agent-dealer 1.1.4 → 1.1.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/bundle/server/dist/coordinator/attempt-waste.js +330 -0
  2. package/bundle/server/dist/coordinator/attempt-waste.test.js +221 -0
  3. package/bundle/server/dist/coordinator/checkpoint.js +55 -0
  4. package/bundle/server/dist/coordinator/checkpoint.test.js +169 -0
  5. package/bundle/server/dist/coordinator/developer-checkpoint.test.js +399 -0
  6. package/bundle/server/dist/coordinator/developer-effect.js +233 -6
  7. package/bundle/server/dist/coordinator/execution-report.js +650 -0
  8. package/bundle/server/dist/coordinator/execution-report.test.js +327 -0
  9. package/bundle/server/dist/coordinator/reviewer-effect.js +1 -1
  10. package/bundle/server/dist/coordinator/session-activity-sampler.test.js +168 -0
  11. package/bundle/server/dist/coordinator/session-activity.js +301 -0
  12. package/bundle/server/dist/coordinator/session-activity.test.js +184 -0
  13. package/bundle/server/dist/coordinator/session-progress.js +246 -3
  14. package/bundle/server/dist/coordinator/session-progress.test.js +78 -1
  15. package/bundle/server/dist/coordinator/session-silence.js +204 -0
  16. package/bundle/server/dist/coordinator/session-silence.test.js +165 -0
  17. package/bundle/server/dist/db/index.js +19 -0
  18. package/bundle/server/dist/db/schema.sql +60 -0
  19. package/bundle/server/dist/db/schema.test.js +28 -0
  20. package/bundle/server/dist/index.js +4 -0
  21. package/bundle/server/dist/read-models/execution-analysis.js +1134 -0
  22. package/bundle/server/dist/read-models/execution-analysis.test.js +335 -0
  23. package/bundle/server/dist/repository/session-activity.js +175 -0
  24. package/bundle/server/dist/repository/session-activity.test.js +281 -0
  25. package/bundle/server/dist/repository/worker-sessions.js +16 -0
  26. package/bundle/server/dist/routes/execution-analysis.js +34 -0
  27. package/bundle/server/dist/routes/execution-analysis.test.js +700 -0
  28. package/bundle/server/dist/routes/execution-report.js +30 -0
  29. package/bundle/server/dist/routes/issues.js +10 -0
  30. package/bundle/server/package.json +2 -2
  31. package/bundle/server/static-ui/assets/index-D9CPv8TJ.css +1 -0
  32. package/bundle/server/static-ui/assets/index-DnsbK_sC.js +60 -0
  33. package/bundle/server/static-ui/index.html +2 -2
  34. package/bundle/shared/dist/attempt-waste.d.ts +59 -0
  35. package/bundle/shared/dist/attempt-waste.js +41 -0
  36. package/bundle/shared/dist/attempt-waste.test.d.ts +1 -0
  37. package/bundle/shared/dist/attempt-waste.test.js +47 -0
  38. package/bundle/shared/dist/execution-analysis.d.ts +3141 -0
  39. package/bundle/shared/dist/execution-analysis.js +298 -0
  40. package/bundle/shared/dist/execution-report.d.ts +2807 -0
  41. package/bundle/shared/dist/execution-report.js +284 -0
  42. package/bundle/shared/dist/execution-report.test.d.ts +1 -0
  43. package/bundle/shared/dist/execution-report.test.js +121 -0
  44. package/bundle/shared/dist/findings.d.ts +2 -2
  45. package/bundle/shared/dist/index.d.ts +53 -50
  46. package/bundle/shared/dist/index.js +3 -0
  47. package/bundle/shared/dist/issues.d.ts +14 -14
  48. package/bundle/shared/dist/queue-entries.d.ts +2 -2
  49. package/bundle/shared/dist/usage-events.d.ts +6 -6
  50. package/bundle/shared/dist/workflow.d.ts +4 -4
  51. package/bundle/shared/dist/workflow.js +8 -0
  52. package/bundle/shared/dist/workflow.test.js +2 -0
  53. package/bundle/shared/package.json +1 -1
  54. package/dist/index.js +3 -0
  55. package/dist/start.js +4 -0
  56. package/dist/status.js +1 -1
  57. package/dist/version-output.test.d.ts +1 -0
  58. package/dist/version-output.test.js +185 -0
  59. package/package.json +1 -1
  60. package/bundle/server/static-ui/assets/index-Bva8TLlT.js +0 -60
  61. package/bundle/server/static-ui/assets/index-DDI4iqx0.css +0 -1
@@ -0,0 +1,330 @@
1
+ import { CheckpointObservedPayload, RetryReusedPayload } from "@agent-dealer/shared";
2
+ import { getDb } from "../db/index.js";
3
+ function msOf(ts) {
4
+ if (!ts)
5
+ return null;
6
+ const ms = Date.parse(ts);
7
+ return Number.isFinite(ms) ? ms : null;
8
+ }
9
+ function withPartialSample(base) {
10
+ const reasons = [...(base.reasons ?? [])];
11
+ if (base.known < base.total && !reasons.includes("partial_sample"))
12
+ reasons.push("partial_sample");
13
+ return { ...base, reasons };
14
+ }
15
+ function orderCheckpoints(checkpoints) {
16
+ return [...checkpoints].sort((a, b) => {
17
+ const ta = msOf(a.observedAt ?? a.ts) ?? 0;
18
+ const tb = msOf(b.observedAt ?? b.ts) ?? 0;
19
+ return ta - tb || a.cursor - b.cursor;
20
+ });
21
+ }
22
+ export function deriveFirstCheckpoint(input) {
23
+ const ordered = orderCheckpoints(input.checkpoints);
24
+ const startMs = msOf(input.workflowStartedAt);
25
+ const first = ordered[0] ?? null;
26
+ if (first) {
27
+ const atMs = msOf(first.observedAt ?? first.ts);
28
+ if (startMs === null || atMs === null) {
29
+ return { kind: first.kind, observedSha: first.observedSha, msSinceWorkflowStart: null, quality: "unavailable", reasons: ["no_defensible_boundary"] };
30
+ }
31
+ if (atMs < startMs) {
32
+ return { kind: first.kind, observedSha: first.observedSha, msSinceWorkflowStart: null, quality: "unavailable", reasons: ["negative_duration"] };
33
+ }
34
+ return { kind: first.kind, observedSha: first.observedSha, msSinceWorkflowStart: atMs - startMs, quality: "exact", reasons: [] };
35
+ }
36
+ const legacyMs = msOf(input.legacyFirstEvidenceAt);
37
+ if (startMs !== null && legacyMs !== null) {
38
+ if (legacyMs < startMs) {
39
+ return { kind: null, observedSha: null, msSinceWorkflowStart: null, quality: "unavailable", reasons: ["negative_duration"] };
40
+ }
41
+ return { kind: null, observedSha: null, msSinceWorkflowStart: legacyMs - startMs, quality: "inferred", reasons: ["backfill"] };
42
+ }
43
+ return { kind: null, observedSha: null, msSinceWorkflowStart: null, quality: "unavailable", reasons: ["missing_checkpoint"] };
44
+ }
45
+ /**
46
+ * Waste includes agent-process/usage from attempts whose primary result is
47
+ * failed/timed_out/cancelled for a failure. A later successful retry never
48
+ * erases prior waste. `cancelled` counts only with failure evidence
49
+ * (errorJson) — a clean operator cancel discarded nothing.
50
+ */
51
+ export function isWastedSession(session) {
52
+ if (session.status === "failed" || session.status === "timed_out")
53
+ return true;
54
+ if (session.status === "cancelled")
55
+ return session.errorJson != null;
56
+ return false;
57
+ }
58
+ function unavailableAggregate(total, reason) {
59
+ return { value: null, quality: "unavailable", reasons: total > 0 ? [reason] : [reason], known: 0, total };
60
+ }
61
+ export function deriveAttemptWaste(input) {
62
+ const publishOnly = input.publishOnlySessionIds ?? new Set();
63
+ const usageBySession = new Map(input.usages.map((u) => [u.workerSessionId, u]));
64
+ const boundsBySession = new Map(input.agentBoundaries.map((b) => [b.sessionId, b]));
65
+ const failed = input.sessions.filter(isWastedSession);
66
+ const publishOnlyFailed = failed.filter((s) => publishOnly.has(s.id));
67
+ const agentFailed = failed.filter((s) => !publishOnly.has(s.id));
68
+ // Runtime: exact agent interval when both boundaries exist, else the inferred
69
+ // spawn envelope (usage duration includes slot wait + post-exit work, §6.1).
70
+ const runtimes = [];
71
+ for (const session of agentFailed) {
72
+ const bounds = boundsBySession.get(session.id);
73
+ if (bounds?.startMs != null && bounds?.endMs != null) {
74
+ if (bounds.endMs < bounds.startMs)
75
+ continue;
76
+ runtimes.push({ value: bounds.endMs - bounds.startMs, quality: "exact", reasons: [] });
77
+ continue;
78
+ }
79
+ const duration = usageBySession.get(session.id)?.durationMs;
80
+ if (duration != null && Number.isFinite(duration)) {
81
+ runtimes.push({
82
+ value: duration,
83
+ quality: "inferred",
84
+ reasons: ["proxy_boundary", "includes_spawn_slot_wait", "includes_post_exit_work"],
85
+ });
86
+ }
87
+ }
88
+ const runtimeMs = runtimes.length === 0
89
+ ? unavailableAggregate(agentFailed.length, "no_defensible_boundary")
90
+ : withPartialSample({
91
+ value: runtimes.reduce((sum, r) => sum + r.value, 0),
92
+ quality: runtimes.some((r) => r.quality === "inferred") ? "inferred" : "exact",
93
+ reasons: [...new Set(runtimes.flatMap((r) => r.reasons))],
94
+ known: runtimes.length,
95
+ total: agentFailed.length,
96
+ });
97
+ // Value metrics: provider-recorded numbers only, never coerced from null.
98
+ const sumKnown = (pick) => {
99
+ const values = [];
100
+ for (const session of agentFailed) {
101
+ const v = usageBySession.get(session.id) ? pick(usageBySession.get(session.id)) : null;
102
+ if (v != null && Number.isFinite(v))
103
+ values.push(v);
104
+ }
105
+ if (values.length === 0)
106
+ return unavailableAggregate(agentFailed.length, "missing_provider_metadata");
107
+ return withPartialSample({
108
+ value: values.reduce((sum, v) => sum + v, 0),
109
+ quality: "exact",
110
+ known: values.length,
111
+ total: agentFailed.length,
112
+ });
113
+ };
114
+ return {
115
+ failedAttempts: agentFailed.length,
116
+ publishOnlyAttempts: publishOnlyFailed.length,
117
+ runtimeMs,
118
+ tokensIn: sumKnown((u) => u.tokensIn),
119
+ tokensOut: sumKnown((u) => u.tokensOut),
120
+ costUsd: sumKnown((u) => u.costUsd),
121
+ };
122
+ }
123
+ export function deriveRetrySummary(input) {
124
+ const reuseBySession = new Map(input.reuse.map((r) => [r.sessionId, r]));
125
+ const hints = input.hints ?? new Map();
126
+ const detail = [];
127
+ for (const session of input.sessions.slice(1)) {
128
+ const record = reuseBySession.get(session.id);
129
+ if (record) {
130
+ detail.push({
131
+ sessionId: session.id,
132
+ kinds: record.kinds,
133
+ cold: record.kinds.length === 0,
134
+ quality: "exact",
135
+ reasons: [],
136
+ });
137
+ continue;
138
+ }
139
+ // Legacy: consume the recovery/routing payload only when defensible.
140
+ const hint = hints.get(session.id);
141
+ if (hint?.publishOnly) {
142
+ detail.push({
143
+ sessionId: session.id,
144
+ kinds: ["publish_only"],
145
+ cold: false,
146
+ quality: "inferred",
147
+ reasons: ["backfill"],
148
+ });
149
+ continue;
150
+ }
151
+ detail.push({
152
+ sessionId: session.id,
153
+ kinds: [],
154
+ cold: null,
155
+ quality: hint?.retryReason ? "inferred" : "unavailable",
156
+ reasons: hint?.retryReason ? ["backfill"] : ["missing_reuse_evidence"],
157
+ });
158
+ }
159
+ return {
160
+ attempts: input.sessions.length,
161
+ retries: detail.length,
162
+ cold: detail.filter((d) => d.cold === true).length,
163
+ reused: detail.filter((d) => d.cold === false).length,
164
+ unknown: detail.filter((d) => d.cold === null).length,
165
+ publishOnly: detail.filter((d) => d.kinds.includes("publish_only")).length,
166
+ attemptsDetail: detail,
167
+ };
168
+ }
169
+ function parseCheckpointRow(row) {
170
+ if (!row.payload_json)
171
+ return null;
172
+ try {
173
+ const parsed = CheckpointObservedPayload.safeParse(JSON.parse(row.payload_json));
174
+ if (!parsed.success)
175
+ return null;
176
+ return {
177
+ sessionId: row.worker_session_id,
178
+ kind: parsed.data.kind,
179
+ observedSha: parsed.data.observedSha,
180
+ observedAt: parsed.data.observedAt,
181
+ origin: parsed.data.origin,
182
+ inputSha: parsed.data.inputSha,
183
+ samplingPrecisionMs: parsed.data.samplingPrecisionMs,
184
+ branch: parsed.data.branch,
185
+ ts: row.ts,
186
+ cursor: row.cursor,
187
+ };
188
+ }
189
+ catch {
190
+ return null;
191
+ }
192
+ }
193
+ /** All checkpoint evidence for an issue, in durable (ts, cursor) order. */
194
+ export function listCheckpointsForIssue(issueId) {
195
+ const rows = getDb()
196
+ .prepare(`SELECT rowid AS cursor, ts, worker_session_id, payload_json FROM workflow_events
197
+ WHERE issue_id = ? AND type = 'checkpoint.observed'
198
+ ORDER BY ts ASC, rowid ASC`)
199
+ .all(issueId);
200
+ return rows
201
+ .map(parseCheckpointRow)
202
+ .filter((r) => r !== null);
203
+ }
204
+ /** The reuse record for one retry session, or null when none was recorded. */
205
+ export function getRetryReuseForSession(workerSessionId) {
206
+ const row = getDb()
207
+ .prepare(`SELECT rowid AS cursor, ts, worker_session_id, payload_json FROM workflow_events
208
+ WHERE worker_session_id = ? AND type = 'retry.reused'
209
+ ORDER BY rowid ASC LIMIT 1`)
210
+ .get(workerSessionId);
211
+ if (!row?.payload_json || !row.worker_session_id)
212
+ return null;
213
+ try {
214
+ const parsed = RetryReusedPayload.safeParse(JSON.parse(row.payload_json));
215
+ if (!parsed.success)
216
+ return null;
217
+ return {
218
+ sessionId: row.worker_session_id,
219
+ kinds: parsed.data.kinds,
220
+ retryReason: parsed.data.retryReason,
221
+ ts: row.ts,
222
+ cursor: row.cursor,
223
+ };
224
+ }
225
+ catch {
226
+ return null;
227
+ }
228
+ }
229
+ /** All reuse records for an issue, in durable (ts, cursor) order. */
230
+ export function listReuseForIssue(issueId) {
231
+ const rows = getDb()
232
+ .prepare(`SELECT rowid AS cursor, ts, worker_session_id, payload_json FROM workflow_events
233
+ WHERE issue_id = ? AND type = 'retry.reused'
234
+ ORDER BY ts ASC, rowid ASC`)
235
+ .all(issueId);
236
+ const out = [];
237
+ for (const row of rows) {
238
+ if (!row.payload_json || !row.worker_session_id)
239
+ continue;
240
+ try {
241
+ const parsed = RetryReusedPayload.safeParse(JSON.parse(row.payload_json));
242
+ if (!parsed.success)
243
+ continue;
244
+ out.push({
245
+ sessionId: row.worker_session_id,
246
+ kinds: parsed.data.kinds,
247
+ retryReason: parsed.data.retryReason,
248
+ ts: row.ts,
249
+ cursor: row.cursor,
250
+ });
251
+ }
252
+ catch {
253
+ // malformed evidence never drives metrics
254
+ }
255
+ }
256
+ return out;
257
+ }
258
+ /** Exact agent-process boundaries per session from durable agent.* events. */
259
+ export function listAgentBoundariesForIssue(issueId) {
260
+ const rows = getDb()
261
+ .prepare(`SELECT worker_session_id AS sessionId, type, ts FROM workflow_events
262
+ WHERE issue_id = ? AND type IN ('agent.started', 'agent.completed')
263
+ ORDER BY ts ASC, rowid ASC`)
264
+ .all(issueId);
265
+ const starts = new Map();
266
+ const ends = new Map();
267
+ for (const row of rows) {
268
+ if (!row.sessionId)
269
+ continue;
270
+ const ms = msOf(row.ts);
271
+ if (ms === null)
272
+ continue;
273
+ if (row.type === "agent.started" && !starts.has(row.sessionId))
274
+ starts.set(row.sessionId, ms);
275
+ if (row.type === "agent.completed" && !ends.has(row.sessionId))
276
+ ends.set(row.sessionId, ms);
277
+ }
278
+ const ids = new Set([...starts.keys(), ...ends.keys()]);
279
+ return [...ids].map((sessionId) => ({
280
+ sessionId,
281
+ startMs: starts.get(sessionId) ?? null,
282
+ endMs: ends.get(sessionId) ?? null,
283
+ }));
284
+ }
285
+ /** Sessions that ran the no-agent publish path (worker terminal carries publishOnly). */
286
+ export function listPublishOnlySessionsForIssue(issueId) {
287
+ const rows = getDb()
288
+ .prepare(`SELECT worker_session_id AS sessionId, payload_json AS payload FROM workflow_events
289
+ WHERE issue_id = ? AND type IN ('worker.completed', 'worker.failed')`)
290
+ .all(issueId);
291
+ const out = new Set();
292
+ for (const row of rows) {
293
+ if (!row.sessionId || !row.payload)
294
+ continue;
295
+ try {
296
+ const payload = JSON.parse(row.payload);
297
+ if (payload.publishOnly === true)
298
+ out.add(row.sessionId);
299
+ }
300
+ catch {
301
+ // ignore malformed payloads
302
+ }
303
+ }
304
+ return out;
305
+ }
306
+ /** Per-session work-item payload hints for legacy retry derivation. */
307
+ export function sessionHintsForIssue(issueId) {
308
+ const rows = getDb()
309
+ .prepare("SELECT worker_session_id AS sessionId, payload_json AS payload FROM work_items WHERE issue_id = ?")
310
+ .all(issueId);
311
+ const out = new Map();
312
+ for (const row of rows) {
313
+ if (!row.sessionId)
314
+ continue;
315
+ let publishOnly = false;
316
+ let retryReason = null;
317
+ if (row.payload) {
318
+ try {
319
+ const payload = JSON.parse(row.payload);
320
+ publishOnly = payload.publishOnly === true;
321
+ retryReason = typeof payload.retryReason === "string" ? payload.retryReason : null;
322
+ }
323
+ catch {
324
+ // keep defaults
325
+ }
326
+ }
327
+ out.set(row.sessionId, { publishOnly, retryReason });
328
+ }
329
+ return out;
330
+ }
@@ -0,0 +1,221 @@
1
+ // packages/server/src/coordinator/attempt-waste.test.ts
2
+ //
3
+ // NOT-172: failed-attempt waste, first checkpoint, and retry-reuse derivation.
4
+ //
5
+ // Pure derivation over durable evidence shapes (no DB): exact agent intervals
6
+ // beat the inferred spawn envelope, null provider fields stay null with
7
+ // known/total counts, publish-only attempts add zero agent waste but remain
8
+ // retries, and legacy payloads derive only as inferred.
9
+ import { test } from "node:test";
10
+ import assert from "node:assert/strict";
11
+ import { deriveAttemptWaste, deriveFirstCheckpoint, deriveRetrySummary, isWastedSession, } from "./attempt-waste.js";
12
+ const T0 = "2026-09-21T00:00:00.000Z";
13
+ const T1 = "2026-09-21T00:05:00.000Z";
14
+ const T2 = "2026-09-21T00:10:00.000Z";
15
+ function session(overrides) {
16
+ return {
17
+ role: "developer",
18
+ status: "failed",
19
+ errorJson: null,
20
+ createdAt: T0,
21
+ ...overrides,
22
+ };
23
+ }
24
+ function checkpoint(overrides = {}) {
25
+ return {
26
+ sessionId: "session-1",
27
+ kind: "commit",
28
+ observedSha: "a".repeat(40),
29
+ observedAt: T1,
30
+ origin: "sampler",
31
+ inputSha: "b".repeat(40),
32
+ samplingPrecisionMs: 10_000,
33
+ branch: "issue-1",
34
+ ts: T1,
35
+ cursor: 7,
36
+ ...overrides,
37
+ };
38
+ }
39
+ // ---------------------------------------------------------- first checkpoint
40
+ test("first checkpoint reports ms since workflow start as exact", () => {
41
+ const first = deriveFirstCheckpoint({ workflowStartedAt: T0, checkpoints: [checkpoint()] });
42
+ assert.equal(first.kind, "commit");
43
+ assert.equal(first.msSinceWorkflowStart, 5 * 60_000);
44
+ assert.equal(first.quality, "exact");
45
+ assert.deepEqual(first.reasons, []);
46
+ });
47
+ test("earliest of several checkpoints wins by (ts, cursor) order", () => {
48
+ const first = deriveFirstCheckpoint({
49
+ workflowStartedAt: T0,
50
+ checkpoints: [
51
+ checkpoint({ observedAt: T2, ts: T2, cursor: 9, kind: "branch_pushed" }),
52
+ checkpoint({ observedAt: T1, ts: T1, cursor: 7, kind: "commit" }),
53
+ ],
54
+ });
55
+ assert.equal(first.kind, "commit");
56
+ assert.equal(first.msSinceWorkflowStart, 5 * 60_000);
57
+ });
58
+ test("no checkpoints and no legacy evidence is unavailable, never zero", () => {
59
+ const first = deriveFirstCheckpoint({ workflowStartedAt: T0, checkpoints: [] });
60
+ assert.equal(first.msSinceWorkflowStart, null);
61
+ assert.equal(first.quality, "unavailable");
62
+ assert.ok(first.reasons.includes("missing_checkpoint"));
63
+ });
64
+ test("legacy evidence derives an inferred first checkpoint, never exact", () => {
65
+ const first = deriveFirstCheckpoint({ workflowStartedAt: T0, checkpoints: [], legacyFirstEvidenceAt: T2 });
66
+ assert.equal(first.msSinceWorkflowStart, 10 * 60_000);
67
+ assert.equal(first.quality, "inferred");
68
+ assert.ok(first.reasons.includes("backfill"));
69
+ });
70
+ test("a checkpoint before workflow start is a negative duration, not clamped", () => {
71
+ const first = deriveFirstCheckpoint({ workflowStartedAt: T2, checkpoints: [checkpoint()] });
72
+ assert.equal(first.msSinceWorkflowStart, null);
73
+ assert.equal(first.quality, "unavailable");
74
+ assert.ok(first.reasons.includes("negative_duration"));
75
+ });
76
+ // --------------------------------------------------------------- waste scope
77
+ test("failed and timed_out count as waste; done never does", () => {
78
+ assert.equal(isWastedSession(session({ id: "a", status: "failed" })), true);
79
+ assert.equal(isWastedSession(session({ id: "b", status: "timed_out" })), true);
80
+ assert.equal(isWastedSession(session({ id: "c", status: "done" })), false);
81
+ assert.equal(isWastedSession(session({ id: "d", status: "queued" })), false);
82
+ assert.equal(isWastedSession(session({ id: "e", status: "running" })), false);
83
+ });
84
+ test("cancelled counts only with failure evidence", () => {
85
+ assert.equal(isWastedSession(session({ id: "a", status: "cancelled", errorJson: JSON.stringify({ reason: "boom" }) })), true);
86
+ assert.equal(isWastedSession(session({ id: "b", status: "cancelled", errorJson: null })), false);
87
+ });
88
+ // ------------------------------------------------------------------ waste sums
89
+ function usage(overrides) {
90
+ return { tokensIn: null, tokensOut: null, costUsd: null, durationMs: null, ...overrides };
91
+ }
92
+ test("a failed session followed by success reports the first attempt as waste", () => {
93
+ const waste = deriveAttemptWaste({
94
+ sessions: [
95
+ session({ id: "s1", status: "failed" }),
96
+ session({ id: "s2", status: "done" }),
97
+ ],
98
+ usages: [
99
+ usage({ workerSessionId: "s1", tokensIn: 100, tokensOut: 50, costUsd: 0.42, durationMs: 60_000 }),
100
+ usage({ workerSessionId: "s2", tokensIn: 200, tokensOut: 60, costUsd: 0.5, durationMs: 70_000 }),
101
+ ],
102
+ agentBoundaries: [],
103
+ });
104
+ assert.equal(waste.failedAttempts, 1);
105
+ // No agent boundaries: the spawn envelope stands in, marked inferred.
106
+ assert.equal(waste.runtimeMs.value, 60_000);
107
+ assert.equal(waste.runtimeMs.quality, "inferred");
108
+ assert.ok(waste.runtimeMs.reasons.includes("includes_spawn_slot_wait"));
109
+ assert.ok(waste.runtimeMs.reasons.includes("includes_post_exit_work"));
110
+ assert.equal(waste.tokensIn.value, 100);
111
+ assert.equal(waste.tokensIn.quality, "exact");
112
+ assert.equal(waste.costUsd.value, 0.42);
113
+ });
114
+ test("exact agent intervals beat the usage envelope when both exist", () => {
115
+ const start = Date.parse(T0);
116
+ const bounds = [{ sessionId: "s1", startMs: start, endMs: start + 45_000 }];
117
+ const waste = deriveAttemptWaste({
118
+ sessions: [session({ id: "s1", status: "timed_out" })],
119
+ usages: [usage({ workerSessionId: "s1", durationMs: 60_000 })],
120
+ agentBoundaries: bounds,
121
+ });
122
+ assert.equal(waste.runtimeMs.value, 45_000);
123
+ assert.equal(waste.runtimeMs.quality, "exact");
124
+ assert.deepEqual(waste.runtimeMs.reasons, []);
125
+ });
126
+ test("missing provider fields stay unavailable with coverage counts, never zero", () => {
127
+ const waste = deriveAttemptWaste({
128
+ sessions: [session({ id: "s1" }), session({ id: "s2", status: "timed_out" })],
129
+ usages: [
130
+ // Cursor-style row: tokens known, cost never reported.
131
+ usage({ workerSessionId: "s1", tokensIn: 100, tokensOut: 50, costUsd: null, durationMs: 60_000 }),
132
+ usage({ workerSessionId: "s2", tokensIn: null, tokensOut: null, costUsd: null, durationMs: null }),
133
+ ],
134
+ agentBoundaries: [],
135
+ });
136
+ assert.equal(waste.tokensIn.value, 100);
137
+ assert.equal(waste.tokensIn.known, 1);
138
+ assert.equal(waste.tokensIn.total, 2);
139
+ assert.ok(waste.tokensIn.reasons.includes("partial_sample"));
140
+ assert.equal(waste.costUsd.value, null);
141
+ assert.equal(waste.costUsd.quality, "unavailable");
142
+ assert.equal(waste.costUsd.known, 0);
143
+ assert.equal(waste.costUsd.total, 2);
144
+ assert.ok(waste.costUsd.reasons.includes("missing_provider_metadata"));
145
+ // Runtime known for one of two — partial, still summed over known only.
146
+ assert.equal(waste.runtimeMs.value, 60_000);
147
+ assert.equal(waste.runtimeMs.known, 1);
148
+ assert.equal(waste.runtimeMs.total, 2);
149
+ });
150
+ test("no failed attempts is unavailable across the board, never zero waste presented as exact", () => {
151
+ const waste = deriveAttemptWaste({
152
+ sessions: [session({ id: "s1", status: "done" })],
153
+ usages: [usage({ workerSessionId: "s1", tokensIn: 1, costUsd: 0.01, durationMs: 5 })],
154
+ agentBoundaries: [],
155
+ });
156
+ assert.equal(waste.failedAttempts, 0);
157
+ for (const agg of [waste.runtimeMs, waste.tokensIn, waste.tokensOut, waste.costUsd]) {
158
+ assert.equal(agg.value, null);
159
+ assert.equal(agg.quality, "unavailable");
160
+ }
161
+ });
162
+ test("coordinator-only publish retries add zero agent waste but remain attempts", () => {
163
+ const waste = deriveAttemptWaste({
164
+ sessions: [session({ id: "s1" }), session({ id: "s2", status: "done" })],
165
+ usages: [usage({ workerSessionId: "s1", tokensIn: 999, tokensOut: 999, costUsd: 9.99, durationMs: 999_999 })],
166
+ agentBoundaries: [{ sessionId: "s1", startMs: 1, endMs: 999_999 }],
167
+ publishOnlySessionIds: new Set(["s1"]),
168
+ });
169
+ assert.equal(waste.failedAttempts, 0, "publish-only is not an agent failure");
170
+ assert.equal(waste.publishOnlyAttempts, 1);
171
+ // Its rows/bounds are structurally excluded — not counted as missing either.
172
+ assert.equal(waste.tokensIn.total, 0);
173
+ assert.equal(waste.runtimeMs.total, 0);
174
+ });
175
+ // ------------------------------------------------------------- retry summary
176
+ test("cold retries are distinguishable from reused-work retries", () => {
177
+ const summary = deriveRetrySummary({
178
+ sessions: [session({ id: "s1", status: "failed" }), session({ id: "s2" }), session({ id: "s3" })],
179
+ reuse: [
180
+ { sessionId: "s2", kinds: ["worktree", "commit"], retryReason: "retry", ts: T1, cursor: 3 },
181
+ { sessionId: "s3", kinds: [], retryReason: "retry", ts: T2, cursor: 5 },
182
+ ],
183
+ });
184
+ assert.equal(summary.attempts, 2 + 1);
185
+ assert.equal(summary.retries, 2);
186
+ assert.equal(summary.reused, 1);
187
+ assert.equal(summary.cold, 1);
188
+ assert.equal(summary.unknown, 0);
189
+ assert.deepEqual(summary.attemptsDetail[0].kinds, ["worktree", "commit"]);
190
+ assert.equal(summary.attemptsDetail[0].cold, false);
191
+ assert.equal(summary.attemptsDetail[1].cold, true);
192
+ });
193
+ test("a retry with no reuse record is unknown, not cold", () => {
194
+ const summary = deriveRetrySummary({
195
+ sessions: [session({ id: "s1" }), session({ id: "s2" })],
196
+ reuse: [],
197
+ });
198
+ assert.equal(summary.retries, 1);
199
+ assert.equal(summary.cold, 0);
200
+ assert.equal(summary.unknown, 1);
201
+ assert.equal(summary.attemptsDetail[0].cold, null);
202
+ assert.equal(summary.attemptsDetail[0].quality, "unavailable");
203
+ });
204
+ test("legacy payloads derive publish_only as inferred, retryReason as unknown", () => {
205
+ const summary = deriveRetrySummary({
206
+ sessions: [session({ id: "s1" }), session({ id: "s2" }), session({ id: "s3" })],
207
+ reuse: [],
208
+ hints: new Map([
209
+ ["s2", { publishOnly: true, retryReason: "presumed dead; republishing 1 commit" }],
210
+ ["s3", { publishOnly: false, retryReason: "Developer session produced no PR." }],
211
+ ]),
212
+ });
213
+ assert.equal(summary.reused, 1);
214
+ assert.equal(summary.publishOnly, 1);
215
+ assert.equal(summary.unknown, 1);
216
+ assert.deepEqual(summary.attemptsDetail[0].kinds, ["publish_only"]);
217
+ assert.equal(summary.attemptsDetail[0].quality, "inferred");
218
+ assert.ok(summary.attemptsDetail[0].reasons.includes("backfill"));
219
+ assert.equal(summary.attemptsDetail[1].cold, null);
220
+ assert.equal(summary.attemptsDetail[1].quality, "inferred");
221
+ });
@@ -0,0 +1,55 @@
1
+ import { appendWorkflowEvent } from "../repository/workflow-events.js";
2
+ import { workerSessionPayload } from "./session-progress.js";
3
+ /**
4
+ * Record one checkpoint kind for a session — at most once per (session, kind),
5
+ * except the salvage-origin commit, which is keyed independently so an agent
6
+ * that commits and then crashes/times out dirty still leaves durable salvage
7
+ * evidence alongside the earlier observation. First-commit derivation still
8
+ * counts once: it orders all commit rows and takes the earliest.
9
+ */
10
+ export function emitCheckpointObserved(ev) {
11
+ appendWorkflowEvent({
12
+ issueId: ev.issueId,
13
+ workflowInstanceId: ev.workflowInstanceId,
14
+ workerSessionId: ev.workerSessionId,
15
+ type: "checkpoint.observed",
16
+ actorType: ev.role,
17
+ stage: ev.stage,
18
+ round: ev.round,
19
+ payload: {
20
+ ...workerSessionPayload({ runtime: null, model: null, sessionId: ev.workerSessionId }),
21
+ kind: ev.kind,
22
+ observedSha: ev.observedSha ?? null,
23
+ observedAt: ev.observedAt ?? new Date().toISOString(),
24
+ origin: ev.kind === "commit" ? (ev.origin ?? null) : null,
25
+ inputSha: ev.inputSha ?? null,
26
+ samplingPrecisionMs: ev.samplingPrecisionMs ?? null,
27
+ branch: ev.branch ?? null,
28
+ },
29
+ idempotencyKey: ev.kind === "commit" && ev.origin === "salvage"
30
+ ? `checkpoint:${ev.workerSessionId}:commit:salvage`
31
+ : `checkpoint:${ev.workerSessionId}:${ev.kind}`,
32
+ });
33
+ }
34
+ /**
35
+ * Record which prior work a retry/recovery actually reused — exactly once per
36
+ * session. A cold retry is recorded with an empty kinds array so it stays
37
+ * distinguishable from "no reuse evidence recorded".
38
+ */
39
+ export function emitRetryReuse(ev) {
40
+ appendWorkflowEvent({
41
+ issueId: ev.issueId,
42
+ workflowInstanceId: ev.workflowInstanceId,
43
+ workerSessionId: ev.workerSessionId,
44
+ type: "retry.reused",
45
+ actorType: ev.role,
46
+ stage: ev.stage,
47
+ round: ev.round,
48
+ payload: {
49
+ ...workerSessionPayload({ runtime: null, model: null, sessionId: ev.workerSessionId }),
50
+ kinds: ev.kinds,
51
+ retryReason: ev.retryReason ?? null,
52
+ },
53
+ idempotencyKey: `retry-reused:${ev.workerSessionId}`,
54
+ });
55
+ }