agent-dealer 1.2.7 → 1.2.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bundle/server/dist/adapters/agent-deck-bind.js +16 -6
- package/bundle/server/dist/adapters/agent-deck-bind.test.js +74 -0
- package/bundle/server/dist/adapters/agent-health.js +14 -3
- package/bundle/server/dist/adapters/github.js +4 -1
- package/bundle/server/dist/adapters/muse-capability.js +443 -24
- package/bundle/server/dist/adapters/muse-capability.test.js +469 -25
- package/bundle/server/dist/adapters/muse-visual-qa.js +114 -0
- package/bundle/server/dist/adapters/muse-visual-qa.test.js +68 -0
- package/bundle/server/dist/capacity/muse-probe.js +56 -9
- package/bundle/server/dist/capacity/muse-probe.test.js +188 -1
- package/bundle/server/dist/coordinator/admission.js +14 -1
- package/bundle/server/dist/coordinator/admission.test.js +197 -4
- package/bundle/server/dist/coordinator/auto-merge.integration.test.js +276 -0
- package/bundle/server/dist/coordinator/auto-merge.js +39 -3
- package/bundle/server/dist/coordinator/commands.js +90 -5
- package/bundle/server/dist/coordinator/deck-outage.integration.test.js +4 -2
- package/bundle/server/dist/coordinator/developer-effect.js +79 -1
- package/bundle/server/dist/coordinator/execution-report.js +6 -0
- package/bundle/server/dist/coordinator/execution-report.test.js +7 -0
- package/bundle/server/dist/coordinator/failure-cause.js +64 -1
- package/bundle/server/dist/coordinator/failure-cause.test.js +92 -1
- package/bundle/server/dist/coordinator/failure-reason.js +7 -0
- package/bundle/server/dist/coordinator/failure-reason.test.js +14 -0
- package/bundle/server/dist/coordinator/human-resolution.js +17 -1
- package/bundle/server/dist/coordinator/merge-conflict-sync.js +647 -0
- package/bundle/server/dist/coordinator/merge-conflict-sync.test.js +121 -0
- package/bundle/server/dist/coordinator/muse-developer.integration.test.js +9 -4
- package/bundle/server/dist/coordinator/muse-spawn.js +122 -29
- package/bundle/server/dist/coordinator/muse-spawn.test.js +210 -0
- package/bundle/server/dist/coordinator/playbook-feedback.js +690 -0
- package/bundle/server/dist/coordinator/playbook-feedback.test.js +702 -0
- package/bundle/server/dist/coordinator/prompts-execution-contract.test.js +107 -0
- package/bundle/server/dist/coordinator/prompts.js +78 -2
- package/bundle/server/dist/coordinator/prompts.test.js +83 -0
- package/bundle/server/dist/coordinator/reflect-trigger.js +33 -169
- package/bundle/server/dist/coordinator/reflect-trigger.test.js +149 -200
- package/bundle/server/dist/coordinator/reviewer-effect.js +10 -1
- package/bundle/server/dist/coordinator/reviewer-result.js +6 -0
- package/bundle/server/dist/coordinator/routing.test.js +14 -0
- package/bundle/server/dist/coordinator/session-timeouts.js +30 -0
- package/bundle/server/dist/coordinator/usage-cap.integration.test.js +1 -1
- package/bundle/server/dist/coordinator/worker-loop.js +18 -4
- package/bundle/server/dist/db/index.js +5 -0
- package/bundle/server/dist/db/schema.sql +3 -0
- package/bundle/server/dist/docs-execution-analysis.test.js +1 -0
- package/bundle/server/dist/repository/artifacts-for-issue.js +3 -3
- package/bundle/server/dist/repository/human-actions.js +14 -0
- package/bundle/server/dist/repository/issues.js +29 -4
- package/bundle/server/dist/repository/worker-sessions.js +51 -2
- package/bundle/server/dist/routes/human-actions.js +10 -8
- package/bundle/server/dist/routes/issues-execution-contract.test.js +321 -0
- package/bundle/server/dist/routes/issues.js +19 -2
- package/bundle/server/dist/runners/muse-code-jsonl.js +106 -11
- package/bundle/server/dist/runners/muse-config-core.js +18 -1
- package/bundle/server/dist/runners/muse-config.test.js +37 -0
- package/bundle/server/dist/runners/muse-serve-session.js +5 -0
- package/bundle/server/dist/runners/spawn-cli.js +116 -0
- package/bundle/server/dist/runners/spawn-cli.test.js +124 -0
- package/bundle/server/package.json +2 -2
- package/bundle/server/static-ui/assets/{index-yLyxRd-7.js → index-B6SVCzMR.js} +14 -14
- package/bundle/server/static-ui/assets/{index-DyAJNyfV.css → index-K_YcYkQU.css} +1 -1
- package/bundle/server/static-ui/index.html +2 -2
- package/bundle/shared/dist/execution-analysis.d.ts +22 -22
- package/bundle/shared/dist/execution-contract.d.ts +117 -0
- package/bundle/shared/dist/execution-contract.js +307 -0
- package/bundle/shared/dist/execution-contract.test.d.ts +2 -0
- package/bundle/shared/dist/execution-contract.test.js +499 -0
- package/bundle/shared/dist/execution-report.d.ts +9 -9
- package/bundle/shared/dist/execution-report.js +2 -0
- package/bundle/shared/dist/failure-cause.d.ts +4 -4
- package/bundle/shared/dist/failure-cause.js +2 -0
- package/bundle/shared/dist/human-actions.d.ts +4 -4
- package/bundle/shared/dist/human-actions.js +9 -0
- package/bundle/shared/dist/index.d.ts +85 -84
- package/bundle/shared/dist/index.js +3 -0
- package/bundle/shared/dist/issues.d.ts +146 -16
- package/bundle/shared/dist/issues.js +8 -0
- package/bundle/shared/dist/issues.test.js +1 -0
- package/bundle/shared/dist/outbound-draft.d.ts +12 -12
- package/bundle/shared/dist/worker-sessions.d.ts +10 -0
- package/bundle/shared/dist/worker-sessions.js +8 -0
- package/bundle/shared/dist/worker-sessions.test.js +1 -0
- package/bundle/shared/dist/workflow.d.ts +4 -4
- package/bundle/shared/dist/workflow.js +10 -0
- package/bundle/shared/dist/workflow.test.js +2 -0
- package/bundle/shared/package.json +1 -1
- package/package.json +1 -1
|
@@ -0,0 +1,702 @@
|
|
|
1
|
+
// packages/server/src/coordinator/playbook-feedback.test.ts
|
|
2
|
+
//
|
|
3
|
+
// NOT-305: actual-use receipts and idempotent signal_only failure reports.
|
|
4
|
+
// Deck's NOT-304 correlation operation is faked via the injectable `deps` seam.
|
|
5
|
+
import { test, before } from "node:test";
|
|
6
|
+
import assert from "node:assert/strict";
|
|
7
|
+
import fs from "node:fs";
|
|
8
|
+
import os from "node:os";
|
|
9
|
+
import path from "node:path";
|
|
10
|
+
process.env.AGENT_DEALER_HOME = fs.mkdtempSync(path.join(os.tmpdir(), "dealer-playbook-feedback-"));
|
|
11
|
+
const { migrate } = await import("../db/index.js");
|
|
12
|
+
const { BUILTIN_AGENT_CURSOR_ID } = await import("@agent-dealer/shared");
|
|
13
|
+
const { createIssue } = await import("../repository/issues.js");
|
|
14
|
+
const { createAgent } = await import("../repository/agents.js");
|
|
15
|
+
const { createWorkerSession, completeSession, getWorkerSession, getOrAssignSessionCorrelationId, } = await import("../repository/worker-sessions.js");
|
|
16
|
+
const { buildProfileSnapshot, serializeProfileSnapshot } = await import("./profile-snapshot.js");
|
|
17
|
+
const { createHumanAction, resolveHumanAction } = await import("../repository/human-actions.js");
|
|
18
|
+
const { createIssueArtifact } = await import("../repository/artifacts.js");
|
|
19
|
+
const { reconcileFinding } = await import("../repository/findings.js");
|
|
20
|
+
const { listArtifactsForIssueByKind } = await import("../repository/artifacts-for-issue.js");
|
|
21
|
+
const { collectPlaybookUseReceipt, collectPlaybookUseReceiptsForIssue, reportDeckFailureSignals, parseCorrelatedFetches, validateSignalOnlyArgs, buildSignalOnlyArgs, DECK_CORRELATION_TOOL, NO_HUMAN_FEEDBACK_EXCERPT, } = await import("./playbook-feedback.js");
|
|
22
|
+
const { triggerIssueReflect } = await import("./reflect-trigger.js");
|
|
23
|
+
before(() => {
|
|
24
|
+
migrate();
|
|
25
|
+
});
|
|
26
|
+
const DECK = "11111111-1111-4111-a111-111111111111";
|
|
27
|
+
const UUID_RE = /^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$/i;
|
|
28
|
+
function seedIssue(developerAgentId) {
|
|
29
|
+
return createIssue({
|
|
30
|
+
title: "Feedback issue",
|
|
31
|
+
repo: "acme/app",
|
|
32
|
+
developerAgentId,
|
|
33
|
+
reviewerAgentId: BUILTIN_AGENT_CURSOR_ID,
|
|
34
|
+
baseBranch: "main",
|
|
35
|
+
maxReviewRounds: 3,
|
|
36
|
+
maxInfraAttempts: 3,
|
|
37
|
+
source: "manual",
|
|
38
|
+
});
|
|
39
|
+
}
|
|
40
|
+
function seedAgent() {
|
|
41
|
+
// NOT-149 shape: createAgent never stores legacy playbookId(s) — new profiles carry
|
|
42
|
+
// none, and receipts/signals must work from actual use alone.
|
|
43
|
+
const agent = createAgent({ name: `dev-${Math.random()}`, runtime: "claude_code", deckId: DECK });
|
|
44
|
+
assert.equal(agent.playbookId, null);
|
|
45
|
+
assert.equal(agent.playbookIdsJson, null);
|
|
46
|
+
return agent;
|
|
47
|
+
}
|
|
48
|
+
/** Terminal developer session whose frozen snapshot names the deck (no playbook IDs). */
|
|
49
|
+
function seedTerminalSession(issueId, agent, status = "done") {
|
|
50
|
+
const snapshot = { ...buildProfileSnapshot(agent, "developer"), deckId: DECK };
|
|
51
|
+
const session = createWorkerSession({
|
|
52
|
+
issueId,
|
|
53
|
+
role: "developer",
|
|
54
|
+
round: 1,
|
|
55
|
+
agentId: agent.id,
|
|
56
|
+
runtime: agent.runtime,
|
|
57
|
+
model: "claude-opus",
|
|
58
|
+
profileSnapshotJson: serializeProfileSnapshot(snapshot),
|
|
59
|
+
});
|
|
60
|
+
completeSession(session.id, { status });
|
|
61
|
+
return getWorkerSession(session.id);
|
|
62
|
+
}
|
|
63
|
+
/**
|
|
64
|
+
* Independent stand-in for Deck's real `propose_playbook_patch` input schema —
|
|
65
|
+
* built from Deck's actual contract (shared PatchEvidenceContent requires a
|
|
66
|
+
* NON-EMPTY user_feedback_excerpt on every report), deliberately NOT by calling
|
|
67
|
+
* Dealer's validateSignalOnlyArgs. If Dealer's validator ever drifts (e.g. treats
|
|
68
|
+
* the excerpt as optional), this fake still rejects the call exactly as a real
|
|
69
|
+
* Deck would, so the test fails instead of passing against its own assumption.
|
|
70
|
+
*/
|
|
71
|
+
function deckSchemaError(args) {
|
|
72
|
+
const allowedTop = new Set([
|
|
73
|
+
"kind", "rationale", "evidence", "signal_ids", "supersedes", "ops", "playbook_id", "new_playbook",
|
|
74
|
+
]);
|
|
75
|
+
for (const key of Object.keys(args)) {
|
|
76
|
+
if (!allowedTop.has(key))
|
|
77
|
+
return `unknown top-level field: ${key}`;
|
|
78
|
+
}
|
|
79
|
+
if (args.kind !== "signal_only")
|
|
80
|
+
return `kind must be "signal_only"`;
|
|
81
|
+
if (typeof args.rationale !== "string" || !args.rationale.trim()) {
|
|
82
|
+
return "rationale must be a non-empty string";
|
|
83
|
+
}
|
|
84
|
+
const evidence = typeof args.evidence === "object" && args.evidence !== null
|
|
85
|
+
? args.evidence
|
|
86
|
+
: null;
|
|
87
|
+
if (!evidence)
|
|
88
|
+
return "evidence must be an object";
|
|
89
|
+
for (const key of Object.keys(evidence)) {
|
|
90
|
+
if (key !== "failure_summary" && key !== "user_feedback_excerpt" && key !== "corrected_output_hint") {
|
|
91
|
+
return `unknown evidence field: ${key}`;
|
|
92
|
+
}
|
|
93
|
+
}
|
|
94
|
+
if (typeof evidence.failure_summary !== "string" || !evidence.failure_summary.trim()) {
|
|
95
|
+
return "evidence.failure_summary must be a non-empty string";
|
|
96
|
+
}
|
|
97
|
+
// Deck's PatchEvidenceContent: user_feedback_excerpt is z.string().min(1) —
|
|
98
|
+
// REQUIRED on every report, including recurring_blocking and blank reasons.
|
|
99
|
+
if (typeof evidence.user_feedback_excerpt !== "string" || !evidence.user_feedback_excerpt.trim()) {
|
|
100
|
+
return "evidence.user_feedback_excerpt must be a non-empty string";
|
|
101
|
+
}
|
|
102
|
+
if ("corrected_output_hint" in evidence) {
|
|
103
|
+
const hint = evidence.corrected_output_hint;
|
|
104
|
+
if (typeof hint !== "string" || !hint.trim()) {
|
|
105
|
+
return "evidence.corrected_output_hint must be a non-empty string when present";
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
if ("playbook_id" in args || "ops" in args)
|
|
109
|
+
return "signal_only must not carry playbook_id or ops";
|
|
110
|
+
return null;
|
|
111
|
+
}
|
|
112
|
+
function makeDeps(opts = {}) {
|
|
113
|
+
const calls = [];
|
|
114
|
+
const deps = {
|
|
115
|
+
checkHealth: async () => opts.healthy ?? true,
|
|
116
|
+
callTool: (async (call) => {
|
|
117
|
+
calls.push({ deckId: call.deckId, toolName: call.toolName, args: call.arguments });
|
|
118
|
+
if (call.toolName === DECK_CORRELATION_TOOL) {
|
|
119
|
+
if (opts.correlationData !== undefined)
|
|
120
|
+
return { ok: true, data: opts.correlationData };
|
|
121
|
+
return {
|
|
122
|
+
ok: true,
|
|
123
|
+
data: {
|
|
124
|
+
deck_id: DECK,
|
|
125
|
+
correlation_id: call.arguments.correlation_id,
|
|
126
|
+
fetches: opts.fetches ?? [],
|
|
127
|
+
},
|
|
128
|
+
};
|
|
129
|
+
}
|
|
130
|
+
if (call.toolName === "propose_playbook_patch") {
|
|
131
|
+
// Strict stand-in for Deck's schema (additionalProperties:false), checked
|
|
132
|
+
// against Deck's own contract above — reject anything the real Deck would
|
|
133
|
+
// reject, so tests prove schema compliance against an independent oracle.
|
|
134
|
+
const schemaError = deckSchemaError(call.arguments);
|
|
135
|
+
if (schemaError) {
|
|
136
|
+
return { ok: false, kind: "infra_failure", reason: `Deck rejected signal payload: ${schemaError}` };
|
|
137
|
+
}
|
|
138
|
+
return opts.propose
|
|
139
|
+
? opts.propose(call.arguments)
|
|
140
|
+
: { ok: true, data: { id: `sig-${calls.length}` } };
|
|
141
|
+
}
|
|
142
|
+
throw new Error(`unexpected tool call: ${call.toolName}`);
|
|
143
|
+
}),
|
|
144
|
+
};
|
|
145
|
+
return { deps, calls };
|
|
146
|
+
}
|
|
147
|
+
function receiptsFor(issueId) {
|
|
148
|
+
return listArtifactsForIssueByKind(issueId, "playbook_use_receipt").map((a) => JSON.parse(a.contentJson));
|
|
149
|
+
}
|
|
150
|
+
function signalsFor(issueId) {
|
|
151
|
+
return listArtifactsForIssueByKind(issueId, "deck_feedback_signal").map((a) => JSON.parse(a.contentJson));
|
|
152
|
+
}
|
|
153
|
+
function sentSignalsFor(issueId) {
|
|
154
|
+
return signalsFor(issueId).filter((s) => s.status === "sent");
|
|
155
|
+
}
|
|
156
|
+
function pendingSignalsFor(issueId) {
|
|
157
|
+
return signalsFor(issueId).filter((s) => s.status === "pending" || s.status === "sending");
|
|
158
|
+
}
|
|
159
|
+
function evidenceOf(args) {
|
|
160
|
+
return args.evidence;
|
|
161
|
+
}
|
|
162
|
+
test("every new worker session persists a unique opaque correlation ID before spawn", () => {
|
|
163
|
+
const agent = seedAgent();
|
|
164
|
+
const issue = seedIssue(agent.id);
|
|
165
|
+
const snapshot = { ...buildProfileSnapshot(agent, "developer"), deckId: DECK };
|
|
166
|
+
const a = createWorkerSession({
|
|
167
|
+
issueId: issue.id, role: "developer", round: 1, agentId: agent.id,
|
|
168
|
+
runtime: agent.runtime, profileSnapshotJson: serializeProfileSnapshot(snapshot),
|
|
169
|
+
});
|
|
170
|
+
const b = createWorkerSession({
|
|
171
|
+
issueId: issue.id, role: "developer", round: 2, agentId: agent.id,
|
|
172
|
+
runtime: agent.runtime, profileSnapshotJson: serializeProfileSnapshot(snapshot),
|
|
173
|
+
});
|
|
174
|
+
assert.match(a.deckCorrelationId, UUID_RE);
|
|
175
|
+
assert.match(b.deckCorrelationId, UUID_RE);
|
|
176
|
+
assert.notEqual(a.deckCorrelationId, b.deckCorrelationId);
|
|
177
|
+
// Persisted, not just returned: a fresh read carries the same ID.
|
|
178
|
+
assert.equal(getWorkerSession(a.id).deckCorrelationId, a.deckCorrelationId);
|
|
179
|
+
// Pre-feature NULL rows are repaired deterministically on read.
|
|
180
|
+
assert.match(getOrAssignSessionCorrelationId(a.id), UUID_RE);
|
|
181
|
+
assert.equal(getOrAssignSessionCorrelationId(a.id), a.deckCorrelationId);
|
|
182
|
+
});
|
|
183
|
+
test("two fetched playbooks persist one receipt naming exactly those two; an unused card is absent", async () => {
|
|
184
|
+
const agent = seedAgent();
|
|
185
|
+
const issue = seedIssue(agent.id);
|
|
186
|
+
const session = seedTerminalSession(issue.id, agent);
|
|
187
|
+
const { deps, calls } = makeDeps({
|
|
188
|
+
fetches: [
|
|
189
|
+
{ playbook_id: "pb-used-a", first_fetched_at: "2026-09-01T10:00:00.000Z", last_fetched_at: "2026-09-01T10:05:00.000Z" },
|
|
190
|
+
{ playbook_id: "pb-used-b", first_fetched_at: "2026-09-01T10:02:00.000Z", last_fetched_at: "2026-09-01T10:09:00.000Z" },
|
|
191
|
+
],
|
|
192
|
+
});
|
|
193
|
+
const result = await collectPlaybookUseReceipt(session.id, deps);
|
|
194
|
+
assert.equal(result.collected, true);
|
|
195
|
+
// The correlation call carries the session's own opaque ID under the launch deck.
|
|
196
|
+
const correlationCalls = calls.filter((c) => c.toolName === DECK_CORRELATION_TOOL);
|
|
197
|
+
assert.equal(correlationCalls.length, 1);
|
|
198
|
+
assert.equal(correlationCalls[0].deckId, DECK);
|
|
199
|
+
assert.equal(correlationCalls[0].args.correlation_id, session.deckCorrelationId);
|
|
200
|
+
const receipts = receiptsFor(issue.id);
|
|
201
|
+
assert.equal(receipts.length, 1);
|
|
202
|
+
const receipt = receipts[0];
|
|
203
|
+
assert.deepStrictEqual(receipt.playbookIds, ["pb-used-a", "pb-used-b"]);
|
|
204
|
+
assert.ok(!receipt.playbookIds.includes("pb-unused"));
|
|
205
|
+
assert.equal(receipt.workerSessionId, session.id);
|
|
206
|
+
assert.equal(receipt.role, "developer");
|
|
207
|
+
assert.equal(receipt.deckId, DECK);
|
|
208
|
+
assert.equal(receipt.correlationId, session.deckCorrelationId);
|
|
209
|
+
assert.equal(receipt.runtime, "claude_code");
|
|
210
|
+
assert.equal(receipt.model, "claude-opus");
|
|
211
|
+
assert.equal(receipt.firstFetchedAt, "2026-09-01T10:00:00.000Z");
|
|
212
|
+
assert.equal(receipt.lastFetchedAt, "2026-09-01T10:09:00.000Z");
|
|
213
|
+
assert.equal(receipt.status, "collected");
|
|
214
|
+
// Non-goal: no prompts, tool I/O, repo names, or issue descriptions in Deck receipts.
|
|
215
|
+
const allowed = new Set([
|
|
216
|
+
"workerSessionId", "role", "deckId", "correlationId", "runtime", "model",
|
|
217
|
+
"playbookIds", "firstFetchedAt", "lastFetchedAt", "status", "reason", "error",
|
|
218
|
+
]);
|
|
219
|
+
for (const key of Object.keys(receipt))
|
|
220
|
+
assert.ok(allowed.has(key), `receipt leaks ${key}`);
|
|
221
|
+
});
|
|
222
|
+
test("receipt collection is idempotent: a repeat after restart does not re-call Deck", async () => {
|
|
223
|
+
const agent = seedAgent();
|
|
224
|
+
const issue = seedIssue(agent.id);
|
|
225
|
+
const session = seedTerminalSession(issue.id, agent);
|
|
226
|
+
const { deps, calls } = makeDeps({ fetches: [{ playbook_id: "pb-x" }] });
|
|
227
|
+
const first = await collectPlaybookUseReceipt(session.id, deps);
|
|
228
|
+
assert.equal(first.collected, true);
|
|
229
|
+
const second = await collectPlaybookUseReceipt(session.id, deps);
|
|
230
|
+
assert.equal(second.collected, true);
|
|
231
|
+
assert.equal(second.collected && "duplicate" in second && second.duplicate, true);
|
|
232
|
+
assert.equal(calls.filter((c) => c.toolName === DECK_CORRELATION_TOOL).length, 1);
|
|
233
|
+
assert.equal(receiptsFor(issue.id).length, 1);
|
|
234
|
+
});
|
|
235
|
+
test("a clean approved run records receipts only — no feedback signal, no playbook patch", async () => {
|
|
236
|
+
const agent = seedAgent();
|
|
237
|
+
const issue = seedIssue(agent.id);
|
|
238
|
+
seedTerminalSession(issue.id, agent);
|
|
239
|
+
const { deps, calls } = makeDeps({ fetches: [{ playbook_id: "pb-used" }] });
|
|
240
|
+
const result = await triggerIssueReflect(issue.id, deps);
|
|
241
|
+
assert.equal(result, "triggered");
|
|
242
|
+
assert.equal(receiptsFor(issue.id).length, 1);
|
|
243
|
+
assert.deepStrictEqual(calls.filter((c) => c.toolName === "propose_playbook_patch"), []);
|
|
244
|
+
assert.deepStrictEqual(signalsFor(issue.id), []);
|
|
245
|
+
});
|
|
246
|
+
test("human retry feedback creates one idempotent signal_only report with refs and failure evidence", async () => {
|
|
247
|
+
const agent = seedAgent();
|
|
248
|
+
const issue = seedIssue(agent.id);
|
|
249
|
+
const session = seedTerminalSession(issue.id, agent);
|
|
250
|
+
const action = createHumanAction({
|
|
251
|
+
issueId: issue.id,
|
|
252
|
+
actionType: "final_review",
|
|
253
|
+
reason: "The widget drops state on rerender.",
|
|
254
|
+
question: "Repair or close?",
|
|
255
|
+
});
|
|
256
|
+
resolveHumanAction(action.id, "operator", { choice: "repair" });
|
|
257
|
+
const { deps, calls } = makeDeps({ fetches: [{ playbook_id: "pb-used-a" }, { playbook_id: "pb-used-b" }] });
|
|
258
|
+
const result = await triggerIssueReflect(issue.id, deps);
|
|
259
|
+
assert.equal(result, "triggered");
|
|
260
|
+
const proposes = calls.filter((c) => c.toolName === "propose_playbook_patch");
|
|
261
|
+
assert.equal(proposes.length, 1);
|
|
262
|
+
const args = proposes[0].args;
|
|
263
|
+
// Only signal_only from this path — never an update proposal, never Notes ops —
|
|
264
|
+
// and only Deck-schema fields. The independent Deck-schema fake above already
|
|
265
|
+
// accepted the call (proving a real Deck would), and Dealer's own validator
|
|
266
|
+
// must agree with that independent oracle.
|
|
267
|
+
assert.equal(deckSchemaError(args), null);
|
|
268
|
+
assert.equal(validateSignalOnlyArgs(args), null);
|
|
269
|
+
assert.equal(args.kind, "signal_only");
|
|
270
|
+
assert.equal("playbook_id" in args, false);
|
|
271
|
+
assert.equal("ops" in args, false);
|
|
272
|
+
for (const banned of ["source_key", "trigger", "issue_ref", "failure", "playbook_ids", "observed_playbook_use"]) {
|
|
273
|
+
assert.equal(banned in args, false, `Dealer-only field ${banned} must not be sent to Deck`);
|
|
274
|
+
}
|
|
275
|
+
// Every Dealer reference lives in Deck-stored fields instead.
|
|
276
|
+
const expectedKey = `dealer:${issue.id}:human_retry:${action.id}`;
|
|
277
|
+
assert.match(args.rationale, /Human correction during Dealer review/);
|
|
278
|
+
assert.ok(args.rationale.includes(expectedKey), "rationale carries the source key");
|
|
279
|
+
assert.ok(args.rationale.includes(issue.id), "rationale carries the issue ref");
|
|
280
|
+
const evidence = evidenceOf(args);
|
|
281
|
+
assert.match(evidence.failure_summary, /repair/);
|
|
282
|
+
assert.match(evidence.failure_summary, /drops state on rerender/);
|
|
283
|
+
assert.ok(evidence.failure_summary.includes(issue.id), "evidence carries the issue ref");
|
|
284
|
+
assert.ok(evidence.failure_summary.includes(session.id), "evidence carries the session ref");
|
|
285
|
+
assert.ok(evidence.failure_summary.includes("pb-used-a") && evidence.failure_summary.includes("pb-used-b"), "evidence names exactly the two actually used playbooks");
|
|
286
|
+
assert.ok(evidence.failure_summary.includes(expectedKey), "evidence carries the source key");
|
|
287
|
+
assert.equal(evidence.user_feedback_excerpt, "The widget drops state on rerender.");
|
|
288
|
+
// One pending intent row (written before the Deck call) plus one sent row.
|
|
289
|
+
assert.equal(pendingSignalsFor(issue.id).length, 1);
|
|
290
|
+
const signals = sentSignalsFor(issue.id);
|
|
291
|
+
assert.equal(signals.length, 1);
|
|
292
|
+
assert.equal(signals[0].trigger, "human_retry");
|
|
293
|
+
assert.equal(signals[0].sourceKey, expectedKey);
|
|
294
|
+
// Simulated restart: same trigger, no duplicate Deck report.
|
|
295
|
+
const again = await triggerIssueReflect(issue.id, deps);
|
|
296
|
+
assert.equal(again, "skipped");
|
|
297
|
+
assert.equal(calls.filter((c) => c.toolName === "propose_playbook_patch").length, 1);
|
|
298
|
+
assert.equal(sentSignalsFor(issue.id).length, 1);
|
|
299
|
+
});
|
|
300
|
+
test("attempts exhaustion creates one signal; a retry choice is reported once, not twice", async () => {
|
|
301
|
+
const agent = seedAgent();
|
|
302
|
+
const issue = seedIssue(agent.id);
|
|
303
|
+
seedTerminalSession(issue.id, agent);
|
|
304
|
+
const exhausted = createHumanAction({
|
|
305
|
+
issueId: issue.id,
|
|
306
|
+
actionType: "attempts_exhausted",
|
|
307
|
+
reason: "Review rounds spent with changes still requested.",
|
|
308
|
+
question: "Retry with another round or close?",
|
|
309
|
+
});
|
|
310
|
+
const { deps, calls } = makeDeps({ fetches: [] });
|
|
311
|
+
const result = await reportDeckFailureSignals(issue.id, deps);
|
|
312
|
+
assert.equal(result.sent.length, 1);
|
|
313
|
+
assert.equal(result.sent[0].trigger, "attempts_exhausted");
|
|
314
|
+
const args = calls.filter((c) => c.toolName === "propose_playbook_patch")[0].args;
|
|
315
|
+
assert.equal(deckSchemaError(args), null);
|
|
316
|
+
assert.equal(validateSignalOnlyArgs(args), null);
|
|
317
|
+
assert.equal(args.kind, "signal_only");
|
|
318
|
+
const exhaustedEvidence = evidenceOf(args);
|
|
319
|
+
assert.match(exhaustedEvidence.failure_summary, /exhausted/);
|
|
320
|
+
assert.ok(exhaustedEvidence.failure_summary.includes("no observed playbook use"), "no receipt means the report says no observed playbook use");
|
|
321
|
+
assert.ok(exhaustedEvidence.failure_summary.includes(issue.id));
|
|
322
|
+
// The human then retries: the same action must not produce a second signal.
|
|
323
|
+
resolveHumanAction(exhausted.id, "operator", { choice: "retry" });
|
|
324
|
+
const afterRetry = await reportDeckFailureSignals(issue.id, deps);
|
|
325
|
+
assert.deepStrictEqual(afterRetry.sent, []);
|
|
326
|
+
assert.equal(calls.filter((c) => c.toolName === "propose_playbook_patch").length, 1);
|
|
327
|
+
assert.equal(sentSignalsFor(issue.id).length, 1);
|
|
328
|
+
// And a restart changes nothing.
|
|
329
|
+
const restart = await reportDeckFailureSignals(issue.id, deps);
|
|
330
|
+
assert.deepStrictEqual(restart.sent, []);
|
|
331
|
+
assert.equal(sentSignalsFor(issue.id).length, 1);
|
|
332
|
+
});
|
|
333
|
+
test("a recurring blocking finding creates one signal; non-blocking recurrence does not", async () => {
|
|
334
|
+
const agent = seedAgent();
|
|
335
|
+
const issue = seedIssue(agent.id);
|
|
336
|
+
seedTerminalSession(issue.id, agent);
|
|
337
|
+
reconcileFinding({
|
|
338
|
+
issueId: issue.id, fingerprint: "missing-null-check", severity: "blocking",
|
|
339
|
+
title: "Missing null check", rationale: "crashes on empty input", round: 1,
|
|
340
|
+
});
|
|
341
|
+
reconcileFinding({
|
|
342
|
+
issueId: issue.id, fingerprint: "missing-null-check", severity: "blocking",
|
|
343
|
+
title: "Missing null check", rationale: "still crashes", round: 2,
|
|
344
|
+
});
|
|
345
|
+
reconcileFinding({
|
|
346
|
+
issueId: issue.id, fingerprint: "typo-in-comment", severity: "non_blocking",
|
|
347
|
+
title: "Typo", rationale: "cosmetic", round: 1,
|
|
348
|
+
});
|
|
349
|
+
reconcileFinding({
|
|
350
|
+
issueId: issue.id, fingerprint: "typo-in-comment", severity: "non_blocking",
|
|
351
|
+
title: "Typo", rationale: "still cosmetic", round: 2,
|
|
352
|
+
});
|
|
353
|
+
const { deps, calls } = makeDeps({ fetches: [] });
|
|
354
|
+
const result = await reportDeckFailureSignals(issue.id, deps);
|
|
355
|
+
assert.equal(result.sent.length, 1);
|
|
356
|
+
assert.equal(result.sent[0].trigger, "recurring_blocking");
|
|
357
|
+
const args = calls.filter((c) => c.toolName === "propose_playbook_patch")[0].args;
|
|
358
|
+
// The recurring_blocking report carries no human reason, yet Deck requires
|
|
359
|
+
// user_feedback_excerpt — both the independent Deck oracle and Dealer's
|
|
360
|
+
// validator must accept what the builder produced.
|
|
361
|
+
assert.equal(deckSchemaError(args), null);
|
|
362
|
+
assert.equal(validateSignalOnlyArgs(args), null);
|
|
363
|
+
assert.equal(args.kind, "signal_only");
|
|
364
|
+
const recurringEvidence = evidenceOf(args);
|
|
365
|
+
assert.match(recurringEvidence.failure_summary, /missing-null-check/);
|
|
366
|
+
assert.doesNotMatch(recurringEvidence.failure_summary, /typo-in-comment/);
|
|
367
|
+
assert.ok(recurringEvidence.failure_summary.includes("no observed playbook use"));
|
|
368
|
+
// The excerpt falls back to the recurring finding's own rationale.
|
|
369
|
+
assert.ok(recurringEvidence.user_feedback_excerpt.trim(), "recurring report carries a non-empty excerpt");
|
|
370
|
+
const restart = await reportDeckFailureSignals(issue.id, deps);
|
|
371
|
+
assert.deepStrictEqual(restart.sent, []);
|
|
372
|
+
assert.equal(sentSignalsFor(issue.id).length, 1);
|
|
373
|
+
});
|
|
374
|
+
test("blank human reasons still produce Deck-acceptable signals with an explicit excerpt", async () => {
|
|
375
|
+
const agent = seedAgent();
|
|
376
|
+
const issue = seedIssue(agent.id);
|
|
377
|
+
seedTerminalSession(issue.id, agent);
|
|
378
|
+
// Whitespace-only reason: excerpt() degrades to null, so the builder must
|
|
379
|
+
// substitute the explicit no-feedback line — a real Deck rejects the omission.
|
|
380
|
+
const repair = createHumanAction({
|
|
381
|
+
issueId: issue.id, actionType: "final_review", reason: " ", question: "Repair?",
|
|
382
|
+
});
|
|
383
|
+
resolveHumanAction(repair.id, "operator", { choice: "repair" });
|
|
384
|
+
const exhausted = createHumanAction({
|
|
385
|
+
issueId: issue.id, actionType: "attempts_exhausted", reason: "", question: "Retry?",
|
|
386
|
+
});
|
|
387
|
+
const { deps, calls } = makeDeps({ fetches: [] });
|
|
388
|
+
const result = await reportDeckFailureSignals(issue.id, deps);
|
|
389
|
+
// The attempts_exhausted action is still pending, so both triggers fire.
|
|
390
|
+
assert.equal(result.sent.length, 2);
|
|
391
|
+
const proposes = calls.filter((c) => c.toolName === "propose_playbook_patch");
|
|
392
|
+
assert.equal(proposes.length, 2);
|
|
393
|
+
for (const call of proposes) {
|
|
394
|
+
// Accepted by the independent Deck-schema oracle (not just Dealer's validator).
|
|
395
|
+
assert.equal(deckSchemaError(call.args), null);
|
|
396
|
+
assert.equal(validateSignalOnlyArgs(call.args), null);
|
|
397
|
+
assert.equal(evidenceOf(call.args).user_feedback_excerpt, NO_HUMAN_FEEDBACK_EXCERPT);
|
|
398
|
+
}
|
|
399
|
+
// Idempotent across restart: no duplicates.
|
|
400
|
+
const restart = await reportDeckFailureSignals(issue.id, deps);
|
|
401
|
+
assert.deepStrictEqual(restart.sent, []);
|
|
402
|
+
assert.equal(calls.filter((c) => c.toolName === "propose_playbook_patch").length, 2);
|
|
403
|
+
});
|
|
404
|
+
test("all three triggers together send three signal_only reports, never an update", async () => {
|
|
405
|
+
const agent = seedAgent();
|
|
406
|
+
const issue = seedIssue(agent.id);
|
|
407
|
+
seedTerminalSession(issue.id, agent);
|
|
408
|
+
const repair = createHumanAction({
|
|
409
|
+
issueId: issue.id, actionType: "final_review", reason: "Needs repair.", question: "Repair?",
|
|
410
|
+
});
|
|
411
|
+
resolveHumanAction(repair.id, "operator", { choice: "repair" });
|
|
412
|
+
createHumanAction({
|
|
413
|
+
issueId: issue.id, actionType: "attempts_exhausted", reason: "Rounds spent.", question: "Retry?",
|
|
414
|
+
});
|
|
415
|
+
reconcileFinding({
|
|
416
|
+
issueId: issue.id, fingerprint: "flaky-lock", severity: "blocking",
|
|
417
|
+
title: "Flaky lock", rationale: "deadlocks", round: 1,
|
|
418
|
+
});
|
|
419
|
+
reconcileFinding({
|
|
420
|
+
issueId: issue.id, fingerprint: "flaky-lock", severity: "blocking",
|
|
421
|
+
title: "Flaky lock", rationale: "deadlocks again", round: 2,
|
|
422
|
+
});
|
|
423
|
+
const { deps, calls } = makeDeps({ fetches: [{ playbook_id: "pb-e" }] });
|
|
424
|
+
const result = await reportDeckFailureSignals(issue.id, deps);
|
|
425
|
+
assert.equal(result.sent.length, 3);
|
|
426
|
+
const proposes = calls.filter((c) => c.toolName === "propose_playbook_patch");
|
|
427
|
+
assert.equal(proposes.length, 3);
|
|
428
|
+
for (const call of proposes) {
|
|
429
|
+
assert.equal(call.args.kind, "signal_only");
|
|
430
|
+
assert.equal(deckSchemaError(call.args), null);
|
|
431
|
+
assert.equal(validateSignalOnlyArgs(call.args), null);
|
|
432
|
+
assert.equal("ops" in call.args, false);
|
|
433
|
+
assert.equal("playbook_id" in call.args, false);
|
|
434
|
+
}
|
|
435
|
+
const triggers = result.sent.map((s) => s.trigger).sort();
|
|
436
|
+
assert.deepStrictEqual(triggers, ["attempts_exhausted", "human_retry", "recurring_blocking"]);
|
|
437
|
+
// Deterministic source keys are unique per trigger.
|
|
438
|
+
assert.equal(new Set(result.sent.map((s) => s.sourceKey)).size, 3);
|
|
439
|
+
});
|
|
440
|
+
test("Deck outage records a visible error and preserves the issue state", async () => {
|
|
441
|
+
const agent = seedAgent();
|
|
442
|
+
const issue = seedIssue(agent.id);
|
|
443
|
+
const session = seedTerminalSession(issue.id, agent, "failed");
|
|
444
|
+
const before = (await import("../repository/issues.js")).getIssue(issue.id);
|
|
445
|
+
const { deps, calls } = makeDeps({ healthy: false });
|
|
446
|
+
const result = await triggerIssueReflect(issue.id, deps);
|
|
447
|
+
assert.equal(result, "failed");
|
|
448
|
+
// Never touches Deck tools when the health gate fails.
|
|
449
|
+
assert.deepStrictEqual(calls, []);
|
|
450
|
+
// Visible status, retryable on a later trigger.
|
|
451
|
+
const statuses = listArtifactsForIssueByKind(issue.id, "reflect_status").map((a) => JSON.parse(a.contentJson));
|
|
452
|
+
assert.ok(statuses.some((s) => s.status === "failed" || s.status === "error" || /offline/i.test(s.error ?? s.reason ?? "")));
|
|
453
|
+
const receipts = receiptsFor(issue.id);
|
|
454
|
+
assert.equal(receipts.length, 1);
|
|
455
|
+
assert.equal(receipts[0].status, "error");
|
|
456
|
+
// The already-decided issue state is untouched.
|
|
457
|
+
assert.equal((await import("../repository/issues.js")).getIssue(issue.id).status, before.status);
|
|
458
|
+
assert.equal(getWorkerSession(session.id).status, "failed");
|
|
459
|
+
});
|
|
460
|
+
test("a malformed correlation response records an error receipt, keeps state, retries later", async () => {
|
|
461
|
+
const agent = seedAgent();
|
|
462
|
+
const issue = seedIssue(agent.id);
|
|
463
|
+
const session = seedTerminalSession(issue.id, agent);
|
|
464
|
+
const first = makeDeps({ correlationData: { unexpected: "shape" } });
|
|
465
|
+
const result = await collectPlaybookUseReceipt(session.id, first.deps);
|
|
466
|
+
assert.equal(result.collected, false);
|
|
467
|
+
const receipts = receiptsFor(issue.id);
|
|
468
|
+
assert.equal(receipts.length, 1);
|
|
469
|
+
assert.equal(receipts[0].status, "error");
|
|
470
|
+
// Deck recovers: the next trigger heals the gap with a collected row.
|
|
471
|
+
const healed = makeDeps({ fetches: [{ playbook_id: "pb-late" }] });
|
|
472
|
+
const retry = await collectPlaybookUseReceipt(session.id, healed.deps);
|
|
473
|
+
assert.equal(retry.collected, true);
|
|
474
|
+
assert.deepStrictEqual(receiptsFor(issue.id).filter((r) => r.status === "collected").map((r) => r.playbookIds), [["pb-late"]]);
|
|
475
|
+
// Issue-level collection now reads the latest (collected) row — the healed gap
|
|
476
|
+
// does not count as an error and a later trigger is not stuck on "failed".
|
|
477
|
+
const healedIssue = await collectPlaybookUseReceiptsForIssue(issue.id, healed.deps);
|
|
478
|
+
assert.equal(healedIssue.errors, 0);
|
|
479
|
+
});
|
|
480
|
+
test("parseCorrelatedFetches rejects identity-less entries instead of guessing IDs", () => {
|
|
481
|
+
assert.deepStrictEqual(parseCorrelatedFetches({ fetches: [] }), []);
|
|
482
|
+
assert.equal(parseCorrelatedFetches({ nope: 1 }), null);
|
|
483
|
+
assert.equal(parseCorrelatedFetches({ fetches: [{ playbook_id: " " }] }), null);
|
|
484
|
+
assert.equal(parseCorrelatedFetches({ fetches: [{ title: "no id" }] }), null);
|
|
485
|
+
assert.deepStrictEqual(parseCorrelatedFetches({ usage: [{ playbookId: "pb-camel" }] }), [
|
|
486
|
+
{ playbookId: "pb-camel", firstFetchedAt: null, lastFetchedAt: null },
|
|
487
|
+
]);
|
|
488
|
+
});
|
|
489
|
+
test("issue-level collection covers every terminal session and ignores running ones", async () => {
|
|
490
|
+
const agent = seedAgent();
|
|
491
|
+
const issue = seedIssue(agent.id);
|
|
492
|
+
const done = seedTerminalSession(issue.id, agent, "done");
|
|
493
|
+
const failed = seedTerminalSession(issue.id, agent, "failed");
|
|
494
|
+
const snapshot = { ...buildProfileSnapshot(agent, "developer"), deckId: DECK };
|
|
495
|
+
// Still running — no receipt yet.
|
|
496
|
+
createWorkerSession({
|
|
497
|
+
issueId: issue.id, role: "reviewer", round: 1, agentId: agent.id,
|
|
498
|
+
runtime: agent.runtime, profileSnapshotJson: serializeProfileSnapshot(snapshot),
|
|
499
|
+
});
|
|
500
|
+
const { deps } = makeDeps({ fetches: [{ playbook_id: "pb-s" }] });
|
|
501
|
+
const result = await collectPlaybookUseReceiptsForIssue(issue.id, deps);
|
|
502
|
+
assert.equal(result.collected, 2);
|
|
503
|
+
const ids = receiptsFor(issue.id).map((r) => r.workerSessionId).sort();
|
|
504
|
+
assert.deepStrictEqual(ids, [done.id, failed.id].sort());
|
|
505
|
+
});
|
|
506
|
+
test("buildSignalOnlyArgs emits only Deck-schema fields with refs in Deck-stored fields", () => {
|
|
507
|
+
const args = buildSignalOnlyArgs({
|
|
508
|
+
trigger: "human_retry",
|
|
509
|
+
sourceKey: "dealer:issue-1:human_retry:act-1",
|
|
510
|
+
humanActionId: "act-1",
|
|
511
|
+
failure: "Human repair on final_review (action act-1). Stated reason: broken.",
|
|
512
|
+
userFeedback: "broken",
|
|
513
|
+
}, { issueId: "issue-1", workerSessionIds: ["ws-1"], playbookIds: ["pb-a", "pb-b"] });
|
|
514
|
+
assert.equal(deckSchemaError(args), null);
|
|
515
|
+
assert.equal(validateSignalOnlyArgs(args), null);
|
|
516
|
+
assert.ok(args.rationale.includes("dealer:issue-1:human_retry:act-1"));
|
|
517
|
+
const evidence = evidenceOf(args);
|
|
518
|
+
assert.ok(evidence.failure_summary.includes("Dealer issue: issue-1"));
|
|
519
|
+
assert.ok(evidence.failure_summary.includes("Worker sessions: ws-1"));
|
|
520
|
+
assert.ok(evidence.failure_summary.includes("pb-a") && evidence.failure_summary.includes("pb-b"));
|
|
521
|
+
assert.equal(evidence.user_feedback_excerpt, "broken");
|
|
522
|
+
// No observed use is explicit, never a legacy fallback — and with no human
|
|
523
|
+
// reason the builder still emits the explicit no-feedback excerpt, because
|
|
524
|
+
// Deck rejects a missing user_feedback_excerpt.
|
|
525
|
+
const noneArgs = buildSignalOnlyArgs({
|
|
526
|
+
trigger: "attempts_exhausted",
|
|
527
|
+
sourceKey: "dealer:issue-1:attempts_exhausted:act-2",
|
|
528
|
+
humanActionId: "act-2",
|
|
529
|
+
failure: "Review attempts were exhausted.",
|
|
530
|
+
userFeedback: null,
|
|
531
|
+
}, { issueId: "issue-1", workerSessionIds: [], playbookIds: [] });
|
|
532
|
+
assert.equal(deckSchemaError(noneArgs), null);
|
|
533
|
+
assert.equal(validateSignalOnlyArgs(noneArgs), null);
|
|
534
|
+
assert.match(evidenceOf(noneArgs).failure_summary, /no observed playbook use/);
|
|
535
|
+
assert.equal(evidenceOf(noneArgs).user_feedback_excerpt, NO_HUMAN_FEEDBACK_EXCERPT);
|
|
536
|
+
// A blank/whitespace-only reason degrades to the same explicit fallback.
|
|
537
|
+
const blankArgs = buildSignalOnlyArgs({
|
|
538
|
+
trigger: "human_retry",
|
|
539
|
+
sourceKey: "dealer:issue-1:human_retry:act-3",
|
|
540
|
+
humanActionId: "act-3",
|
|
541
|
+
failure: "Human repair on final_review (action act-3).",
|
|
542
|
+
userFeedback: " ",
|
|
543
|
+
}, { issueId: "issue-1", workerSessionIds: ["ws-1"], playbookIds: [] });
|
|
544
|
+
assert.equal(deckSchemaError(blankArgs), null);
|
|
545
|
+
assert.equal(validateSignalOnlyArgs(blankArgs), null);
|
|
546
|
+
assert.equal(evidenceOf(blankArgs).user_feedback_excerpt, NO_HUMAN_FEEDBACK_EXCERPT);
|
|
547
|
+
// Dealer's validator agrees with Deck's contract: a missing excerpt is rejected.
|
|
548
|
+
assert.match(validateSignalOnlyArgs({
|
|
549
|
+
kind: "signal_only",
|
|
550
|
+
rationale: "r",
|
|
551
|
+
evidence: { failure_summary: "f" },
|
|
552
|
+
}), /user_feedback_excerpt/);
|
|
553
|
+
// The validator rejects every Dealer-only smuggled field a real Deck would drop.
|
|
554
|
+
for (const extra of ["source_key", "trigger", "issue_ref", "failure", "playbook_ids", "observed_playbook_use"]) {
|
|
555
|
+
const rejected = validateSignalOnlyArgs({
|
|
556
|
+
kind: "signal_only",
|
|
557
|
+
rationale: "r",
|
|
558
|
+
evidence: { failure_summary: "f" },
|
|
559
|
+
[extra]: "x",
|
|
560
|
+
});
|
|
561
|
+
assert.match(rejected, /unknown top-level field/, `${extra} must be rejected`);
|
|
562
|
+
}
|
|
563
|
+
assert.equal(validateSignalOnlyArgs({ kind: "signal_only", rationale: "r", evidence: { nope: 1 } }), "unknown evidence field: nope");
|
|
564
|
+
});
|
|
565
|
+
test("a failed Deck send leaves the pending intent; the retry reuses the same key", async () => {
|
|
566
|
+
const agent = seedAgent();
|
|
567
|
+
const issue = seedIssue(agent.id);
|
|
568
|
+
seedTerminalSession(issue.id, agent);
|
|
569
|
+
const action = createHumanAction({
|
|
570
|
+
issueId: issue.id, actionType: "final_review", reason: "Flaky on retry.", question: "Repair?",
|
|
571
|
+
});
|
|
572
|
+
resolveHumanAction(action.id, "operator", { choice: "repair" });
|
|
573
|
+
const expectedKey = `dealer:${issue.id}:human_retry:${action.id}`;
|
|
574
|
+
const failing = makeDeps({
|
|
575
|
+
fetches: [],
|
|
576
|
+
propose: () => ({ ok: false, kind: "infra_failure", reason: "temporary deck error" }),
|
|
577
|
+
});
|
|
578
|
+
const first = await reportDeckFailureSignals(issue.id, failing.deps);
|
|
579
|
+
assert.deepStrictEqual(first.sent, []);
|
|
580
|
+
assert.equal(failing.calls.filter((c) => c.toolName === "propose_playbook_patch").length, 1);
|
|
581
|
+
// No sent row, but the pending intent survives the failure for the retry.
|
|
582
|
+
assert.equal(sentSignalsFor(issue.id).length, 0);
|
|
583
|
+
assert.equal(pendingSignalsFor(issue.id).length, 1);
|
|
584
|
+
assert.equal(pendingSignalsFor(issue.id)[0].sourceKey, expectedKey);
|
|
585
|
+
const succeeding = makeDeps({ fetches: [] });
|
|
586
|
+
const second = await reportDeckFailureSignals(issue.id, succeeding.deps);
|
|
587
|
+
assert.equal(second.sent.length, 1);
|
|
588
|
+
assert.equal(second.sent[0].sourceKey, expectedKey);
|
|
589
|
+
// No second pending row: the retry reconciled to the same intent.
|
|
590
|
+
assert.equal(pendingSignalsFor(issue.id).length, 1);
|
|
591
|
+
assert.equal(sentSignalsFor(issue.id).length, 1);
|
|
592
|
+
const retryEvidence = evidenceOf(succeeding.calls.filter((c) => c.toolName === "propose_playbook_patch")[0].args);
|
|
593
|
+
assert.ok(retryEvidence.failure_summary.includes(expectedKey));
|
|
594
|
+
const failedEvidence = evidenceOf(failing.calls.filter((c) => c.toolName === "propose_playbook_patch")[0].args);
|
|
595
|
+
assert.equal(retryEvidence.failure_summary, failedEvidence.failure_summary);
|
|
596
|
+
});
|
|
597
|
+
test("a crash between the Deck send and the sent-write reconciles to the same key", async () => {
|
|
598
|
+
const agent = seedAgent();
|
|
599
|
+
const issue = seedIssue(agent.id);
|
|
600
|
+
const session = seedTerminalSession(issue.id, agent);
|
|
601
|
+
const action = createHumanAction({
|
|
602
|
+
issueId: issue.id, actionType: "final_review", reason: "Drops state.", question: "Repair?",
|
|
603
|
+
});
|
|
604
|
+
resolveHumanAction(action.id, "operator", { choice: "repair" });
|
|
605
|
+
const expectedKey = `dealer:${issue.id}:human_retry:${action.id}`;
|
|
606
|
+
// Simulate the crash window: a first attempt wrote its pending intent and
|
|
607
|
+
// reached Deck, but the process died before writing the sent row.
|
|
608
|
+
createIssueArtifact({
|
|
609
|
+
issueId: issue.id,
|
|
610
|
+
kind: "deck_feedback_signal",
|
|
611
|
+
author: "system",
|
|
612
|
+
content: {
|
|
613
|
+
sourceKey: expectedKey,
|
|
614
|
+
trigger: "human_retry",
|
|
615
|
+
humanActionId: action.id,
|
|
616
|
+
signalId: null,
|
|
617
|
+
deckId: DECK,
|
|
618
|
+
workerSessionIds: [session.id],
|
|
619
|
+
playbookIds: [],
|
|
620
|
+
observedPlaybookUse: "none",
|
|
621
|
+
failure: "Human repair on final_review (action placeholder).",
|
|
622
|
+
status: "pending",
|
|
623
|
+
},
|
|
624
|
+
});
|
|
625
|
+
const { deps, calls } = makeDeps({ fetches: [] });
|
|
626
|
+
const result = await reportDeckFailureSignals(issue.id, deps);
|
|
627
|
+
assert.equal(result.sent.length, 1);
|
|
628
|
+
assert.equal(result.sent[0].sourceKey, expectedKey);
|
|
629
|
+
// Exactly one send, carrying the identical Deck-visible key, and no second
|
|
630
|
+
// pending row — Deck can recognize the repeat as the same report.
|
|
631
|
+
const proposes = calls.filter((c) => c.toolName === "propose_playbook_patch");
|
|
632
|
+
assert.equal(proposes.length, 1);
|
|
633
|
+
assert.ok(evidenceOf(proposes[0].args).failure_summary.includes(expectedKey));
|
|
634
|
+
assert.equal(pendingSignalsFor(issue.id).length, 1);
|
|
635
|
+
assert.equal(sentSignalsFor(issue.id).length, 1);
|
|
636
|
+
// A further restart sends nothing more.
|
|
637
|
+
const restart = await reportDeckFailureSignals(issue.id, deps);
|
|
638
|
+
assert.deepStrictEqual(restart.sent, []);
|
|
639
|
+
assert.equal(calls.filter((c) => c.toolName === "propose_playbook_patch").length, 1);
|
|
640
|
+
});
|
|
641
|
+
test("an empty collected receipt is provisional: the next trigger re-reads instead of freezing", async () => {
|
|
642
|
+
const agent = seedAgent();
|
|
643
|
+
const issue = seedIssue(agent.id);
|
|
644
|
+
const session = seedTerminalSession(issue.id, agent);
|
|
645
|
+
// Session-end probe runs before Deck ingested anything: empty but collected.
|
|
646
|
+
const early = makeDeps({ fetches: [] });
|
|
647
|
+
const first = await collectPlaybookUseReceipt(session.id, early.deps);
|
|
648
|
+
assert.equal(first.collected, true);
|
|
649
|
+
if (first.collected) {
|
|
650
|
+
assert.equal(first.duplicate, false);
|
|
651
|
+
assert.deepStrictEqual(first.receipt.playbookIds, []);
|
|
652
|
+
}
|
|
653
|
+
// Issue-level trigger re-reads rather than freezing the early empty list.
|
|
654
|
+
const late = makeDeps({ fetches: [{ playbook_id: "pb-late" }] });
|
|
655
|
+
const second = await collectPlaybookUseReceipt(session.id, late.deps);
|
|
656
|
+
assert.equal(second.collected, true);
|
|
657
|
+
if (second.collected) {
|
|
658
|
+
assert.equal(second.duplicate, false);
|
|
659
|
+
assert.deepStrictEqual(second.receipt.playbookIds, ["pb-late"]);
|
|
660
|
+
}
|
|
661
|
+
// The fresh row supersedes the provisional one; readers take the latest
|
|
662
|
+
// (newest-first, so index 0 is the superseding row).
|
|
663
|
+
assert.equal(receiptsFor(issue.id).length, 2);
|
|
664
|
+
assert.deepStrictEqual(receiptsFor(issue.id)[0].status, "collected");
|
|
665
|
+
assert.deepStrictEqual(receiptsFor(issue.id)[0].playbookIds, ["pb-late"]);
|
|
666
|
+
// A non-empty collected receipt IS terminal: no further Deck calls.
|
|
667
|
+
const third = await collectPlaybookUseReceipt(session.id, late.deps);
|
|
668
|
+
assert.equal(third.collected, true);
|
|
669
|
+
if (third.collected)
|
|
670
|
+
assert.equal(third.duplicate, true);
|
|
671
|
+
assert.equal(late.calls.filter((c) => c.toolName === DECK_CORRELATION_TOOL).length, 1);
|
|
672
|
+
});
|
|
673
|
+
test("concurrent collection for one session shares a single Deck call", async () => {
|
|
674
|
+
const agent = seedAgent();
|
|
675
|
+
const issue = seedIssue(agent.id);
|
|
676
|
+
const session = seedTerminalSession(issue.id, agent);
|
|
677
|
+
let release;
|
|
678
|
+
const gate = new Promise((resolve) => {
|
|
679
|
+
release = resolve;
|
|
680
|
+
});
|
|
681
|
+
let correlationCalls = 0;
|
|
682
|
+
const deps = {
|
|
683
|
+
checkHealth: async () => true,
|
|
684
|
+
callTool: (async (call) => {
|
|
685
|
+
if (call.toolName === DECK_CORRELATION_TOOL) {
|
|
686
|
+
correlationCalls++;
|
|
687
|
+
return gate;
|
|
688
|
+
}
|
|
689
|
+
throw new Error(`unexpected tool call: ${call.toolName}`);
|
|
690
|
+
}),
|
|
691
|
+
};
|
|
692
|
+
// Session-end probe and issue-level trigger racing for the just-finished session.
|
|
693
|
+
const first = collectPlaybookUseReceipt(session.id, deps);
|
|
694
|
+
const second = collectPlaybookUseReceipt(session.id, deps);
|
|
695
|
+
release({ ok: true, data: { fetches: [{ playbook_id: "pb-shared" }] } });
|
|
696
|
+
const [r1, r2] = await Promise.all([first, second]);
|
|
697
|
+
assert.equal(r1.collected, true);
|
|
698
|
+
assert.equal(r2.collected, true);
|
|
699
|
+
assert.equal(correlationCalls, 1);
|
|
700
|
+
assert.equal(receiptsFor(issue.id).length, 1);
|
|
701
|
+
assert.deepStrictEqual(receiptsFor(issue.id)[0].playbookIds, ["pb-shared"]);
|
|
702
|
+
});
|