agent-dealer 1.2.8 → 1.2.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/bundle/server/dist/adapters/github.js +80 -4
  2. package/bundle/server/dist/adapters/github.test.js +83 -1
  3. package/bundle/server/dist/adapters/worktree-deps.js +96 -0
  4. package/bundle/server/dist/adapters/worktree-deps.test.js +144 -0
  5. package/bundle/server/dist/coordinator/auto-merge.js +205 -1
  6. package/bundle/server/dist/coordinator/checks-wait-config.js +19 -0
  7. package/bundle/server/dist/coordinator/commands.js +194 -18
  8. package/bundle/server/dist/coordinator/commands.test.js +47 -0
  9. package/bundle/server/dist/coordinator/developer-effect.js +145 -3
  10. package/bundle/server/dist/coordinator/developer-effect.test.js +288 -3
  11. package/bundle/server/dist/coordinator/human-resolution.js +19 -0
  12. package/bundle/server/dist/coordinator/human-resolution.test.js +33 -0
  13. package/bundle/server/dist/coordinator/operator-criteria.js +168 -0
  14. package/bundle/server/dist/coordinator/operator-criteria.test.js +78 -0
  15. package/bundle/server/dist/coordinator/operator-verification.integration.test.js +347 -0
  16. package/bundle/server/dist/coordinator/playbook-feedback.js +3 -0
  17. package/bundle/server/dist/coordinator/projection.js +10 -4
  18. package/bundle/server/dist/coordinator/projection.test.js +13 -0
  19. package/bundle/server/dist/coordinator/prompts.js +134 -1
  20. package/bundle/server/dist/coordinator/prompts.test.js +148 -0
  21. package/bundle/server/dist/coordinator/recovery.js +57 -3
  22. package/bundle/server/dist/coordinator/recovery.test.js +67 -8
  23. package/bundle/server/dist/coordinator/repair-cycle.integration.test.js +1 -0
  24. package/bundle/server/dist/coordinator/reviewer-effect.js +6 -0
  25. package/bundle/server/dist/coordinator/routing.js +45 -6
  26. package/bundle/server/dist/coordinator/routing.test.js +62 -15
  27. package/bundle/server/dist/coordinator/usage-cap-defer.js +61 -0
  28. package/bundle/server/dist/db/ci-attempts-migration.test.js +70 -0
  29. package/bundle/server/dist/db/index.js +13 -0
  30. package/bundle/server/dist/db/schema.sql +2 -0
  31. package/bundle/server/dist/repository/issues.js +39 -3
  32. package/bundle/server/dist/repository/work-items.js +7 -0
  33. package/bundle/server/dist/runners/muse-code-args.js +25 -0
  34. package/bundle/server/dist/runners/muse-code-args.test.js +30 -0
  35. package/bundle/server/package.json +2 -2
  36. package/bundle/server/static-ui/assets/index-CU6fNntg.js +64 -0
  37. package/bundle/server/static-ui/index.html +1 -1
  38. package/bundle/shared/dist/human-actions.d.ts +4 -4
  39. package/bundle/shared/dist/human-actions.js +8 -0
  40. package/bundle/shared/dist/issues.d.ts +30 -0
  41. package/bundle/shared/dist/issues.js +14 -0
  42. package/bundle/shared/dist/issues.test.js +2 -0
  43. package/bundle/shared/dist/workflow.d.ts +4 -4
  44. package/bundle/shared/dist/workflow.js +6 -0
  45. package/bundle/shared/dist/workflow.test.js +1 -0
  46. package/bundle/shared/package.json +1 -1
  47. package/dist/action.js +4 -1
  48. package/dist/action.test.js +42 -0
  49. package/dist/index.js +1 -1
  50. package/package.json +1 -1
  51. package/bundle/server/static-ui/assets/index-B6SVCzMR.js +0 -64
@@ -1,5 +1,7 @@
1
+ import { MUSE_SANDBOX_CAPABILITIES } from "../runners/muse-code-args.js";
1
2
  import { checkMuseVisualQa, museVisualQaPromptSection, } from "../adapters/muse-visual-qa.js";
2
3
  import { formatVerificationReceiptSection, } from "./verification-receipt.js";
4
+ import { formatOperatorCriteria, } from "./operator-criteria.js";
3
5
  /**
4
6
  * NOT-306: render the frozen execution contract as dedicated sections. These
5
7
  * are the ticket's own execution semantics — the worker follows them as
@@ -67,6 +69,113 @@ function conflictRepairSection(directive) {
67
69
  ``,
68
70
  ];
69
71
  }
72
+ /** NOT-314: the frozen snapshot's `[operator]` criteria for the developer.
73
+ * Empty/absent renders nothing so operator-free tasks produce byte-for-byte
74
+ * the prompt they always did. */
75
+ function operatorDeveloperSection(criteria) {
76
+ if (!criteria?.length)
77
+ return [];
78
+ return [
79
+ `## Operator verification required ([operator] criteria)`,
80
+ `These acceptance criteria can only be proven by a human operator with real credentials, a real login, or a paid session — Dealer blocks the merge until the operator records the result, so do not work around the gate. Do not attempt these; ship the ready-to-run probe and a doc, put the exact command in the PR body:`,
81
+ ``,
82
+ ...formatOperatorCriteria(criteria),
83
+ ``,
84
+ ];
85
+ }
86
+ /** NOT-314: an operator_verification `repair` note, verbatim. Empty/absent
87
+ * renders nothing so ordinary rounds produce byte-for-byte the prompt they always did. */
88
+ function operatorRepairSection(note) {
89
+ if (!note?.trim())
90
+ return [];
91
+ return [
92
+ `## Operator verification feedback`,
93
+ `A human reviewed the operator-gated criteria and sent the work back instead of verifying. Address this before anything else, and do not ask the operator to re-verify the same head without fixing it:`,
94
+ ``,
95
+ note.trim(),
96
+ ``,
97
+ ];
98
+ }
99
+ /** NOT-314: the frozen snapshot's `[operator]` criteria for the reviewer.
100
+ * Missing operator evidence is NOT a defect (Dealer gates the merge itself) —
101
+ * but the probe and the doc must exist. Empty/absent renders nothing. */
102
+ function operatorReviewerSection(criteria) {
103
+ if (!criteria?.length)
104
+ return [];
105
+ return [
106
+ `## Operator-gated criteria ([operator]) — not yours to verify`,
107
+ `Missing operator evidence is NOT a defect: Dealer blocks the merge until a human operator records the result, so never raise a blocking finding and never escalate for an unverified [operator] criterion. What you MUST check instead is that the probe and doc exist: the diff must ship the ready-to-run probe for each criterion below, the doc must name the exact command, and the PR body must carry it. A missing probe or doc IS a blocking finding.`,
108
+ ``,
109
+ ...formatOperatorCriteria(criteria),
110
+ ``,
111
+ ];
112
+ }
113
+ /**
114
+ * NOT-316: the Muse developer sandbox limits, rendered from
115
+ * `MUSE_SANDBOX_CAPABILITIES` (which is tied by test to the real
116
+ * `--sandbox-network` launch flag) instead of hand-written prose, so the
117
+ * prompt cannot drift from what the sandbox actually does. Muse developers
118
+ * only — returns [] otherwise, so non-Muse prompts are byte-for-byte
119
+ * unchanged.
120
+ */
121
+ function museEnvironmentLimitsSection() {
122
+ const caps = MUSE_SANDBOX_CAPABILITIES;
123
+ return [
124
+ `## Environment limits (sandbox network: ${caps.sandboxNetwork})`,
125
+ `This session runs inside the Muse sandbox with network \`${caps.network}\` — these limits come from the launch flags and cannot be worked around:`,
126
+ ``,
127
+ `- Network: ${caps.network} (no DNS — the npm registry and GitHub API are unreachable).`,
128
+ `- Loopback listeners: ${caps.loopbackListen ? "yes" : "no"} — \`listen()\` fails with EPERM, so do not start servers.`,
129
+ `- Browser: ${caps.browser ? "yes" : "no"}.`,
130
+ `- Credentials/Keychain: ${caps.credentialsKeychain ? "yes" : "no"}.`,
131
+ `After one failed attempt at something the environment cannot do (install, listen, browser, credentials), stop and hand off; do not work around the sandbox.`,
132
+ ``,
133
+ ];
134
+ }
135
+ function acceptanceCriteriaHasTag(acceptanceCriteria, tag) {
136
+ return acceptanceCriteria.includes(tag);
137
+ }
138
+ /**
139
+ * NOT-316: what `[agent]` / `[ci]` tags on acceptance criteria mean for a Muse
140
+ * developer. Rendered only when the AC text carries the tag (per-tag lines)
141
+ * and only for Muse developers — otherwise returns [] so untagged and
142
+ * non-Muse prompts are byte-for-byte what they always were. `[operator]`
143
+ * criteria are owned by the operator-gate section (`operatorDeveloperSection`)
144
+ * and deliberately not duplicated here.
145
+ */
146
+ function tagSemanticsDeveloperSection(acceptanceCriteria) {
147
+ const lines = [];
148
+ if (acceptanceCriteriaHasTag(acceptanceCriteria, "[agent]")) {
149
+ lines.push(`- \`[agent]\`: verify yourself — prove the criterion with your own in-session runs (targeted tests); your verification is the evidence.`);
150
+ }
151
+ if (acceptanceCriteriaHasTag(acceptanceCriteria, "[ci]")) {
152
+ lines.push(`- \`[ci]\`: implement or extend the named CI job/test and push — CI is the evidence, do not reproduce it locally.`);
153
+ }
154
+ if (!lines.length)
155
+ return [];
156
+ return [`## Acceptance-criterion tags ([agent] / [ci])`, ...lines, ``];
157
+ }
158
+ /**
159
+ * NOT-316: how the reviewer judges tagged criteria. Rendered only when the AC
160
+ * text carries the tag (per-tag lines) — untagged reviewer prompts are
161
+ * byte-for-byte what they always were. `[ci]` is judged by the PR check
162
+ * status in the evidence, never by local runs; an `[operator]` criterion is
163
+ * gated by Dealer and missing operator evidence is never a defect (the
164
+ * probe/doc existence check lives in `operatorReviewerSection` and is not
165
+ * duplicated here).
166
+ */
167
+ function tagSemanticsReviewerSection(acceptanceCriteria) {
168
+ const lines = [];
169
+ if (acceptanceCriteriaHasTag(acceptanceCriteria, "[ci]")) {
170
+ lines.push(`- Judge \`[ci]\` criteria by the PR check status in the evidence (the CI checks section), not by local runs.`);
171
+ }
172
+ if (acceptanceCriteriaHasTag(acceptanceCriteria, "[operator]")) {
173
+ lines.push(`- Do not mark an \`[operator]\` criterion as a defect — it is gated by Dealer.`);
174
+ }
175
+ if (!lines.length)
176
+ return [];
177
+ return [`## Criterion tags ([ci] / [operator]) — how to judge`, ...lines, ``];
178
+ }
70
179
  /** NOT-272: a resolved product_scope_decision's human note, verbatim. Empty/absent
71
180
  * renders nothing so noteless resolves produce byte-for-byte the prompt they always did. */
72
181
  function scopeDecisionSection(note) {
@@ -140,7 +249,18 @@ export function buildDeveloperPrompt(input) {
140
249
  parts.push(...conflictRepairSection(input.conflictRepair));
141
250
  // NOT-272: the scope decision leads — the developer reads it before the task itself.
142
251
  parts.push(...scopeDecisionSection(input.scopeDecisionNote));
143
- parts.push(`## Task`, input.taskSnapshot.title, input.taskSnapshot.description, ``, `## Acceptance criteria`, input.taskSnapshot.acceptanceCriteria, ``, ...executionContractSection(input.taskSnapshot.executionContract));
252
+ // NOT-314: operator repair feedback leads for the same reason.
253
+ parts.push(...operatorRepairSection(input.operatorRepairNote));
254
+ parts.push(`## Task`, input.taskSnapshot.title, input.taskSnapshot.description, ``, `## Acceptance criteria`, input.taskSnapshot.acceptanceCriteria, ``,
255
+ // NOT-314: operator criteria render separately right after the criteria —
256
+ // the worker must not attempt them, only ship the probe + doc.
257
+ ...operatorDeveloperSection(input.operatorCriteria),
258
+ // NOT-316: tag semantics render only for Muse developers whose AC text
259
+ // carries the tags — otherwise nothing, so untagged and non-Muse prompts
260
+ // stay byte-for-byte.
261
+ ...(input.museDeveloper
262
+ ? tagSemanticsDeveloperSection(input.taskSnapshot.acceptanceCriteria)
263
+ : []), ...executionContractSection(input.taskSnapshot.executionContract));
144
264
  if (input.findings?.length) {
145
265
  parts.push(`## Findings to address`);
146
266
  for (const f of input.findings) {
@@ -153,6 +273,10 @@ export function buildDeveloperPrompt(input) {
153
273
  parts.push(...agentDeckSection(input.worktreePath, input.deckId));
154
274
  if (input.museDeveloper)
155
275
  parts.push(...museCronProhibitionSection());
276
+ // NOT-316: sandbox limits rendered from MUSE_SANDBOX_CAPABILITIES — Muse
277
+ // developers only, so non-Muse prompts are untouched.
278
+ if (input.museDeveloper)
279
+ parts.push(...museEnvironmentLimitsSection());
156
280
  // NOT-303: the screenshot-path preflight verdict is embedded before the session
157
281
  // starts, so the worker reads it instead of discovering the Chrome.app abort
158
282
  // mid-session. Non-Muse developers are untouched.
@@ -160,6 +284,9 @@ export function buildDeveloperPrompt(input) {
160
284
  parts.push(...museVisualQaPromptSection(input.museVisualQa ?? checkMuseVisualQa()));
161
285
  }
162
286
  parts.push(`## Required`,
287
+ // NOT-315: the coordinator provisions dependencies before the session spawns —
288
+ // the sandbox has no network, so an in-session install can never work.
289
+ `Dependencies are installed; do not run npm install (no network).`,
163
290
  // NOT-115: encourage incremental commits so a mid-session death leaves less
164
291
  // uncommitted work for dirty_worktree escalation — soft mitigation only.
165
292
  // NOT-146: timed-spawn handoff bar is commits + targeted tests for the change.
@@ -251,6 +378,12 @@ export function buildReviewerPrompt(input) {
251
378
  `## Acceptance criteria`,
252
379
  input.taskSnapshot.acceptanceCriteria,
253
380
  ``,
381
+ // NOT-314: missing operator evidence is NOT a defect — but the probe and
382
+ // doc must exist. Only rendered when the snapshot carries operator criteria.
383
+ ...operatorReviewerSection(input.operatorCriteria),
384
+ // NOT-316: tagged-criterion judging lines — only when the AC text carries
385
+ // the tags, so untagged reviewer prompts stay byte-for-byte.
386
+ ...tagSemanticsReviewerSection(input.taskSnapshot.acceptanceCriteria),
254
387
  ...executionContractSection(input.taskSnapshot.executionContract),
255
388
  `## Diff (base ${input.baseSha.slice(0, 8)} → head ${input.headSha.slice(0, 8)})`,
256
389
  "```diff",
@@ -335,3 +335,151 @@ test("NOT-310: no conflict directive produces byte-for-byte the same prompt as b
335
335
  assert.equal(buildDeveloperPrompt({ ...base, conflictRepair: undefined }), without);
336
336
  assert.equal(buildDeveloperPrompt({ ...base, conflictRepair: { baseBranch: "", branch: "", files: [] } }), without);
337
337
  });
338
+ // NOT-314: the developer prompt lists operator criteria separately — the worker
339
+ // ships the probe + doc and never attempts the criterion itself.
340
+ test("NOT-314: developer prompt lists operator criteria with the do-not-attempt instruction", () => {
341
+ const prompt = buildDeveloperPrompt({
342
+ taskSnapshot,
343
+ round: 1,
344
+ operatorCriteria: [
345
+ {
346
+ text: "Operator can sign in with SSO and see the org dashboard [operator]",
347
+ commands: ["`npm run probe:sso -- --env staging`"],
348
+ },
349
+ ],
350
+ });
351
+ assert.match(prompt, /## Operator verification required/);
352
+ assert.match(prompt, /Do not attempt these; ship the ready-to-run probe and a doc, put the exact command in the PR body/);
353
+ assert.match(prompt, /sign in with SSO/);
354
+ assert.match(prompt, /npm run probe:sso -- --env staging/);
355
+ });
356
+ test("NOT-314: no operator criteria produces byte-for-byte the same developer prompt as before", () => {
357
+ const base = { taskSnapshot, round: 2 };
358
+ const without = buildDeveloperPrompt(base);
359
+ assert.doesNotMatch(without, /## Operator verification/);
360
+ assert.equal(buildDeveloperPrompt({ ...base, operatorCriteria: undefined }), without);
361
+ assert.equal(buildDeveloperPrompt({ ...base, operatorCriteria: [] }), without);
362
+ });
363
+ test("NOT-314: operator repair note renders verbatim under its own heading before the task", () => {
364
+ const note = "The probe 404s — its path moved; fix the script and the doc.";
365
+ const prompt = buildDeveloperPrompt({ taskSnapshot, round: 2, operatorRepairNote: note });
366
+ assert.match(prompt, /## Operator verification feedback/);
367
+ assert.ok(prompt.includes(note), "note text must appear verbatim");
368
+ assert.ok(prompt.indexOf("## Operator verification feedback") < prompt.indexOf("## Task"), "repair feedback must precede the task");
369
+ const without = buildDeveloperPrompt({ taskSnapshot, round: 2 });
370
+ assert.doesNotMatch(without, /## Operator verification feedback/);
371
+ assert.equal(buildDeveloperPrompt({ taskSnapshot, round: 2, operatorRepairNote: " " }), without);
372
+ });
373
+ // NOT-314: the reviewer must not flag missing operator evidence — Dealer gates
374
+ // the merge — but the probe and doc must exist.
375
+ test("NOT-314: reviewer prompt states missing operator evidence is not a defect, probe and doc must exist", () => {
376
+ const prompt = buildReviewerPrompt({
377
+ ...reviewerBase,
378
+ operatorCriteria: [
379
+ {
380
+ text: "Operator can complete a paid checkout [operator]",
381
+ commands: ["`npm run probe:checkout -- --env staging`"],
382
+ },
383
+ ],
384
+ });
385
+ assert.match(prompt, /## Operator-gated criteria/);
386
+ assert.match(prompt, /Missing operator evidence is NOT a defect/);
387
+ assert.match(prompt, /never raise a blocking finding/);
388
+ assert.match(prompt, /probe and doc/);
389
+ assert.match(prompt, /A missing probe or doc IS a blocking finding/);
390
+ assert.match(prompt, /paid checkout/);
391
+ assert.match(prompt, /npm run probe:checkout/);
392
+ });
393
+ test("NOT-314: no operator criteria produces byte-for-byte the same reviewer prompt as before", () => {
394
+ const without = buildReviewerPrompt(reviewerBase);
395
+ assert.doesNotMatch(without, /Operator-gated criteria/);
396
+ assert.equal(buildReviewerPrompt({ ...reviewerBase, operatorCriteria: undefined }), without);
397
+ assert.equal(buildReviewerPrompt({ ...reviewerBase, operatorCriteria: [] }), without);
398
+ });
399
+ // NOT-315: dependencies arrive installed (coordinator-side `npm ci`); the sandbox has
400
+ // no network, so the worker must never try to install them itself.
401
+ test("NOT-315: developer prompt tells the worker dependencies are installed and not to run npm install", () => {
402
+ for (const round of [1, 2]) {
403
+ const prompt = buildDeveloperPrompt({ taskSnapshot, round });
404
+ assert.match(prompt, /Dependencies are installed; do not run npm install \(no network\)\./);
405
+ }
406
+ const retry = buildDeveloperPrompt({ taskSnapshot, round: 1, retryReason: "Developer session failed or crashed." });
407
+ assert.match(retry, /Dependencies are installed; do not run npm install \(no network\)\./);
408
+ });
409
+ // NOT-316: a Muse developer learns the sandbox limits from its prompt — the
410
+ // section is rendered from MUSE_SANDBOX_CAPABILITIES, not hand-written prose.
411
+ // Other runtimes are unchanged.
412
+ test("NOT-316: museDeveloper prompt contains the Environment limits section; other runtimes do not", async () => {
413
+ const { MUSE_SANDBOX_CAPABILITIES } = await import("../runners/muse-code-args.js");
414
+ const muse = buildDeveloperPrompt({ taskSnapshot, round: 1, museDeveloper: true });
415
+ assert.match(muse, /## Environment limits/);
416
+ assert.ok(muse.includes(`sandbox network: ${MUSE_SANDBOX_CAPABILITIES.sandboxNetwork}`), "section header names the sandbox-network value from the constant");
417
+ assert.ok(muse.includes(`Network: ${MUSE_SANDBOX_CAPABILITIES.network}`), "network limit comes from the constant");
418
+ assert.match(muse, /listen.*EPERM/i);
419
+ assert.match(muse, /Browser: no/);
420
+ assert.match(muse, /Credentials\/Keychain: no/);
421
+ assert.match(muse, /After one failed attempt.*stop and hand off; do not work around the sandbox\./);
422
+ const other = buildDeveloperPrompt({ taskSnapshot, round: 1 });
423
+ assert.doesNotMatch(other, /## Environment limits/);
424
+ assert.doesNotMatch(other, /do not work around the sandbox/);
425
+ });
426
+ // NOT-316: tag semantics appear only when the AC text carries the tags, per
427
+ // tag, and only for Muse developers — an untagged ticket's prompt carries no
428
+ // tag section at all.
429
+ test("NOT-316: developer tag semantics render per tag only for a Muse developer with tagged ACs", () => {
430
+ const taggedBoth = {
431
+ ...taskSnapshot,
432
+ acceptanceCriteria: "- [ ] [agent] Widget renders in the preview.\n- [ ] [ci] Extend the `verify` workflow for widgets.",
433
+ };
434
+ const museBoth = buildDeveloperPrompt({ taskSnapshot: taggedBoth, round: 1, museDeveloper: true });
435
+ assert.match(museBoth, /## Acceptance-criterion tags/);
436
+ assert.match(museBoth, /`\[agent\]`: verify yourself/);
437
+ assert.match(museBoth, /`\[ci\]`: implement or extend the named CI job\/test and push/);
438
+ assert.match(museBoth, /CI is the evidence, do not reproduce it locally/);
439
+ const agentOnly = buildDeveloperPrompt({
440
+ taskSnapshot: { ...taskSnapshot, acceptanceCriteria: "- [ ] [agent] Widget renders." },
441
+ round: 1,
442
+ museDeveloper: true,
443
+ });
444
+ assert.match(agentOnly, /`\[agent\]`: verify yourself/);
445
+ assert.doesNotMatch(agentOnly, /`\[ci\]`: implement or extend/);
446
+ const ciOnly = buildDeveloperPrompt({
447
+ taskSnapshot: { ...taskSnapshot, acceptanceCriteria: "- [ ] [ci] Extend the `verify` workflow." },
448
+ round: 1,
449
+ museDeveloper: true,
450
+ });
451
+ assert.match(ciOnly, /`\[ci\]`: implement or extend the named CI job\/test and push/);
452
+ assert.doesNotMatch(ciOnly, /`\[agent\]`: verify yourself/);
453
+ // Untagged Muse prompt: no tag section — snapshot-style byte check that the
454
+ // tag feature adds nothing beyond the section itself.
455
+ const untaggedMuse = buildDeveloperPrompt({ taskSnapshot, round: 1, museDeveloper: true });
456
+ assert.doesNotMatch(untaggedMuse, /## Acceptance-criterion tags/);
457
+ assert.doesNotMatch(untaggedMuse, /`\[agent\]`: verify yourself/);
458
+ assert.doesNotMatch(untaggedMuse, /`\[ci\]`: implement or extend/);
459
+ assert.equal(buildDeveloperPrompt({ taskSnapshot, round: 1, museDeveloper: true }), untaggedMuse);
460
+ // Non-Muse prompts never carry tag semantics, even with tagged ACs.
461
+ const otherTagged = buildDeveloperPrompt({ taskSnapshot: taggedBoth, round: 1 });
462
+ assert.doesNotMatch(otherTagged, /## Acceptance-criterion tags/);
463
+ assert.doesNotMatch(otherTagged, /## Environment limits/);
464
+ });
465
+ // NOT-316: the reviewer judges [ci] by the PR check status and never flags a
466
+ // missing [operator] result as a defect.
467
+ test("NOT-316: reviewer prompt judges [ci] by check status and never defects [operator]", () => {
468
+ const tagged = buildReviewerPrompt({
469
+ ...reviewerBase,
470
+ taskSnapshot: {
471
+ ...taskSnapshot,
472
+ acceptanceCriteria: "- [ ] [ci] The `verify` workflow covers widgets.\n- [ ] [operator] Operator completes a paid checkout.",
473
+ },
474
+ });
475
+ assert.match(tagged, /## Criterion tags/);
476
+ assert.match(tagged, /Judge `\[ci\]` criteria by the PR check status in the evidence/);
477
+ assert.match(tagged, /not by local runs/);
478
+ assert.match(tagged, /Do not mark an `\[operator\]` criterion as a defect/);
479
+ assert.match(tagged, /gated by Dealer/);
480
+ });
481
+ test("NOT-316: untagged reviewer prompt carries no tag-judging section", () => {
482
+ const without = buildReviewerPrompt(reviewerBase);
483
+ assert.doesNotMatch(without, /## Criterion tags/);
484
+ assert.equal(buildReviewerPrompt(reviewerBase), without);
485
+ });
@@ -26,23 +26,72 @@
26
26
  // effect; and one item that fails to route never rolls back the others.
27
27
  import { canTransitionIssue } from "@agent-dealer/shared";
28
28
  import { getDb } from "../db/index.js";
29
- import { getIssue, incrementIssueInfraAttempts, transitionIssue } from "../repository/issues.js";
29
+ import { getIssue, incrementIssueInfraAttempts, listIssues, transitionIssue } from "../repository/issues.js";
30
30
  import { appendWorkflowEvent, getActiveWorkflowInstance } from "../repository/workflow-events.js";
31
31
  import { completeSession, getWorkerSession } from "../repository/worker-sessions.js";
32
32
  import { runtimeAvailability } from "../repository/runtime-availability.js";
33
- import { deferWorkItem, finishWorkItem, listExpiredLeases, refreshHeartbeat, requeueWorkItem, } from "../repository/work-items.js";
33
+ import { deferWorkItem, enqueueWorkItem, finishWorkItem, listExpiredLeases, listWorkItemsForIssue, refreshHeartbeat, requeueWorkItem, } from "../repository/work-items.js";
34
34
  import { infraAttemptsRemain } from "./routing.js";
35
35
  import { activeClockJumpGrace } from "./clock-jump.js";
36
36
  import { inspectWorkerProcess, terminateWorkerProcess } from "./process-liveness.js";
37
37
  import { maxAliveHoldMsFor } from "./session-timeouts.js";
38
38
  import { emitHostSuspended } from "./agent-boundaries.js";
39
- import { routeAppliedOutcome } from "./commands.js";
39
+ import { queuedProfileSnapshot, routeAppliedOutcome } from "./commands.js";
40
40
  import { recoverStrandedAutoMerges } from "./auto-merge.js";
41
41
  import { workerSessionPayload } from "./session-progress.js";
42
42
  import { PRESUMED_DEAD_REASON, presumedDeadReclaimReason } from "./failure-reason.js";
43
43
  import { recordCausesForWorkerFailedEvent } from "./failure-cause.js";
44
44
  import { baseRefCandidates, developerBranchName, hasPublishableWork, inspectBranchProgress, } from "./branch-progress.js";
45
45
  const num = (name, dflt) => Number(process.env[name] ?? dflt);
46
+ /**
47
+ * NOT-333: an issue parked in `reviewing` with an open workflow instance but no
48
+ * pending/leased work item is stranded — nothing will ever claim it (the NOT-200
49
+ * shape: a resume round that pushed nothing deduped its reviewer enqueue against the
50
+ * already-terminal reviewer row). Re-enqueue one reviewer round at the current head.
51
+ *
52
+ * Bounded by construction: the fresh pending row makes the very next pass a no-op, so
53
+ * this fires at most once per stranding. Item history is required — reaching
54
+ * `reviewing` always enqueues a reviewer, so a bare row with no work items at all has
55
+ * nothing to resume and is left alone.
56
+ */
57
+ export function recoverStrandedReviewing() {
58
+ const recovered = [];
59
+ for (const issue of listIssues("reviewing")) {
60
+ const instance = getActiveWorkflowInstance(issue.id);
61
+ if (!instance)
62
+ continue;
63
+ const items = listWorkItemsForIssue(issue.id);
64
+ if (items.length === 0)
65
+ continue;
66
+ if (items.some((w) => w.status === "pending" || w.status === "leased"))
67
+ continue;
68
+ const headSha = issue.headSha ?? null;
69
+ const baseKey = `${instance.id}:reviewer:${issue.currentRound}${headSha ? `:${headSha}` : ""}:stranded-recovery`;
70
+ const taken = new Set(items.map((w) => w.idempotencyKey));
71
+ let key = baseKey;
72
+ for (let n = 2; taken.has(key); n++)
73
+ key = `${baseKey}:${n}`;
74
+ console.warn("[coordinator] issue stranded in reviewing with no active work — re-enqueueing reviewer", {
75
+ issueId: issue.id,
76
+ workflowInstanceId: instance.id,
77
+ round: issue.currentRound,
78
+ headSha,
79
+ });
80
+ const next = enqueueWorkItem({
81
+ issueId: issue.id,
82
+ workflowInstanceId: instance.id,
83
+ kind: "reviewer",
84
+ round: issue.currentRound,
85
+ payload: {
86
+ ...(headSha ? { inputSha: headSha } : {}),
87
+ profileSnapshot: queuedProfileSnapshot(issue, "reviewer"),
88
+ },
89
+ idempotencyKey: key,
90
+ });
91
+ recovered.push(next.id);
92
+ }
93
+ return recovered;
94
+ }
46
95
  /** Fail a worker_session still `running` for an item whose worker is gone. */
47
96
  function failOrphanSession(workerSessionId) {
48
97
  if (!workerSessionId)
@@ -489,6 +538,10 @@ export async function recoverCoordinator(opts) {
489
538
  console.warn(`[coordinator] ${republished.length} reclaim(s) routed to republish — branch already had commits, no new agent session`, { workItemIds: republished });
490
539
  }
491
540
  const stranded = await recoverStrandedAutoMerges();
541
+ // NOT-333: re-enqueue a reviewer for issues stranded in `reviewing` with no active
542
+ // work (runs every tick alongside the lease reclaim above — the fresh pending row
543
+ // makes the next pass a no-op, so this recovers exactly once per stranding).
544
+ const strandedReviewingRecovered = recoverStrandedReviewing();
492
545
  return {
493
546
  reclaimed,
494
547
  republished,
@@ -499,5 +552,6 @@ export async function recoverCoordinator(opts) {
499
552
  unverifiedOrphans,
500
553
  heldAcrossClockJump,
501
554
  autoMergesFinalized: stranded.finalized,
555
+ strandedReviewingRecovered,
502
556
  };
503
557
  }
@@ -9,10 +9,11 @@ process.env.MAX_COORDINATOR_CONCURRENCY = "2";
9
9
  process.env.COORDINATOR_FAIL_BACKOFF_MS = "0";
10
10
  const { migrate, getDb } = await import("../db/index.js");
11
11
  const { BUILTIN_AGENT_CLAUDE_ID, BUILTIN_AGENT_CURSOR_ID } = await import("@agent-dealer/shared");
12
- const { createIssue, getIssue } = await import("../repository/issues.js");
12
+ const { createIssue, getIssue, transitionIssue } = await import("../repository/issues.js");
13
+ const { getActiveWorkflowInstance } = await import("../repository/workflow-events.js");
13
14
  const { listHumanActionsForIssue } = await import("../repository/human-actions.js");
14
15
  const { listWorkerSessionsForIssue, createWorkerSession, startSession } = await import("../repository/worker-sessions.js");
15
- const { getWorkItem, listWorkItemsForIssue, claimWorkItem, bindWorkItemSession, finishWorkItem, refreshHeartbeat } = await import("../repository/work-items.js");
16
+ const { getWorkItem, listWorkItemsForIssue, claimWorkItem, bindWorkItemSession, enqueueWorkItem, finishWorkItem, refreshHeartbeat } = await import("../repository/work-items.js");
16
17
  const { startWorkflow } = await import("./commands.js");
17
18
  const { recoverCoordinator } = await import("./recovery.js");
18
19
  const FUTURE = () => Date.now() + 3_600_000; // a clock well past any test lease
@@ -126,7 +127,7 @@ test("recovery ignores a lease that has not expired", async () => {
126
127
  const issueId = newIssue();
127
128
  startWorkflow(issueId);
128
129
  claimWorkItem("healthy", { leaseMs: 600_000 });
129
- assert.deepEqual(await recoverCoordinator({ now: Date.now() }), { reclaimed: [], republished: [], deadLettered: [], deferredForCap: [], heldAlive: [], heldAliveExpired: [], unverifiedOrphans: [], heldAcrossClockJump: [], autoMergesFinalized: [] });
130
+ assert.deepEqual(await recoverCoordinator({ now: Date.now() }), { reclaimed: [], republished: [], deadLettered: [], deferredForCap: [], heldAlive: [], heldAliveExpired: [], unverifiedOrphans: [], heldAcrossClockJump: [], autoMergesFinalized: [], strandedReviewingRecovered: [] });
130
131
  });
131
132
  test("recovery leaves a lease alone when a heartbeat renewed it after the snapshot", async () => {
132
133
  // Regression for the reviewer's "recovery can reclaim a freshly renewed lease".
@@ -138,7 +139,7 @@ test("recovery leaves a lease alone when a heartbeat renewed it after the snapsh
138
139
  // …but before its CAS runs, the worker heartbeats and renews the lease:
139
140
  refreshHeartbeat(devItem.id, claimed.leaseToken, { leaseMs: 600_000 });
140
141
  const res = await recoverCoordinator({ now: Date.now() + 1_000 });
141
- assert.deepEqual(res, { reclaimed: [], republished: [], deadLettered: [], deferredForCap: [], heldAlive: [], heldAliveExpired: [], unverifiedOrphans: [], heldAcrossClockJump: [], autoMergesFinalized: [] });
142
+ assert.deepEqual(res, { reclaimed: [], republished: [], deadLettered: [], deferredForCap: [], heldAlive: [], heldAliveExpired: [], unverifiedOrphans: [], heldAcrossClockJump: [], autoMergesFinalized: [], strandedReviewingRecovered: [] });
142
143
  assert.equal(getWorkItem(devItem.id).status, "leased");
143
144
  });
144
145
  test("an expired lease past the infra-attempt limit is dead-lettered AND routed in one step", async () => {
@@ -162,7 +163,7 @@ test("an expired lease past the infra-attempt limit is dead-lettered AND routed
162
163
  const last = JSON.parse(failed[failed.length - 1].payloadJson);
163
164
  assert.match(last.reason ?? "", /presumed dead/);
164
165
  // A second recovery pass is a no-op — the item is already dead, nothing to reclaim.
165
- assert.deepEqual(await recoverCoordinator({ now: FUTURE() }), { reclaimed: [], republished: [], deadLettered: [], deferredForCap: [], heldAlive: [], heldAliveExpired: [], unverifiedOrphans: [], heldAcrossClockJump: [], autoMergesFinalized: [] });
166
+ assert.deepEqual(await recoverCoordinator({ now: FUTURE() }), { reclaimed: [], republished: [], deadLettered: [], deferredForCap: [], heldAlive: [], heldAliveExpired: [], unverifiedOrphans: [], heldAcrossClockJump: [], autoMergesFinalized: [], strandedReviewingRecovered: [] });
166
167
  assert.equal(listHumanActionsForIssue(issueId).filter((a) => a.status === "open").length, 1, "no duplicate human action from a second recovery");
167
168
  });
168
169
  test("recovery loses its CAS to a worker that completed concurrently", async () => {
@@ -172,7 +173,7 @@ test("recovery loses its CAS to a worker that completed concurrently", async ()
172
173
  const claimed = claimWorkItem("o", { leaseMs: 1 });
173
174
  // Worker finishes just before recovery's transaction runs.
174
175
  finishWorkItem(devItem.id, claimed.leaseToken, { status: "done", result: { kind: "no_pr" } });
175
- assert.deepEqual(await recoverCoordinator({ now: FUTURE() }), { reclaimed: [], republished: [], deadLettered: [], deferredForCap: [], heldAlive: [], heldAliveExpired: [], unverifiedOrphans: [], heldAcrossClockJump: [], autoMergesFinalized: [] });
176
+ assert.deepEqual(await recoverCoordinator({ now: FUTURE() }), { reclaimed: [], republished: [], deadLettered: [], deferredForCap: [], heldAlive: [], heldAliveExpired: [], unverifiedOrphans: [], heldAcrossClockJump: [], autoMergesFinalized: [], strandedReviewingRecovered: [] });
176
177
  assert.equal(getWorkItem(devItem.id).status, "done");
177
178
  });
178
179
  test("an expired lease on a usage-capped runtime is deferred, not dead-lettered (NOT-111 recovery gap)", async () => {
@@ -210,7 +211,8 @@ test("an expired lease on a usage-capped runtime is deferred, not dead-lettered
210
211
  heldAliveExpired: [],
211
212
  unverifiedOrphans: [],
212
213
  heldAcrossClockJump: [],
213
- autoMergesFinalized: []
214
+ autoMergesFinalized: [],
215
+ strandedReviewingRecovered: []
214
216
  });
215
217
  assert.equal(getWorkItem(devItem.id).status, "pending");
216
218
  assert.equal(getWorkItem(devItem.id).attemptCount, 0, "claim's attempt bump is reverted, like a live cap deferral");
@@ -223,5 +225,62 @@ test("an expired lease on a usage-capped runtime is deferred, not dead-lettered
223
225
  }
224
226
  });
225
227
  test("recoverCoordinator is a no-op on a clean queue", async () => {
226
- assert.deepEqual(await recoverCoordinator({ now: FUTURE() }), { reclaimed: [], republished: [], deadLettered: [], deferredForCap: [], heldAlive: [], heldAliveExpired: [], unverifiedOrphans: [], heldAcrossClockJump: [], autoMergesFinalized: [] });
228
+ assert.deepEqual(await recoverCoordinator({ now: FUTURE() }), { reclaimed: [], republished: [], deadLettered: [], deferredForCap: [], heldAlive: [], heldAliveExpired: [], unverifiedOrphans: [], heldAcrossClockJump: [], autoMergesFinalized: [], strandedReviewingRecovered: [] });
229
+ });
230
+ /** Settles an issue's developer round so the test can park the issue in `reviewing`. */
231
+ function settleDeveloper(issueId) {
232
+ const dev = claimWorkItem(`seed-${issueId}`, { leaseMs: 600_000 });
233
+ assert.equal(dev.issueId, issueId);
234
+ assert.equal(finishWorkItem(dev.id, dev.leaseToken, { status: "done", result: { kind: "no_pr" } }).status, "done");
235
+ }
236
+ test("NOT-333: recovery re-enqueues exactly once for an issue stranded in reviewing, and is a no-op otherwise", async () => {
237
+ // The NOT-200 stranding: reviewing, open instance, head pinned, but every work
238
+ // item terminal — the resume round's reviewer enqueue deduped against the done row.
239
+ const strandedId = newIssue();
240
+ startWorkflow(strandedId);
241
+ settleDeveloper(strandedId);
242
+ transitionIssue(strandedId, "reviewing", { headSha: "abc123" });
243
+ const instance = getActiveWorkflowInstance(strandedId);
244
+ const oldReviewer = enqueueWorkItem({
245
+ issueId: strandedId,
246
+ workflowInstanceId: instance.id,
247
+ kind: "reviewer",
248
+ round: 1,
249
+ payload: { inputSha: "abc123" },
250
+ idempotencyKey: `${instance.id}:reviewer:1:abc123`,
251
+ });
252
+ const leasedOld = claimWorkItem(`seed-rev-${strandedId}`, { leaseMs: 600_000 });
253
+ assert.equal(leasedOld.id, oldReviewer.id);
254
+ assert.equal(finishWorkItem(oldReviewer.id, leasedOld.leaseToken, { status: "done", result: { kind: "verdict" } }).status, "done");
255
+ // A healthy issue: reviewing WITH a pending reviewer must be left alone.
256
+ const healthyId = newIssue();
257
+ startWorkflow(healthyId);
258
+ settleDeveloper(healthyId);
259
+ transitionIssue(healthyId, "reviewing", { headSha: "def456" });
260
+ const healthyInstance = getActiveWorkflowInstance(healthyId);
261
+ const healthyReviewer = enqueueWorkItem({
262
+ issueId: healthyId,
263
+ workflowInstanceId: healthyInstance.id,
264
+ kind: "reviewer",
265
+ round: 1,
266
+ payload: { inputSha: "def456" },
267
+ idempotencyKey: `${healthyInstance.id}:reviewer:1:def456`,
268
+ });
269
+ // First tick recovers the stranded issue exactly once.
270
+ const first = await recoverCoordinator({ now: Date.now() });
271
+ assert.equal(first.strandedReviewingRecovered.length, 1);
272
+ const recovered = getWorkItem(first.strandedReviewingRecovered[0]);
273
+ assert.equal(recovered.issueId, strandedId);
274
+ assert.equal(recovered.kind, "reviewer");
275
+ assert.equal(recovered.status, "pending");
276
+ assert.notEqual(recovered.id, oldReviewer.id);
277
+ assert.equal(JSON.parse(recovered.payloadJson).inputSha, "abc123");
278
+ // ...and touches nothing else.
279
+ assert.equal(listWorkItemsForIssue(healthyId).length, 2);
280
+ assert.equal(getWorkItem(healthyReviewer.id).status, "pending");
281
+ // Second tick is a no-op — the fresh pending row bounds the guard, no loop.
282
+ const second = await recoverCoordinator({ now: Date.now() });
283
+ assert.deepEqual(second.strandedReviewingRecovered, []);
284
+ assert.equal(listWorkItemsForIssue(strandedId).filter((i) => i.kind === "reviewer").length, 2);
285
+ assert.equal(listWorkItemsForIssue(strandedId).filter((i) => i.kind === "reviewer" && i.status === "pending").length, 1);
227
286
  });
@@ -197,6 +197,7 @@ test("developer round 1 → reviewer changes_requested → developer round 2 on
197
197
  assert.equal(issue.status, "final_review", "the cycle must end at a human final_review, not stuck or escalated");
198
198
  assert.equal(issue.currentRound, 2, "exactly one genuine repair round — changes_requested — was spent");
199
199
  assert.equal(issue.infraAttempts, 0, "nothing in this run was an infra failure");
200
+ assert.equal(issue.ciAttempts, 0, "nothing in this run was a CI failure either");
200
201
  assert.ok(listHumanActionsForIssue(issueId).find((a) => a.actionType === "final_review" && a.status === "open"));
201
202
  // Same branch reused across both rounds — never re-created.
202
203
  const branch = issueBranchName(issueId);
@@ -26,6 +26,7 @@ import os from "node:os";
26
26
  import path from "node:path";
27
27
  import { parseProfileSnapshot, roleCeiling } from "@agent-dealer/shared";
28
28
  import { getTaskSnapshot } from "./commands.js";
29
+ import { extractOperatorCriteria } from "./operator-criteria.js";
29
30
  import { buildReviewerPrompt, formatDiffForPrompt, TOTAL_DIFF_LIMIT } from "./prompts.js";
30
31
  import { guidanceForNextSession } from "./guidance.js";
31
32
  import { realReviewerSpawn, reviewerSessionLogPath } from "./spawn.js";
@@ -336,6 +337,10 @@ export async function runReviewerEffect(ctx, deps = defaultDeps) {
336
337
  const { truncated: diffTruncated, omittedPaths } = formatDiffForPrompt(diff);
337
338
  const openFindings = listFindingsForIssue(issue.id).filter((f) => f.status === "open" || f.status === "recurring");
338
339
  const guidance = guidanceForNextSession(issue.id, sessionId);
340
+ // NOT-314: the frozen snapshot's operator criteria — the reviewer must not
341
+ // treat missing operator evidence as a defect (Dealer gates the merge), but
342
+ // the probe and doc must exist. Rendered only when non-empty.
343
+ const operatorCriteria = extractOperatorCriteria(taskSnapshot.acceptanceCriteria);
339
344
  const prompt = buildReviewerPrompt({
340
345
  taskSnapshot,
341
346
  round: workItem.round,
@@ -348,6 +353,7 @@ export async function runReviewerEffect(ctx, deps = defaultDeps) {
348
353
  worktreePath,
349
354
  deckId: snapshot?.deckId ?? null,
350
355
  guidance: guidance.length ? guidance : undefined,
356
+ operatorCriteria: operatorCriteria.length > 0 ? operatorCriteria : undefined,
351
357
  });
352
358
  // NOT-83 review finding — see developer-effect.ts's identical check for the full
353
359
  // rationale: re-verify this item is still leased right before the real spawn, since an