agent-dealer 1.0.0 → 1.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/bundle/server/dist/adapters/linear-graphql.js +165 -0
  2. package/bundle/server/dist/adapters/linear-graphql.test.js +129 -0
  3. package/bundle/server/dist/adapters/linear-inbox.js +12 -21
  4. package/bundle/server/dist/adapters/linear-sync.js +6 -19
  5. package/bundle/server/dist/adapters/managed-repo.js +11 -0
  6. package/bundle/server/dist/coordinator/auto-merge.integration.test.js +55 -4
  7. package/bundle/server/dist/coordinator/auto-merge.js +72 -4
  8. package/bundle/server/dist/coordinator/auto-merge.timeout.test.js +45 -3
  9. package/bundle/server/dist/coordinator/commands.js +17 -11
  10. package/bundle/server/dist/coordinator/commands.test.js +21 -4
  11. package/bundle/server/dist/coordinator/dependencies.js +5 -1
  12. package/bundle/server/dist/coordinator/dependency-readiness.integration.test.js +33 -1
  13. package/bundle/server/dist/coordinator/human-resolution.js +5 -2
  14. package/bundle/server/dist/coordinator/human-resolution.test.js +13 -1
  15. package/bundle/server/dist/coordinator/prompts.js +21 -13
  16. package/bundle/server/dist/coordinator/prompts.test.js +21 -0
  17. package/bundle/server/dist/coordinator/reviewer-effect.js +56 -11
  18. package/bundle/server/dist/coordinator/reviewer-effect.test.js +67 -9
  19. package/bundle/server/dist/coordinator/reviewer-result.js +49 -1
  20. package/bundle/server/dist/coordinator/reviewer-result.test.js +38 -0
  21. package/bundle/server/dist/coordinator/routing.js +15 -4
  22. package/bundle/server/dist/coordinator/routing.test.js +17 -2
  23. package/bundle/server/dist/coordinator/worker-loop.test.js +2 -0
  24. package/bundle/server/dist/db/index.js +6 -0
  25. package/bundle/server/dist/db/migrate-to-issues.test.js +1 -0
  26. package/bundle/server/dist/dev-review-cli-happy-path.integration.test.js +3 -2
  27. package/bundle/server/dist/direct-start-interrupt-probe.js +58 -0
  28. package/bundle/server/dist/direct-start-liveness.integration.test.js +114 -24
  29. package/bundle/server/dist/direct-start-temp-home-cleanup.js +124 -0
  30. package/bundle/server/dist/direct-start-temp-home-cleanup.test.js +92 -0
  31. package/bundle/server/dist/repository/queue-entries.js +49 -11
  32. package/bundle/server/dist/routes/human-actions.test.js +2 -0
  33. package/bundle/server/dist/routes/queue-reorder.integration.test.js +144 -0
  34. package/bundle/server/dist/routes/queue.js +21 -2
  35. package/bundle/server/package.json +2 -2
  36. package/bundle/server/static-ui/assets/index-0kT1vk6L.js +60 -0
  37. package/bundle/server/static-ui/assets/{index-BxTx4b1d.css → index-hXICi1rX.css} +1 -1
  38. package/bundle/server/static-ui/index.html +2 -2
  39. package/bundle/shared/dist/queue-entries.d.ts +47 -0
  40. package/bundle/shared/dist/queue-entries.js +14 -0
  41. package/bundle/shared/package.json +1 -1
  42. package/dist/action.test.js +2 -2
  43. package/dist/index.js +1 -0
  44. package/dist/install.js +1 -1
  45. package/dist/lifecycle.contract.test.js +2 -2
  46. package/dist/queue.d.ts +5 -0
  47. package/dist/queue.js +50 -0
  48. package/dist/queue.test.js +39 -0
  49. package/dist/setup.js +5 -1
  50. package/package.json +1 -1
  51. package/bundle/server/static-ui/assets/index-BTtPGYpY.js +0 -60
@@ -1,7 +1,17 @@
1
- // Unit coverage for bounded gh timeout classification (NOT-102 review finding).
2
- import { test } from "node:test";
1
+ // Unit coverage for bounded gh timeout classification (NOT-102) and NOT-151 spawn ENOENT mapping.
2
+ import { test, before } from "node:test";
3
3
  import assert from "node:assert/strict";
4
- import { GH_MERGE_TIMEOUT_MS, ghErrorReason, isGhTimeoutError, } from "./auto-merge.js";
4
+ import fs from "node:fs";
5
+ import os from "node:os";
6
+ import path from "node:path";
7
+ import { execFileSync } from "node:child_process";
8
+ process.env.AGENT_DEALER_HOME = fs.mkdtempSync(path.join(os.tmpdir(), "dealer-not151-unit-"));
9
+ const { GH_MERGE_TIMEOUT_MS, ghErrorReason, ghSpawnEnoentReason, isGhTimeoutError, resolveAutoMergeCwd, } = await import("./auto-merge.js");
10
+ const { managedRepoPath } = await import("../adapters/managed-repo.js");
11
+ before(async () => {
12
+ const { migrate } = await import("../db/index.js");
13
+ migrate();
14
+ });
5
15
  test("isGhTimeoutError detects killed / SIGTERM from execFile timeout", () => {
6
16
  assert.equal(isGhTimeoutError({ killed: true }), true);
7
17
  assert.equal(isGhTimeoutError({ signal: "SIGTERM" }), true);
@@ -12,3 +22,35 @@ test("ghErrorReason maps timeout before stderr", () => {
12
22
  assert.equal(ghErrorReason({ stderr: " checks failed \n" }, "fallback"), "checks failed");
13
23
  assert.equal(ghErrorReason({}, "gh pr merge failed"), "gh pr merge failed");
14
24
  });
25
+ test("NOT-151: ghSpawnEnoentReason distinguishes bad cwd from missing gh", () => {
26
+ const missing = path.join(os.tmpdir(), "dealer-missing-merge-cwd-xyz");
27
+ assert.match(ghSpawnEnoentReason({ code: "ENOENT", message: "spawn gh ENOENT" }, missing) ?? "", /invalid merge cwd/);
28
+ const existing = fs.mkdtempSync(path.join(os.tmpdir(), "dealer-merge-cwd-"));
29
+ assert.match(ghSpawnEnoentReason({ code: "ENOENT", message: "spawn gh ENOENT" }, existing) ?? "", /gh not on PATH/);
30
+ assert.equal(ghSpawnEnoentReason({ stderr: "checks failed" }, existing), null);
31
+ });
32
+ test("NOT-151: resolveAutoMergeCwd maps portable identity to managed path when clone exists", () => {
33
+ const identity = "github.com/not-so-fat/agent-dealer";
34
+ const managed = managedRepoPath(identity);
35
+ fs.mkdirSync(path.join(managed, ".git"), { recursive: true });
36
+ const ok = resolveAutoMergeCwd(identity);
37
+ assert.equal(ok.ok, true);
38
+ if (ok.ok)
39
+ assert.equal(ok.cwd, managed);
40
+ });
41
+ test("NOT-151: resolveAutoMergeCwd fails closed when managed clone is missing", () => {
42
+ const missing = resolveAutoMergeCwd("github.com/missing/not-cloned");
43
+ assert.equal(missing.ok, false);
44
+ if (!missing.ok) {
45
+ assert.match(missing.reason, /Managed clone missing/);
46
+ assert.doesNotMatch(missing.reason, /ENOENT/);
47
+ }
48
+ });
49
+ test("NOT-151: resolveAutoMergeCwd accepts a real legacy local checkout", () => {
50
+ const dir = fs.mkdtempSync(path.join(os.tmpdir(), "legacy-merge-cwd-"));
51
+ execFileSync("git", ["init", "-b", "main"], { cwd: dir });
52
+ const ok = resolveAutoMergeCwd(dir);
53
+ assert.equal(ok.ok, true);
54
+ if (ok.ok)
55
+ assert.equal(ok.cwd, dir);
56
+ });
@@ -3,6 +3,7 @@ import { getIssue, incrementIssueRound, incrementIssueInfraAttempts, resetIssueI
3
3
  import { appendWorkflowEvent, completeWorkflowInstance, getActiveWorkflowInstance, getWorkflowInstance, startWorkflowInstance, WorkflowAlreadyActiveError, } from "../repository/workflow-events.js";
4
4
  import { createHumanAction, findOpenHumanAction, getHumanAction, listHumanActionsForIssue, resolveHumanAction, } from "../repository/human-actions.js";
5
5
  import { reconcileFinding } from "../repository/findings.js";
6
+ import { normalizeReviewerResult } from "./reviewer-result.js";
6
7
  import { getAgent } from "../repository/agents.js";
7
8
  import { githubIssuesSync } from "../adapters/agent-health.js";
8
9
  import { createIssueArtifact, latestIssueArtifact } from "../repository/artifacts.js";
@@ -499,12 +500,14 @@ function applyReviewer(issue, instance, item, outcome) {
499
500
  autoMerge: issue.autoMerge,
500
501
  }, issue.headSha);
501
502
  const hasVerdict = outcome.kind === "verdict";
503
+ // Normalize before emit/finding reconcile so remapped blocking findings (NOT-150) persist.
504
+ const verdictResult = outcome.kind === "verdict" ? normalizeReviewerResult(outcome.result) : null;
502
505
  const { projection, effect, advance } = projectReviewerRoute(route, issue.currentRound, hasVerdict);
503
506
  const ev = eventEmitter(issue, instance, item.workerSessionId, projection.issueStatus, issue.currentRound);
504
507
  const patch = {};
505
508
  for (const type of projection.events) {
506
- if (type === "review.submitted" && outcome.kind === "verdict") {
507
- ev.emit("review.submitted", { actorType: "reviewer", payload: outcome.result });
509
+ if (type === "review.submitted" && verdictResult) {
510
+ ev.emit("review.submitted", { actorType: "reviewer", payload: verdictResult });
508
511
  }
509
512
  else if (type === "worker.completed" || type === "worker.failed") {
510
513
  const session = item.workerSessionId ? getWorkerSession(item.workerSessionId) : null;
@@ -544,8 +547,8 @@ function applyReviewer(issue, instance, item, outcome) {
544
547
  patch.headSha = outcome.currentHeadSha;
545
548
  }
546
549
  // Thread reviewer findings across rounds (PRD §6.4) — every blocking/non-blocking finding.
547
- if (outcome.kind === "verdict") {
548
- for (const f of outcome.result.findings) {
550
+ if (verdictResult) {
551
+ for (const f of verdictResult.findings) {
549
552
  reconcileFinding({
550
553
  issueId: issue.id,
551
554
  fingerprint: f.fingerprint,
@@ -651,7 +654,7 @@ function applyEffect(issue, instance, effect, route, issueNow, ev, causativeItem
651
654
  function questionFor(actionType, reason, resumeAsReviewer = false) {
652
655
  switch (actionType) {
653
656
  case "final_review":
654
- return "Accept and merge this work, send it back for another repair round, or close it?";
657
+ return "Merge this work, send it back for another repair round, or close it?";
655
658
  case "attempts_exhausted":
656
659
  return "The review-round limit is reached. Retry with a fresh round, or close the issue?";
657
660
  case "policy_escalation":
@@ -684,9 +687,9 @@ export function responseOptionsFor(actionType, resumeAsReviewer = false) {
684
687
  switch (actionType) {
685
688
  case "final_review":
686
689
  return [
687
- { choice: "complete", label: "Accept — merge & mark done" },
690
+ { choice: "merge", label: "Merge" },
688
691
  { choice: "repair", label: "Another repair round" },
689
- { choice: "close", label: "Close without accepting" },
692
+ { choice: "close", label: "Close" },
690
693
  ];
691
694
  case "attempts_exhausted":
692
695
  return [
@@ -748,9 +751,11 @@ export function resolveHumanActionAndAdvance(actionId, resolvedBy, choice) {
748
751
  if (!issue)
749
752
  return { ok: false, code: 404, error: "Issue not found" };
750
753
  const instance = getActiveWorkflowInstance(action.issueId);
751
- // NOT-102: human accept must undraft+merge (never mark done while leaving a draft PR).
754
+ // NOT-102 / NOT-150: human Merge (or legacy "complete") must undraft+merge.
752
755
  // Park like auto-merge, then the async wrapper runs finalizeAutoMerge outside this txn.
753
- if (instance && resolution.actionType === "final_review" && resolution.choice === "complete") {
756
+ if (instance &&
757
+ resolution.actionType === "final_review" &&
758
+ (resolution.choice === "merge" || resolution.choice === "complete")) {
754
759
  return getDb().transaction(() => {
755
760
  resolveHumanAction(actionId, resolvedBy, { choice });
756
761
  appendWorkflowEvent({
@@ -761,7 +766,7 @@ export function resolveHumanActionAndAdvance(actionId, resolvedBy, choice) {
761
766
  actorRef: resolvedBy,
762
767
  stage: "final_review",
763
768
  round: issue.currentRound,
764
- payload: { actionType: "final_review", choice: "complete", pendingMerge: true },
769
+ payload: { actionType: "final_review", choice: resolution.choice, pendingMerge: true },
765
770
  });
766
771
  transitionIssue(issue.id, "final_review", {
767
772
  currentOwner: "system",
@@ -966,7 +971,8 @@ function resolveLegacyTerminalAction(action, issue, resolvedBy, resolution) {
966
971
  if (resolution.choice === "close") {
967
972
  nextStatus = "closed";
968
973
  }
969
- else if (resolution.actionType === "final_review" && resolution.choice === "complete") {
974
+ else if (resolution.actionType === "final_review" &&
975
+ (resolution.choice === "merge" || resolution.choice === "complete")) {
970
976
  nextStatus = "done";
971
977
  }
972
978
  else if (resolution.actionType === "final_review" && resolution.choice === "repair") {
@@ -17,11 +17,13 @@ const { startWorkflow, applyCompletion, resolveHumanActionAndAdvance, resolveHum
17
17
  const { ReviewerResult } = await import("./reviewer-result.js");
18
18
  const { listArtifactsForIssue } = await import("../repository/artifacts-for-issue.js");
19
19
  const { setMergePrForTests, clearFinalizeInflightForTests } = await import("./auto-merge.js");
20
+ const { stubManagedCloneForTests } = await import("../adapters/managed-repo.js");
20
21
  before(() => migrate());
21
22
  beforeEach(() => {
22
23
  getDb().exec("DELETE FROM work_items");
23
24
  clearFinalizeInflightForTests();
24
25
  setMergePrForTests(async () => ({ ok: true }));
26
+ stubManagedCloneForTests("acme/app");
25
27
  });
26
28
  function newIssue(opts = {}) {
27
29
  return createIssue({
@@ -400,13 +402,16 @@ test("fail → retry → exhaust → resume → fail again does not collide with
400
402
  assert.equal(pending.length, 1, "a fresh item must be enqueued — the old (round, infraAttempts)-keyed row must not be silently reused");
401
403
  assert.notEqual(pending[0].id, firstRetryItem.id, "must be a NEW work item, not the pre-escalation retry's now-terminal row");
402
404
  });
403
- test("resolving policy_escalation:resume after a reviewer's escalated verdict (a real code-level question) still resumes as the developer", async () => {
405
+ test("resolving product_scope_decision:resume after a reviewer's escalated+question resumes as the developer", async () => {
404
406
  const issueId = newIssue();
405
407
  startWorkflow(issueId);
406
408
  await complete(issueId, cleanHandoff);
407
- await complete(issueId, { kind: "verdict", result: okReview("escalated") });
408
- const action = listHumanActionsForIssue(issueId).find((a) => a.actionType === "policy_escalation");
409
- assert.equal(action.continuationPreviewJson, null, "an escalated-verdict policy_escalation carries no reviewer-resume continuation");
409
+ await complete(issueId, {
410
+ kind: "verdict",
411
+ result: { ...okReview("escalated"), productScopeQuestion: "Should deleted users retain sessions?" },
412
+ });
413
+ const action = listHumanActionsForIssue(issueId).find((a) => a.actionType === "product_scope_decision");
414
+ assert.ok(action, "true product escalate opens product_scope_decision, not policy_escalation");
410
415
  const resolved = resolveHumanActionAndAdvance(action.id, "yusuke", "resume");
411
416
  assert.equal(resolved.ok, true);
412
417
  const issue = getIssue(issueId);
@@ -414,6 +419,18 @@ test("resolving policy_escalation:resume after a reviewer's escalated verdict (a
414
419
  const pending = listWorkItemsForIssue(issueId).filter((i) => i.status === "pending");
415
420
  assert.deepEqual(pending.map((i) => i.kind), ["developer"]);
416
421
  });
422
+ test("NOT-150: bare escalated verdict remaps to automatic repair, not policy_escalation", async () => {
423
+ const issueId = newIssue();
424
+ startWorkflow(issueId);
425
+ await complete(issueId, cleanHandoff);
426
+ await complete(issueId, { kind: "verdict", result: okReview("escalated") });
427
+ const issue = getIssue(issueId);
428
+ assert.equal(issue.status, "repairing");
429
+ assert.ok(!listHumanActionsForIssue(issueId).find((a) => a.actionType === "policy_escalation"));
430
+ assert.equal(listWorkItemsForIssue(issueId).filter((i) => i.kind === "developer" && i.status === "pending").length, 1);
431
+ const findings = listFindingsForIssue(issueId);
432
+ assert.ok(findings.some((f) => f.severity === "blocking"), "bare escalate remap must thread a blocking finding into repair");
433
+ });
417
434
  test("a stale review re-queues a reviewer at the new head without consuming a round", async () => {
418
435
  const issueId = newIssue();
419
436
  startWorkflow(issueId);
@@ -16,6 +16,7 @@
16
16
  // A persisted table can replace the provider later without touching the rule.
17
17
  import { isTerminalIssueStatus } from "@agent-dealer/shared";
18
18
  import { fetchLinearBlockers } from "../adapters/linear-inbox.js";
19
+ import { LinearHttpError } from "../adapters/linear-graphql.js";
19
20
  import { listIssuesByExternalId } from "../repository/issues.js";
20
21
  const EMPTY_SNAPSHOT = new Map();
21
22
  const DEFAULT_CACHE_TTL_MS = 60_000;
@@ -124,7 +125,10 @@ export async function blockersFor(issues) {
124
125
  fetched = await pending;
125
126
  }
126
127
  catch (err) {
127
- backoffUntil = Date.now() + failureBackoffMs;
128
+ // NOT-152: a 429 / exhausted budget must wait for Linear's reset window, not only the
129
+ // short FAILURE_BACKOFF_MS — otherwise admission re-hammers once that window lapses.
130
+ const rateLimitMs = err instanceof LinearHttpError && err.retryAfterMs != null ? err.retryAfterMs : 0;
131
+ backoffUntil = Date.now() + Math.max(failureBackoffMs, rateLimitMs);
128
132
  throw err;
129
133
  }
130
134
  finally {
@@ -322,6 +322,35 @@ test("a failing Linear is asked once per backoff window, not once per tick", asy
322
322
  setLinearBlockerFetcherForTests(async (ids) => new Map(ids.map((id) => [id, []])));
323
323
  assert.equal((await admitNext())?.issueId, issue.id);
324
324
  });
325
+ test("HTTP 429 backs off until Linear's reset header, not only the short failure window", async () => {
326
+ // NOT-152: a rate-limit must outlive FAILURE_BACKOFF_MS when X-RateLimit-Requests-Reset says so.
327
+ const { LinearHttpError } = await import("../adapters/linear-graphql.js");
328
+ const issue = seedIssue({ source: "linear", externalId: "lin-a" });
329
+ let fetches = 0;
330
+ setBlockerFailureBackoffForTests(50);
331
+ setLinearBlockerFetcherForTests(async () => {
332
+ fetches++;
333
+ throw new LinearHttpError({
334
+ status: 429,
335
+ operation: "fetchLinearBlockers",
336
+ rateLimit: {
337
+ requestsRemaining: "0",
338
+ requestsReset: String(Date.now() + 60_000),
339
+ },
340
+ bodySnippet: "rate limited",
341
+ retryAfterMs: 60_000,
342
+ });
343
+ });
344
+ await assert.rejects(() => blockersFor([issue]), (err) => {
345
+ assert.ok(err instanceof LinearHttpError);
346
+ assert.equal(err.status, 429);
347
+ return true;
348
+ });
349
+ assert.equal(fetches, 1);
350
+ await new Promise((r) => setTimeout(r, 80));
351
+ await assert.rejects(() => blockersFor([issue]), /backing off/);
352
+ assert.equal(fetches, 1, "must not re-hit Linear while the rate-limit window is open");
353
+ });
325
354
  test("overlapping ticks share one fetch: the second parks instead of opening a second call", async () => {
326
355
  const issue = seedIssue({ source: "linear", externalId: "lin-a" });
327
356
  let release = () => { };
@@ -371,7 +400,10 @@ function stubLinearIssues(batch, rounds = []) {
371
400
  else {
372
401
  data = { issues: { nodes: batch, pageInfo: { hasNextPage: false, endCursor: null } } };
373
402
  }
374
- return { ok: true, json: async () => ({ data }) };
403
+ return new Response(JSON.stringify({ data }), {
404
+ status: 200,
405
+ headers: { "Content-Type": "application/json" },
406
+ });
375
407
  });
376
408
  return {
377
409
  requests,
@@ -8,7 +8,8 @@
8
8
  * reopen an issue or enqueue developer/reviewer work).
9
9
  */
10
10
  const VALID_CHOICES = {
11
- final_review: ["complete", "repair", "close"],
11
+ // "complete" kept as a synonym for "merge" so older open actions / CLI callers still resolve.
12
+ final_review: ["merge", "complete", "repair", "close"],
12
13
  attempts_exhausted: ["retry", "close"],
13
14
  policy_escalation: ["resume", "close"],
14
15
  product_scope_decision: ["resume"],
@@ -52,8 +53,10 @@ export function parseHumanResolution(actionType, choice) {
52
53
  export function resolveHumanActionOutcome(resolution) {
53
54
  switch (resolution.actionType) {
54
55
  case "final_review":
55
- if (resolution.choice === "complete")
56
+ // Merge (and legacy "complete") undraft+merge via commands.ts, then mark done.
57
+ if (resolution.choice === "merge" || resolution.choice === "complete") {
56
58
  return { issueStatus: "done", workflowOutcome: "done", triggerReflect: true };
59
+ }
57
60
  if (resolution.choice === "repair")
58
61
  return { issueStatus: "repairing", startNewRound: true, roundKind: "review" };
59
62
  if (resolution.choice === "close")
@@ -23,12 +23,24 @@ test("parseHumanResolution rejects reflection_interaction_required even though V
23
23
  test("resolveHumanActionOutcome throws rather than silently closing on an invalid choice reaching it directly", () => {
24
24
  assert.throws(() => resolveHumanActionOutcome({ actionType: "final_review", choice: "bogus" }), /Unrecognized final_review choice/);
25
25
  });
26
- test("final_review complete marks the issue done and triggers reflect", () => {
26
+ test("final_review merge marks the issue done and triggers reflect", () => {
27
+ const result = resolveHumanActionOutcome({ actionType: "final_review", choice: "merge" });
28
+ assert.equal(result.issueStatus, "done");
29
+ assert.equal(result.workflowOutcome, "done");
30
+ assert.equal(result.triggerReflect, true);
31
+ });
32
+ test("final_review complete (legacy synonym) still marks done", () => {
27
33
  const result = resolveHumanActionOutcome({ actionType: "final_review", choice: "complete" });
28
34
  assert.equal(result.issueStatus, "done");
29
35
  assert.equal(result.workflowOutcome, "done");
30
36
  assert.equal(result.triggerReflect, true);
31
37
  });
38
+ test("parseHumanResolution accepts merge for final_review", () => {
39
+ assert.deepStrictEqual(parseHumanResolution("final_review", "merge"), {
40
+ actionType: "final_review",
41
+ choice: "merge",
42
+ });
43
+ });
32
44
  test("final_review repair sends the issue back for another round without reflect", () => {
33
45
  const result = resolveHumanActionOutcome({ actionType: "final_review", choice: "repair" });
34
46
  assert.equal(result.issueStatus, "repairing");
@@ -73,16 +73,18 @@ export function buildDeveloperPrompt(input) {
73
73
  }
74
74
  const REVIEWER_RESULT_SHAPE = '{"verdict":"approved"|"changes_requested"|"escalated","baseSha":"...","headSha":"...","acceptanceCriteriaAssessment":"...","evidenceAssessment":"...","findings":[{"fingerprint":"stable-slug","severity":"blocking"|"non_blocking","title":"...","rationale":"...","file":"...","line":0}],"risks":["..."],"productScopeQuestion":"..."}';
75
75
  function reviewerContractSection(baseSha, headSha) {
76
+ // Verdict table matches docs/PRD_ISSUE_COORDINATION.md §6.4 / design NOT-150 — do not drift.
76
77
  return [
77
78
  `## Required final JSON block`,
78
79
  `End your reply with exactly one fenced \`\`\`json block shaped like:`,
79
80
  REVIEWER_RESULT_SHAPE,
80
- `Rules:`,
81
+ `Rules (verdict contract):`,
81
82
  `- Set "baseSha" to exactly "${baseSha}" and "headSha" to exactly "${headSha}" — these are the coordinator-verified SHAs you were checked out at, not values you compute.`,
82
83
  `- "fingerprint" must be a short, stable slug for the finding (e.g. "missing-null-check-args-ts") so the same issue re-found next round is recognized as recurring, not duplicated.`,
83
84
  `- "findings" holds every blocking AND non-blocking observation; "risks" is uncertainties that are not findings tied to a location.`,
84
- `- Use "changes_requested" whenever any finding is "blocking". Use "approved" only when there are none.`,
85
- `- Use "escalated" only when the acceptance criteria themselves are ambiguous, contradictory, or the diff reveals a missing product decision — not for ordinary code problems. Set "productScopeQuestion" to that question; omit it otherwise.`,
85
+ `- "approved": AC met for this tip and no finding is "blocking" (non_blocking nits allowed).`,
86
+ `- "changes_requested": any "blocking" finding a coding pass can address — including incomplete review because AC-critical files were omitted/truncated from the diff. List omitted paths in a blocking finding.`,
87
+ `- "escalated": only when acceptance criteria / product scope are ambiguous, contradictory, or need a human product call — not ordinary code defects, not "diff too large". You MUST set non-empty "productScopeQuestion"; omit the field otherwise.`,
86
88
  `- You cannot edit files, push, or publish anything — you only return this JSON. The coordinator publishes it to GitHub on your behalf.`,
87
89
  ];
88
90
  }
@@ -91,37 +93,43 @@ function reviewerContractSection(baseSha, headSha) {
91
93
  * earlier, much smaller per-file cap still truncated mid-file on a genuinely large PR,
92
94
  * cutting off before the code under review). Whole files only — never a mid-hunk cut,
93
95
  * which would be actively misleading — so a file either fits completely or is entirely
94
- * omitted and counted in `truncated`. `runReviewerEffect` treats `truncated: true` as
95
- * grounds to override whatever verdict the reviewer reports: the reviewer cannot see the
96
- * full revision, so no verdict against it can be trusted, and this must be enforced in
97
- * code — a prompt instruction alone is not a structural guarantee (the same reasoning
98
- * `args.ts`/`permissions.ts` already apply to enforcement in general).
96
+ * omitted and listed in `omittedPaths`. Truncation policy (PRD §6.4 / NOT-150): coordinator
97
+ * remaps illegal escalate / approved+blocking; never blind escalate → Resume|Close.
99
98
  */
100
99
  export const TOTAL_DIFF_LIMIT = 300_000;
100
+ function pathFromDiffGitHeader(headerLine) {
101
+ // `diff --git a/path b/path` — prefer the b/ side; fall back to a/.
102
+ const m = headerLine.match(/^diff --git a\/(.+?) b\/(.+)$/);
103
+ if (!m)
104
+ return null;
105
+ return m[2] || m[1] || null;
106
+ }
101
107
  export function formatDiffForPrompt(diff) {
102
108
  const trimmed = diff.trim();
103
109
  if (!trimmed)
104
- return { text: "(empty diff)", truncated: false };
110
+ return { text: "(empty diff)", truncated: false, omittedPaths: [] };
105
111
  const blocks = trimmed.split(/(?=^diff --git )/m).filter(Boolean);
106
112
  const manifest = blocks.map((b) => b.slice(0, b.indexOf("\n"))).join("\n");
107
113
  const pieces = [];
114
+ const omittedPaths = [];
108
115
  let total = 0;
109
- let omittedFiles = 0;
110
116
  for (const block of blocks) {
111
117
  if (total + block.length > TOTAL_DIFF_LIMIT) {
112
- omittedFiles++;
118
+ const header = block.slice(0, block.indexOf("\n"));
119
+ omittedPaths.push(pathFromDiffGitHeader(header) ?? (header || "unknown"));
113
120
  continue;
114
121
  }
115
122
  pieces.push(block);
116
123
  total += block.length;
117
124
  }
118
- const truncated = omittedFiles > 0;
125
+ const truncated = omittedPaths.length > 0;
119
126
  const footer = truncated
120
- ? `\n... [${omittedFiles} changed file(s) omitted — this diff exceeds ${TOTAL_DIFF_LIMIT} characters. You cannot examine the complete revision; the coordinator will not accept "approved" or "changes_requested" from this session regardless of what you report.]`
127
+ ? `\n... [${omittedPaths.length} changed file(s) omitted — diff exceeds ${TOTAL_DIFF_LIMIT} characters. Omitted: ${omittedPaths.join(", ")}. If any omitted path is AC-critical, verdict MUST be "changes_requested" with a blocking finding listing those paths. If AC is still certifiable from the visible tip with only non_blocking findings, "approved" is allowed. Do NOT use "escalated" for truncation — escalate only with productScopeQuestion for a true product gap.]`
121
128
  : "";
122
129
  return {
123
130
  text: `Changed files (${blocks.length}):\n${manifest}\n\n${pieces.join("\n")}${footer}`,
124
131
  truncated,
132
+ omittedPaths,
125
133
  };
126
134
  }
127
135
  export function buildReviewerPrompt(input) {
@@ -145,6 +145,27 @@ test("reviewer prompt embeds the diff, echoes the exact SHAs to report, and forb
145
145
  assert.match(prompt, /"headSha" to exactly "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"/);
146
146
  assert.match(prompt, /You cannot edit files, push, or publish anything/);
147
147
  });
148
+ test("NOT-150: reviewer verdict rules match the design table (blocking ⇒ changes_requested; escalate needs productScopeQuestion)", () => {
149
+ const prompt = buildReviewerPrompt(reviewerBase);
150
+ assert.match(prompt, /"approved": AC met/);
151
+ assert.match(prompt, /no finding is "blocking"/);
152
+ assert.match(prompt, /"changes_requested": any "blocking" finding/);
153
+ assert.match(prompt, /"escalated": only when acceptance criteria/);
154
+ assert.match(prompt, /MUST set non-empty "productScopeQuestion"/);
155
+ assert.match(prompt, /not ordinary code defects/);
156
+ assert.match(prompt, /not "diff too large"/);
157
+ });
158
+ test("NOT-150: truncated-diff footer does not reject approved/changes_requested; forbids escalate-for-truncation", async () => {
159
+ const { formatDiffForPrompt, TOTAL_DIFF_LIMIT } = await import("./prompts.js");
160
+ const big = "x".repeat(TOTAL_DIFF_LIMIT + 1);
161
+ const diff = `diff --git a/small.ts b/small.ts\n+ok\n\ndiff --git a/huge.ts b/huge.ts\n+${big}\n`;
162
+ const formatted = formatDiffForPrompt(diff);
163
+ assert.equal(formatted.truncated, true);
164
+ assert.ok(formatted.omittedPaths.some((p) => p.includes("huge.ts")));
165
+ assert.match(formatted.text, /changes_requested/);
166
+ assert.match(formatted.text, /Do NOT use "escalated" for truncation/);
167
+ assert.doesNotMatch(formatted.text, /will not accept "approved" or "changes_requested"/);
168
+ });
148
169
  test("reviewer prompt includes the developer's conclusion, checks summary, and prior findings when given", () => {
149
170
  const prompt = buildReviewerPrompt({
150
171
  ...reviewerBase,
@@ -29,7 +29,7 @@ import { getTaskSnapshot } from "./commands.js";
29
29
  import { buildReviewerPrompt, formatDiffForPrompt, TOTAL_DIFF_LIMIT } from "./prompts.js";
30
30
  import { guidanceForNextSession } from "./guidance.js";
31
31
  import { realReviewerSpawn, reviewerSessionLogPath } from "./spawn.js";
32
- import { parseReviewerResult, ReviewerResult as ReviewerResultSchema } from "./reviewer-result.js";
32
+ import { parseReviewerResult, normalizeReviewerResult, INCOMPLETE_REVIEW_FINGERPRINT, ReviewerResult as ReviewerResultSchema, } from "./reviewer-result.js";
33
33
  import { createRoleWorktree, safeRemoveWorktree, isWorktreeClean, mergeBase, fetchRef, diffShas, } from "../adapters/git-worktree.js";
34
34
  import { ensureIssueRepoCheckout, roleWorktreePathForResolution, resolveCheckoutBaseBranch, } from "../adapters/managed-repo.js";
35
35
  import { prepareWorkerDeckConnection, releaseWorkerDeckConnection } from "../adapters/agent-deck-bind.js";
@@ -108,6 +108,47 @@ async function bestEffortRemove(repo, worktreePath) {
108
108
  // leave it for crash-recovery inspection — cleanup is a courtesy, not part of the contract
109
109
  }
110
110
  }
111
+ /** Inject a stable blocking incomplete-review finding listing omitted paths (NOT-150). */
112
+ function withIncompleteReviewFinding(result, omittedPaths) {
113
+ if (result.findings.some((f) => f.fingerprint === INCOMPLETE_REVIEW_FINGERPRINT)) {
114
+ const { productScopeQuestion: _drop, ...rest } = result;
115
+ return { ...rest, verdict: "changes_requested" };
116
+ }
117
+ const paths = omittedPaths.length > 0 ? omittedPaths.join(", ") : "(unlisted omitted files)";
118
+ const { productScopeQuestion: _drop, ...rest } = result;
119
+ return {
120
+ ...rest,
121
+ verdict: "changes_requested",
122
+ findings: [
123
+ ...result.findings,
124
+ {
125
+ fingerprint: INCOMPLETE_REVIEW_FINGERPRINT,
126
+ severity: "blocking",
127
+ title: "Diff truncated — incomplete review",
128
+ rationale: `Reviewer prompt omitted path(s): ${paths}. Cannot fully certify acceptance criteria without them; shrink the change set or split the PR so the next review sees the full AC-critical surface.`,
129
+ },
130
+ ],
131
+ };
132
+ }
133
+ /**
134
+ * Truncation remap (PRD §6.4 / design NOT-150): never blind escalate → Resume|Close.
135
+ * Shippable approved (no blocking) stays approved; otherwise ensure changes_requested with
136
+ * the incomplete-review finding listing omitted paths (in addition to any other blocking).
137
+ */
138
+ function applyTruncationVerdictPolicy(result, omittedPaths) {
139
+ const normalized = normalizeReviewerResult(result);
140
+ const hasProductQ = !!normalized.productScopeQuestion?.trim();
141
+ const hasBlocking = normalized.findings.some((f) => f.severity === "blocking");
142
+ if (normalized.verdict === "escalated" && hasProductQ) {
143
+ return normalized;
144
+ }
145
+ if (normalized.verdict === "approved" && !hasBlocking) {
146
+ return normalized;
147
+ }
148
+ // Truncated + not shippable-approved → changes_requested with incomplete-review (omitted paths).
149
+ const { productScopeQuestion: _drop, ...rest } = normalized;
150
+ return withIncompleteReviewFinding({ ...rest, verdict: "changes_requested" }, omittedPaths);
151
+ }
111
152
  function readImplementationConclusion(issueId) {
112
153
  const artifact = latestIssueArtifact(issueId, "implementation_conclusion");
113
154
  if (!artifact?.contentJson)
@@ -281,7 +322,7 @@ export async function runReviewerEffect(ctx, deps = defaultDeps) {
281
322
  await fetchRef(worktreePath, baseBranch);
282
323
  const baseSha = await mergeBase({ repo: worktreePath, base: `origin/${baseBranch}`, head: headSha });
283
324
  const diff = await diffShas({ worktreePath, baseSha, headSha });
284
- const { truncated: diffTruncated } = formatDiffForPrompt(diff);
325
+ const { truncated: diffTruncated, omittedPaths } = formatDiffForPrompt(diff);
285
326
  const openFindings = listFindingsForIssue(issue.id).filter((f) => f.status === "open" || f.status === "recurring");
286
327
  const guidance = guidanceForNextSession(issue.id, sessionId);
287
328
  const prompt = buildReviewerPrompt({
@@ -402,20 +443,23 @@ export async function runReviewerEffect(ctx, deps = defaultDeps) {
402
443
  return { kind: "session_failed" };
403
444
  }
404
445
  result = parsed;
405
- // A prompt instruction alone ("don't approve an incomplete diff") is not a structural
406
- // guarantee — the same reasoning this codebase already applies to tool permissions
407
- // (args.ts/permissions.ts). If the diff had to be truncated, the reviewer's verdict is
408
- // overridden to "escalated" in code, regardless of what it actually reported, so a
409
- // truncated review can never reach final_review or an automatic repair loop.
410
- if (diffTruncated && result.verdict !== "escalated") {
446
+ // Truncation policy (PRD §6.4 / design NOT-150): never blind-escalate to Resume|Close.
447
+ // Persist evidence; remap bare escalate / approved+blocking; keep shippable approved.
448
+ if (diffTruncated) {
449
+ const reportedVerdict = result.verdict;
450
+ result = applyTruncationVerdictPolicy(result, omittedPaths);
411
451
  createIssueArtifact({
412
452
  issueId: issue.id,
413
453
  workerSessionId: sessionId,
414
454
  kind: "diff_truncated_evidence",
415
455
  author: "system",
416
- content: { reportedVerdict: result.verdict, overriddenTo: "escalated", diffCharLimit: TOTAL_DIFF_LIMIT },
456
+ content: {
457
+ reportedVerdict,
458
+ overriddenTo: result.verdict,
459
+ diffCharLimit: TOTAL_DIFF_LIMIT,
460
+ omittedPaths,
461
+ },
417
462
  });
418
- result = { ...result, verdict: "escalated" };
419
463
  }
420
464
  }
421
465
  catch {
@@ -456,7 +500,8 @@ export async function runReviewerEffect(ctx, deps = defaultDeps) {
456
500
  // diff). A `row` with no recorded result (still `claimed` after every wait
457
501
  // attempt, or the winner itself failed) has nothing safe to report — escalate.
458
502
  if (claim.row?.state === "published" && claim.row.resultJson) {
459
- const winnerResult = ReviewerResultSchema.parse(JSON.parse(claim.row.resultJson));
503
+ const winnerParsed = ReviewerResultSchema.parse(JSON.parse(claim.row.resultJson));
504
+ const winnerResult = normalizeReviewerResult(winnerParsed);
460
505
  return { kind: "verdict", result: winnerResult };
461
506
  }
462
507
  return { kind: "publish_failed" };