agent-dealer 1.0.0 → 1.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bundle/server/dist/adapters/linear-graphql.js +165 -0
- package/bundle/server/dist/adapters/linear-graphql.test.js +129 -0
- package/bundle/server/dist/adapters/linear-inbox.js +12 -21
- package/bundle/server/dist/adapters/linear-sync.js +6 -19
- package/bundle/server/dist/adapters/managed-repo.js +11 -0
- package/bundle/server/dist/coordinator/auto-merge.integration.test.js +55 -4
- package/bundle/server/dist/coordinator/auto-merge.js +72 -4
- package/bundle/server/dist/coordinator/auto-merge.timeout.test.js +45 -3
- package/bundle/server/dist/coordinator/commands.js +17 -11
- package/bundle/server/dist/coordinator/commands.test.js +21 -4
- package/bundle/server/dist/coordinator/dependencies.js +5 -1
- package/bundle/server/dist/coordinator/dependency-readiness.integration.test.js +33 -1
- package/bundle/server/dist/coordinator/human-resolution.js +5 -2
- package/bundle/server/dist/coordinator/human-resolution.test.js +13 -1
- package/bundle/server/dist/coordinator/prompts.js +21 -13
- package/bundle/server/dist/coordinator/prompts.test.js +21 -0
- package/bundle/server/dist/coordinator/reviewer-effect.js +56 -11
- package/bundle/server/dist/coordinator/reviewer-effect.test.js +67 -9
- package/bundle/server/dist/coordinator/reviewer-result.js +49 -1
- package/bundle/server/dist/coordinator/reviewer-result.test.js +38 -0
- package/bundle/server/dist/coordinator/routing.js +15 -4
- package/bundle/server/dist/coordinator/routing.test.js +17 -2
- package/bundle/server/dist/coordinator/worker-loop.test.js +2 -0
- package/bundle/server/dist/db/index.js +6 -0
- package/bundle/server/dist/db/migrate-to-issues.test.js +1 -0
- package/bundle/server/dist/dev-review-cli-happy-path.integration.test.js +3 -2
- package/bundle/server/dist/direct-start-interrupt-probe.js +58 -0
- package/bundle/server/dist/direct-start-liveness.integration.test.js +114 -24
- package/bundle/server/dist/direct-start-temp-home-cleanup.js +124 -0
- package/bundle/server/dist/direct-start-temp-home-cleanup.test.js +92 -0
- package/bundle/server/dist/repository/queue-entries.js +49 -11
- package/bundle/server/dist/routes/human-actions.test.js +2 -0
- package/bundle/server/dist/routes/queue-reorder.integration.test.js +144 -0
- package/bundle/server/dist/routes/queue.js +21 -2
- package/bundle/server/package.json +2 -2
- package/bundle/server/static-ui/assets/index-0kT1vk6L.js +60 -0
- package/bundle/server/static-ui/assets/{index-BxTx4b1d.css → index-hXICi1rX.css} +1 -1
- package/bundle/server/static-ui/index.html +2 -2
- package/bundle/shared/dist/queue-entries.d.ts +47 -0
- package/bundle/shared/dist/queue-entries.js +14 -0
- package/bundle/shared/package.json +1 -1
- package/dist/action.test.js +2 -2
- package/dist/index.js +1 -0
- package/dist/install.js +1 -1
- package/dist/lifecycle.contract.test.js +2 -2
- package/dist/queue.d.ts +5 -0
- package/dist/queue.js +50 -0
- package/dist/queue.test.js +39 -0
- package/dist/setup.js +5 -1
- package/package.json +1 -1
- package/bundle/server/static-ui/assets/index-BTtPGYpY.js +0 -60
|
@@ -1,7 +1,17 @@
|
|
|
1
|
-
// Unit coverage for bounded gh timeout classification (NOT-102
|
|
2
|
-
import { test } from "node:test";
|
|
1
|
+
// Unit coverage for bounded gh timeout classification (NOT-102) and NOT-151 spawn ENOENT mapping.
|
|
2
|
+
import { test, before } from "node:test";
|
|
3
3
|
import assert from "node:assert/strict";
|
|
4
|
-
import
|
|
4
|
+
import fs from "node:fs";
|
|
5
|
+
import os from "node:os";
|
|
6
|
+
import path from "node:path";
|
|
7
|
+
import { execFileSync } from "node:child_process";
|
|
8
|
+
process.env.AGENT_DEALER_HOME = fs.mkdtempSync(path.join(os.tmpdir(), "dealer-not151-unit-"));
|
|
9
|
+
const { GH_MERGE_TIMEOUT_MS, ghErrorReason, ghSpawnEnoentReason, isGhTimeoutError, resolveAutoMergeCwd, } = await import("./auto-merge.js");
|
|
10
|
+
const { managedRepoPath } = await import("../adapters/managed-repo.js");
|
|
11
|
+
before(async () => {
|
|
12
|
+
const { migrate } = await import("../db/index.js");
|
|
13
|
+
migrate();
|
|
14
|
+
});
|
|
5
15
|
test("isGhTimeoutError detects killed / SIGTERM from execFile timeout", () => {
|
|
6
16
|
assert.equal(isGhTimeoutError({ killed: true }), true);
|
|
7
17
|
assert.equal(isGhTimeoutError({ signal: "SIGTERM" }), true);
|
|
@@ -12,3 +22,35 @@ test("ghErrorReason maps timeout before stderr", () => {
|
|
|
12
22
|
assert.equal(ghErrorReason({ stderr: " checks failed \n" }, "fallback"), "checks failed");
|
|
13
23
|
assert.equal(ghErrorReason({}, "gh pr merge failed"), "gh pr merge failed");
|
|
14
24
|
});
|
|
25
|
+
test("NOT-151: ghSpawnEnoentReason distinguishes bad cwd from missing gh", () => {
|
|
26
|
+
const missing = path.join(os.tmpdir(), "dealer-missing-merge-cwd-xyz");
|
|
27
|
+
assert.match(ghSpawnEnoentReason({ code: "ENOENT", message: "spawn gh ENOENT" }, missing) ?? "", /invalid merge cwd/);
|
|
28
|
+
const existing = fs.mkdtempSync(path.join(os.tmpdir(), "dealer-merge-cwd-"));
|
|
29
|
+
assert.match(ghSpawnEnoentReason({ code: "ENOENT", message: "spawn gh ENOENT" }, existing) ?? "", /gh not on PATH/);
|
|
30
|
+
assert.equal(ghSpawnEnoentReason({ stderr: "checks failed" }, existing), null);
|
|
31
|
+
});
|
|
32
|
+
test("NOT-151: resolveAutoMergeCwd maps portable identity to managed path when clone exists", () => {
|
|
33
|
+
const identity = "github.com/not-so-fat/agent-dealer";
|
|
34
|
+
const managed = managedRepoPath(identity);
|
|
35
|
+
fs.mkdirSync(path.join(managed, ".git"), { recursive: true });
|
|
36
|
+
const ok = resolveAutoMergeCwd(identity);
|
|
37
|
+
assert.equal(ok.ok, true);
|
|
38
|
+
if (ok.ok)
|
|
39
|
+
assert.equal(ok.cwd, managed);
|
|
40
|
+
});
|
|
41
|
+
test("NOT-151: resolveAutoMergeCwd fails closed when managed clone is missing", () => {
|
|
42
|
+
const missing = resolveAutoMergeCwd("github.com/missing/not-cloned");
|
|
43
|
+
assert.equal(missing.ok, false);
|
|
44
|
+
if (!missing.ok) {
|
|
45
|
+
assert.match(missing.reason, /Managed clone missing/);
|
|
46
|
+
assert.doesNotMatch(missing.reason, /ENOENT/);
|
|
47
|
+
}
|
|
48
|
+
});
|
|
49
|
+
test("NOT-151: resolveAutoMergeCwd accepts a real legacy local checkout", () => {
|
|
50
|
+
const dir = fs.mkdtempSync(path.join(os.tmpdir(), "legacy-merge-cwd-"));
|
|
51
|
+
execFileSync("git", ["init", "-b", "main"], { cwd: dir });
|
|
52
|
+
const ok = resolveAutoMergeCwd(dir);
|
|
53
|
+
assert.equal(ok.ok, true);
|
|
54
|
+
if (ok.ok)
|
|
55
|
+
assert.equal(ok.cwd, dir);
|
|
56
|
+
});
|
|
@@ -3,6 +3,7 @@ import { getIssue, incrementIssueRound, incrementIssueInfraAttempts, resetIssueI
|
|
|
3
3
|
import { appendWorkflowEvent, completeWorkflowInstance, getActiveWorkflowInstance, getWorkflowInstance, startWorkflowInstance, WorkflowAlreadyActiveError, } from "../repository/workflow-events.js";
|
|
4
4
|
import { createHumanAction, findOpenHumanAction, getHumanAction, listHumanActionsForIssue, resolveHumanAction, } from "../repository/human-actions.js";
|
|
5
5
|
import { reconcileFinding } from "../repository/findings.js";
|
|
6
|
+
import { normalizeReviewerResult } from "./reviewer-result.js";
|
|
6
7
|
import { getAgent } from "../repository/agents.js";
|
|
7
8
|
import { githubIssuesSync } from "../adapters/agent-health.js";
|
|
8
9
|
import { createIssueArtifact, latestIssueArtifact } from "../repository/artifacts.js";
|
|
@@ -499,12 +500,14 @@ function applyReviewer(issue, instance, item, outcome) {
|
|
|
499
500
|
autoMerge: issue.autoMerge,
|
|
500
501
|
}, issue.headSha);
|
|
501
502
|
const hasVerdict = outcome.kind === "verdict";
|
|
503
|
+
// Normalize before emit/finding reconcile so remapped blocking findings (NOT-150) persist.
|
|
504
|
+
const verdictResult = outcome.kind === "verdict" ? normalizeReviewerResult(outcome.result) : null;
|
|
502
505
|
const { projection, effect, advance } = projectReviewerRoute(route, issue.currentRound, hasVerdict);
|
|
503
506
|
const ev = eventEmitter(issue, instance, item.workerSessionId, projection.issueStatus, issue.currentRound);
|
|
504
507
|
const patch = {};
|
|
505
508
|
for (const type of projection.events) {
|
|
506
|
-
if (type === "review.submitted" &&
|
|
507
|
-
ev.emit("review.submitted", { actorType: "reviewer", payload:
|
|
509
|
+
if (type === "review.submitted" && verdictResult) {
|
|
510
|
+
ev.emit("review.submitted", { actorType: "reviewer", payload: verdictResult });
|
|
508
511
|
}
|
|
509
512
|
else if (type === "worker.completed" || type === "worker.failed") {
|
|
510
513
|
const session = item.workerSessionId ? getWorkerSession(item.workerSessionId) : null;
|
|
@@ -544,8 +547,8 @@ function applyReviewer(issue, instance, item, outcome) {
|
|
|
544
547
|
patch.headSha = outcome.currentHeadSha;
|
|
545
548
|
}
|
|
546
549
|
// Thread reviewer findings across rounds (PRD §6.4) — every blocking/non-blocking finding.
|
|
547
|
-
if (
|
|
548
|
-
for (const f of
|
|
550
|
+
if (verdictResult) {
|
|
551
|
+
for (const f of verdictResult.findings) {
|
|
549
552
|
reconcileFinding({
|
|
550
553
|
issueId: issue.id,
|
|
551
554
|
fingerprint: f.fingerprint,
|
|
@@ -651,7 +654,7 @@ function applyEffect(issue, instance, effect, route, issueNow, ev, causativeItem
|
|
|
651
654
|
function questionFor(actionType, reason, resumeAsReviewer = false) {
|
|
652
655
|
switch (actionType) {
|
|
653
656
|
case "final_review":
|
|
654
|
-
return "
|
|
657
|
+
return "Merge this work, send it back for another repair round, or close it?";
|
|
655
658
|
case "attempts_exhausted":
|
|
656
659
|
return "The review-round limit is reached. Retry with a fresh round, or close the issue?";
|
|
657
660
|
case "policy_escalation":
|
|
@@ -684,9 +687,9 @@ export function responseOptionsFor(actionType, resumeAsReviewer = false) {
|
|
|
684
687
|
switch (actionType) {
|
|
685
688
|
case "final_review":
|
|
686
689
|
return [
|
|
687
|
-
{ choice: "
|
|
690
|
+
{ choice: "merge", label: "Merge" },
|
|
688
691
|
{ choice: "repair", label: "Another repair round" },
|
|
689
|
-
{ choice: "close", label: "Close
|
|
692
|
+
{ choice: "close", label: "Close" },
|
|
690
693
|
];
|
|
691
694
|
case "attempts_exhausted":
|
|
692
695
|
return [
|
|
@@ -748,9 +751,11 @@ export function resolveHumanActionAndAdvance(actionId, resolvedBy, choice) {
|
|
|
748
751
|
if (!issue)
|
|
749
752
|
return { ok: false, code: 404, error: "Issue not found" };
|
|
750
753
|
const instance = getActiveWorkflowInstance(action.issueId);
|
|
751
|
-
// NOT-102: human
|
|
754
|
+
// NOT-102 / NOT-150: human Merge (or legacy "complete") must undraft+merge.
|
|
752
755
|
// Park like auto-merge, then the async wrapper runs finalizeAutoMerge outside this txn.
|
|
753
|
-
if (instance &&
|
|
756
|
+
if (instance &&
|
|
757
|
+
resolution.actionType === "final_review" &&
|
|
758
|
+
(resolution.choice === "merge" || resolution.choice === "complete")) {
|
|
754
759
|
return getDb().transaction(() => {
|
|
755
760
|
resolveHumanAction(actionId, resolvedBy, { choice });
|
|
756
761
|
appendWorkflowEvent({
|
|
@@ -761,7 +766,7 @@ export function resolveHumanActionAndAdvance(actionId, resolvedBy, choice) {
|
|
|
761
766
|
actorRef: resolvedBy,
|
|
762
767
|
stage: "final_review",
|
|
763
768
|
round: issue.currentRound,
|
|
764
|
-
payload: { actionType: "final_review", choice:
|
|
769
|
+
payload: { actionType: "final_review", choice: resolution.choice, pendingMerge: true },
|
|
765
770
|
});
|
|
766
771
|
transitionIssue(issue.id, "final_review", {
|
|
767
772
|
currentOwner: "system",
|
|
@@ -966,7 +971,8 @@ function resolveLegacyTerminalAction(action, issue, resolvedBy, resolution) {
|
|
|
966
971
|
if (resolution.choice === "close") {
|
|
967
972
|
nextStatus = "closed";
|
|
968
973
|
}
|
|
969
|
-
else if (resolution.actionType === "final_review" &&
|
|
974
|
+
else if (resolution.actionType === "final_review" &&
|
|
975
|
+
(resolution.choice === "merge" || resolution.choice === "complete")) {
|
|
970
976
|
nextStatus = "done";
|
|
971
977
|
}
|
|
972
978
|
else if (resolution.actionType === "final_review" && resolution.choice === "repair") {
|
|
@@ -17,11 +17,13 @@ const { startWorkflow, applyCompletion, resolveHumanActionAndAdvance, resolveHum
|
|
|
17
17
|
const { ReviewerResult } = await import("./reviewer-result.js");
|
|
18
18
|
const { listArtifactsForIssue } = await import("../repository/artifacts-for-issue.js");
|
|
19
19
|
const { setMergePrForTests, clearFinalizeInflightForTests } = await import("./auto-merge.js");
|
|
20
|
+
const { stubManagedCloneForTests } = await import("../adapters/managed-repo.js");
|
|
20
21
|
before(() => migrate());
|
|
21
22
|
beforeEach(() => {
|
|
22
23
|
getDb().exec("DELETE FROM work_items");
|
|
23
24
|
clearFinalizeInflightForTests();
|
|
24
25
|
setMergePrForTests(async () => ({ ok: true }));
|
|
26
|
+
stubManagedCloneForTests("acme/app");
|
|
25
27
|
});
|
|
26
28
|
function newIssue(opts = {}) {
|
|
27
29
|
return createIssue({
|
|
@@ -400,13 +402,16 @@ test("fail → retry → exhaust → resume → fail again does not collide with
|
|
|
400
402
|
assert.equal(pending.length, 1, "a fresh item must be enqueued — the old (round, infraAttempts)-keyed row must not be silently reused");
|
|
401
403
|
assert.notEqual(pending[0].id, firstRetryItem.id, "must be a NEW work item, not the pre-escalation retry's now-terminal row");
|
|
402
404
|
});
|
|
403
|
-
test("resolving
|
|
405
|
+
test("resolving product_scope_decision:resume after a reviewer's escalated+question resumes as the developer", async () => {
|
|
404
406
|
const issueId = newIssue();
|
|
405
407
|
startWorkflow(issueId);
|
|
406
408
|
await complete(issueId, cleanHandoff);
|
|
407
|
-
await complete(issueId, {
|
|
408
|
-
|
|
409
|
-
|
|
409
|
+
await complete(issueId, {
|
|
410
|
+
kind: "verdict",
|
|
411
|
+
result: { ...okReview("escalated"), productScopeQuestion: "Should deleted users retain sessions?" },
|
|
412
|
+
});
|
|
413
|
+
const action = listHumanActionsForIssue(issueId).find((a) => a.actionType === "product_scope_decision");
|
|
414
|
+
assert.ok(action, "true product escalate opens product_scope_decision, not policy_escalation");
|
|
410
415
|
const resolved = resolveHumanActionAndAdvance(action.id, "yusuke", "resume");
|
|
411
416
|
assert.equal(resolved.ok, true);
|
|
412
417
|
const issue = getIssue(issueId);
|
|
@@ -414,6 +419,18 @@ test("resolving policy_escalation:resume after a reviewer's escalated verdict (a
|
|
|
414
419
|
const pending = listWorkItemsForIssue(issueId).filter((i) => i.status === "pending");
|
|
415
420
|
assert.deepEqual(pending.map((i) => i.kind), ["developer"]);
|
|
416
421
|
});
|
|
422
|
+
test("NOT-150: bare escalated verdict remaps to automatic repair, not policy_escalation", async () => {
|
|
423
|
+
const issueId = newIssue();
|
|
424
|
+
startWorkflow(issueId);
|
|
425
|
+
await complete(issueId, cleanHandoff);
|
|
426
|
+
await complete(issueId, { kind: "verdict", result: okReview("escalated") });
|
|
427
|
+
const issue = getIssue(issueId);
|
|
428
|
+
assert.equal(issue.status, "repairing");
|
|
429
|
+
assert.ok(!listHumanActionsForIssue(issueId).find((a) => a.actionType === "policy_escalation"));
|
|
430
|
+
assert.equal(listWorkItemsForIssue(issueId).filter((i) => i.kind === "developer" && i.status === "pending").length, 1);
|
|
431
|
+
const findings = listFindingsForIssue(issueId);
|
|
432
|
+
assert.ok(findings.some((f) => f.severity === "blocking"), "bare escalate remap must thread a blocking finding into repair");
|
|
433
|
+
});
|
|
417
434
|
test("a stale review re-queues a reviewer at the new head without consuming a round", async () => {
|
|
418
435
|
const issueId = newIssue();
|
|
419
436
|
startWorkflow(issueId);
|
|
@@ -16,6 +16,7 @@
|
|
|
16
16
|
// A persisted table can replace the provider later without touching the rule.
|
|
17
17
|
import { isTerminalIssueStatus } from "@agent-dealer/shared";
|
|
18
18
|
import { fetchLinearBlockers } from "../adapters/linear-inbox.js";
|
|
19
|
+
import { LinearHttpError } from "../adapters/linear-graphql.js";
|
|
19
20
|
import { listIssuesByExternalId } from "../repository/issues.js";
|
|
20
21
|
const EMPTY_SNAPSHOT = new Map();
|
|
21
22
|
const DEFAULT_CACHE_TTL_MS = 60_000;
|
|
@@ -124,7 +125,10 @@ export async function blockersFor(issues) {
|
|
|
124
125
|
fetched = await pending;
|
|
125
126
|
}
|
|
126
127
|
catch (err) {
|
|
127
|
-
|
|
128
|
+
// NOT-152: a 429 / exhausted budget must wait for Linear's reset window, not only the
|
|
129
|
+
// short FAILURE_BACKOFF_MS — otherwise admission re-hammers once that window lapses.
|
|
130
|
+
const rateLimitMs = err instanceof LinearHttpError && err.retryAfterMs != null ? err.retryAfterMs : 0;
|
|
131
|
+
backoffUntil = Date.now() + Math.max(failureBackoffMs, rateLimitMs);
|
|
128
132
|
throw err;
|
|
129
133
|
}
|
|
130
134
|
finally {
|
|
@@ -322,6 +322,35 @@ test("a failing Linear is asked once per backoff window, not once per tick", asy
|
|
|
322
322
|
setLinearBlockerFetcherForTests(async (ids) => new Map(ids.map((id) => [id, []])));
|
|
323
323
|
assert.equal((await admitNext())?.issueId, issue.id);
|
|
324
324
|
});
|
|
325
|
+
test("HTTP 429 backs off until Linear's reset header, not only the short failure window", async () => {
|
|
326
|
+
// NOT-152: a rate-limit must outlive FAILURE_BACKOFF_MS when X-RateLimit-Requests-Reset says so.
|
|
327
|
+
const { LinearHttpError } = await import("../adapters/linear-graphql.js");
|
|
328
|
+
const issue = seedIssue({ source: "linear", externalId: "lin-a" });
|
|
329
|
+
let fetches = 0;
|
|
330
|
+
setBlockerFailureBackoffForTests(50);
|
|
331
|
+
setLinearBlockerFetcherForTests(async () => {
|
|
332
|
+
fetches++;
|
|
333
|
+
throw new LinearHttpError({
|
|
334
|
+
status: 429,
|
|
335
|
+
operation: "fetchLinearBlockers",
|
|
336
|
+
rateLimit: {
|
|
337
|
+
requestsRemaining: "0",
|
|
338
|
+
requestsReset: String(Date.now() + 60_000),
|
|
339
|
+
},
|
|
340
|
+
bodySnippet: "rate limited",
|
|
341
|
+
retryAfterMs: 60_000,
|
|
342
|
+
});
|
|
343
|
+
});
|
|
344
|
+
await assert.rejects(() => blockersFor([issue]), (err) => {
|
|
345
|
+
assert.ok(err instanceof LinearHttpError);
|
|
346
|
+
assert.equal(err.status, 429);
|
|
347
|
+
return true;
|
|
348
|
+
});
|
|
349
|
+
assert.equal(fetches, 1);
|
|
350
|
+
await new Promise((r) => setTimeout(r, 80));
|
|
351
|
+
await assert.rejects(() => blockersFor([issue]), /backing off/);
|
|
352
|
+
assert.equal(fetches, 1, "must not re-hit Linear while the rate-limit window is open");
|
|
353
|
+
});
|
|
325
354
|
test("overlapping ticks share one fetch: the second parks instead of opening a second call", async () => {
|
|
326
355
|
const issue = seedIssue({ source: "linear", externalId: "lin-a" });
|
|
327
356
|
let release = () => { };
|
|
@@ -371,7 +400,10 @@ function stubLinearIssues(batch, rounds = []) {
|
|
|
371
400
|
else {
|
|
372
401
|
data = { issues: { nodes: batch, pageInfo: { hasNextPage: false, endCursor: null } } };
|
|
373
402
|
}
|
|
374
|
-
return
|
|
403
|
+
return new Response(JSON.stringify({ data }), {
|
|
404
|
+
status: 200,
|
|
405
|
+
headers: { "Content-Type": "application/json" },
|
|
406
|
+
});
|
|
375
407
|
});
|
|
376
408
|
return {
|
|
377
409
|
requests,
|
|
@@ -8,7 +8,8 @@
|
|
|
8
8
|
* reopen an issue or enqueue developer/reviewer work).
|
|
9
9
|
*/
|
|
10
10
|
const VALID_CHOICES = {
|
|
11
|
-
|
|
11
|
+
// "complete" kept as a synonym for "merge" so older open actions / CLI callers still resolve.
|
|
12
|
+
final_review: ["merge", "complete", "repair", "close"],
|
|
12
13
|
attempts_exhausted: ["retry", "close"],
|
|
13
14
|
policy_escalation: ["resume", "close"],
|
|
14
15
|
product_scope_decision: ["resume"],
|
|
@@ -52,8 +53,10 @@ export function parseHumanResolution(actionType, choice) {
|
|
|
52
53
|
export function resolveHumanActionOutcome(resolution) {
|
|
53
54
|
switch (resolution.actionType) {
|
|
54
55
|
case "final_review":
|
|
55
|
-
|
|
56
|
+
// Merge (and legacy "complete") undraft+merge via commands.ts, then mark done.
|
|
57
|
+
if (resolution.choice === "merge" || resolution.choice === "complete") {
|
|
56
58
|
return { issueStatus: "done", workflowOutcome: "done", triggerReflect: true };
|
|
59
|
+
}
|
|
57
60
|
if (resolution.choice === "repair")
|
|
58
61
|
return { issueStatus: "repairing", startNewRound: true, roundKind: "review" };
|
|
59
62
|
if (resolution.choice === "close")
|
|
@@ -23,12 +23,24 @@ test("parseHumanResolution rejects reflection_interaction_required even though V
|
|
|
23
23
|
test("resolveHumanActionOutcome throws rather than silently closing on an invalid choice reaching it directly", () => {
|
|
24
24
|
assert.throws(() => resolveHumanActionOutcome({ actionType: "final_review", choice: "bogus" }), /Unrecognized final_review choice/);
|
|
25
25
|
});
|
|
26
|
-
test("final_review
|
|
26
|
+
test("final_review merge marks the issue done and triggers reflect", () => {
|
|
27
|
+
const result = resolveHumanActionOutcome({ actionType: "final_review", choice: "merge" });
|
|
28
|
+
assert.equal(result.issueStatus, "done");
|
|
29
|
+
assert.equal(result.workflowOutcome, "done");
|
|
30
|
+
assert.equal(result.triggerReflect, true);
|
|
31
|
+
});
|
|
32
|
+
test("final_review complete (legacy synonym) still marks done", () => {
|
|
27
33
|
const result = resolveHumanActionOutcome({ actionType: "final_review", choice: "complete" });
|
|
28
34
|
assert.equal(result.issueStatus, "done");
|
|
29
35
|
assert.equal(result.workflowOutcome, "done");
|
|
30
36
|
assert.equal(result.triggerReflect, true);
|
|
31
37
|
});
|
|
38
|
+
test("parseHumanResolution accepts merge for final_review", () => {
|
|
39
|
+
assert.deepStrictEqual(parseHumanResolution("final_review", "merge"), {
|
|
40
|
+
actionType: "final_review",
|
|
41
|
+
choice: "merge",
|
|
42
|
+
});
|
|
43
|
+
});
|
|
32
44
|
test("final_review repair sends the issue back for another round without reflect", () => {
|
|
33
45
|
const result = resolveHumanActionOutcome({ actionType: "final_review", choice: "repair" });
|
|
34
46
|
assert.equal(result.issueStatus, "repairing");
|
|
@@ -73,16 +73,18 @@ export function buildDeveloperPrompt(input) {
|
|
|
73
73
|
}
|
|
74
74
|
const REVIEWER_RESULT_SHAPE = '{"verdict":"approved"|"changes_requested"|"escalated","baseSha":"...","headSha":"...","acceptanceCriteriaAssessment":"...","evidenceAssessment":"...","findings":[{"fingerprint":"stable-slug","severity":"blocking"|"non_blocking","title":"...","rationale":"...","file":"...","line":0}],"risks":["..."],"productScopeQuestion":"..."}';
|
|
75
75
|
function reviewerContractSection(baseSha, headSha) {
|
|
76
|
+
// Verdict table matches docs/PRD_ISSUE_COORDINATION.md §6.4 / design NOT-150 — do not drift.
|
|
76
77
|
return [
|
|
77
78
|
`## Required final JSON block`,
|
|
78
79
|
`End your reply with exactly one fenced \`\`\`json block shaped like:`,
|
|
79
80
|
REVIEWER_RESULT_SHAPE,
|
|
80
|
-
`Rules:`,
|
|
81
|
+
`Rules (verdict contract):`,
|
|
81
82
|
`- Set "baseSha" to exactly "${baseSha}" and "headSha" to exactly "${headSha}" — these are the coordinator-verified SHAs you were checked out at, not values you compute.`,
|
|
82
83
|
`- "fingerprint" must be a short, stable slug for the finding (e.g. "missing-null-check-args-ts") so the same issue re-found next round is recognized as recurring, not duplicated.`,
|
|
83
84
|
`- "findings" holds every blocking AND non-blocking observation; "risks" is uncertainties that are not findings tied to a location.`,
|
|
84
|
-
`-
|
|
85
|
-
`-
|
|
85
|
+
`- "approved": AC met for this tip and no finding is "blocking" (non_blocking nits allowed).`,
|
|
86
|
+
`- "changes_requested": any "blocking" finding a coding pass can address — including incomplete review because AC-critical files were omitted/truncated from the diff. List omitted paths in a blocking finding.`,
|
|
87
|
+
`- "escalated": only when acceptance criteria / product scope are ambiguous, contradictory, or need a human product call — not ordinary code defects, not "diff too large". You MUST set non-empty "productScopeQuestion"; omit the field otherwise.`,
|
|
86
88
|
`- You cannot edit files, push, or publish anything — you only return this JSON. The coordinator publishes it to GitHub on your behalf.`,
|
|
87
89
|
];
|
|
88
90
|
}
|
|
@@ -91,37 +93,43 @@ function reviewerContractSection(baseSha, headSha) {
|
|
|
91
93
|
* earlier, much smaller per-file cap still truncated mid-file on a genuinely large PR,
|
|
92
94
|
* cutting off before the code under review). Whole files only — never a mid-hunk cut,
|
|
93
95
|
* which would be actively misleading — so a file either fits completely or is entirely
|
|
94
|
-
* omitted and
|
|
95
|
-
*
|
|
96
|
-
* full revision, so no verdict against it can be trusted, and this must be enforced in
|
|
97
|
-
* code — a prompt instruction alone is not a structural guarantee (the same reasoning
|
|
98
|
-
* `args.ts`/`permissions.ts` already apply to enforcement in general).
|
|
96
|
+
* omitted and listed in `omittedPaths`. Truncation policy (PRD §6.4 / NOT-150): coordinator
|
|
97
|
+
* remaps illegal escalate / approved+blocking; never blind escalate → Resume|Close.
|
|
99
98
|
*/
|
|
100
99
|
export const TOTAL_DIFF_LIMIT = 300_000;
|
|
100
|
+
function pathFromDiffGitHeader(headerLine) {
|
|
101
|
+
// `diff --git a/path b/path` — prefer the b/ side; fall back to a/.
|
|
102
|
+
const m = headerLine.match(/^diff --git a\/(.+?) b\/(.+)$/);
|
|
103
|
+
if (!m)
|
|
104
|
+
return null;
|
|
105
|
+
return m[2] || m[1] || null;
|
|
106
|
+
}
|
|
101
107
|
export function formatDiffForPrompt(diff) {
|
|
102
108
|
const trimmed = diff.trim();
|
|
103
109
|
if (!trimmed)
|
|
104
|
-
return { text: "(empty diff)", truncated: false };
|
|
110
|
+
return { text: "(empty diff)", truncated: false, omittedPaths: [] };
|
|
105
111
|
const blocks = trimmed.split(/(?=^diff --git )/m).filter(Boolean);
|
|
106
112
|
const manifest = blocks.map((b) => b.slice(0, b.indexOf("\n"))).join("\n");
|
|
107
113
|
const pieces = [];
|
|
114
|
+
const omittedPaths = [];
|
|
108
115
|
let total = 0;
|
|
109
|
-
let omittedFiles = 0;
|
|
110
116
|
for (const block of blocks) {
|
|
111
117
|
if (total + block.length > TOTAL_DIFF_LIMIT) {
|
|
112
|
-
|
|
118
|
+
const header = block.slice(0, block.indexOf("\n"));
|
|
119
|
+
omittedPaths.push(pathFromDiffGitHeader(header) ?? (header || "unknown"));
|
|
113
120
|
continue;
|
|
114
121
|
}
|
|
115
122
|
pieces.push(block);
|
|
116
123
|
total += block.length;
|
|
117
124
|
}
|
|
118
|
-
const truncated =
|
|
125
|
+
const truncated = omittedPaths.length > 0;
|
|
119
126
|
const footer = truncated
|
|
120
|
-
? `\n... [${
|
|
127
|
+
? `\n... [${omittedPaths.length} changed file(s) omitted — diff exceeds ${TOTAL_DIFF_LIMIT} characters. Omitted: ${omittedPaths.join(", ")}. If any omitted path is AC-critical, verdict MUST be "changes_requested" with a blocking finding listing those paths. If AC is still certifiable from the visible tip with only non_blocking findings, "approved" is allowed. Do NOT use "escalated" for truncation — escalate only with productScopeQuestion for a true product gap.]`
|
|
121
128
|
: "";
|
|
122
129
|
return {
|
|
123
130
|
text: `Changed files (${blocks.length}):\n${manifest}\n\n${pieces.join("\n")}${footer}`,
|
|
124
131
|
truncated,
|
|
132
|
+
omittedPaths,
|
|
125
133
|
};
|
|
126
134
|
}
|
|
127
135
|
export function buildReviewerPrompt(input) {
|
|
@@ -145,6 +145,27 @@ test("reviewer prompt embeds the diff, echoes the exact SHAs to report, and forb
|
|
|
145
145
|
assert.match(prompt, /"headSha" to exactly "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"/);
|
|
146
146
|
assert.match(prompt, /You cannot edit files, push, or publish anything/);
|
|
147
147
|
});
|
|
148
|
+
test("NOT-150: reviewer verdict rules match the design table (blocking ⇒ changes_requested; escalate needs productScopeQuestion)", () => {
|
|
149
|
+
const prompt = buildReviewerPrompt(reviewerBase);
|
|
150
|
+
assert.match(prompt, /"approved": AC met/);
|
|
151
|
+
assert.match(prompt, /no finding is "blocking"/);
|
|
152
|
+
assert.match(prompt, /"changes_requested": any "blocking" finding/);
|
|
153
|
+
assert.match(prompt, /"escalated": only when acceptance criteria/);
|
|
154
|
+
assert.match(prompt, /MUST set non-empty "productScopeQuestion"/);
|
|
155
|
+
assert.match(prompt, /not ordinary code defects/);
|
|
156
|
+
assert.match(prompt, /not "diff too large"/);
|
|
157
|
+
});
|
|
158
|
+
test("NOT-150: truncated-diff footer does not reject approved/changes_requested; forbids escalate-for-truncation", async () => {
|
|
159
|
+
const { formatDiffForPrompt, TOTAL_DIFF_LIMIT } = await import("./prompts.js");
|
|
160
|
+
const big = "x".repeat(TOTAL_DIFF_LIMIT + 1);
|
|
161
|
+
const diff = `diff --git a/small.ts b/small.ts\n+ok\n\ndiff --git a/huge.ts b/huge.ts\n+${big}\n`;
|
|
162
|
+
const formatted = formatDiffForPrompt(diff);
|
|
163
|
+
assert.equal(formatted.truncated, true);
|
|
164
|
+
assert.ok(formatted.omittedPaths.some((p) => p.includes("huge.ts")));
|
|
165
|
+
assert.match(formatted.text, /changes_requested/);
|
|
166
|
+
assert.match(formatted.text, /Do NOT use "escalated" for truncation/);
|
|
167
|
+
assert.doesNotMatch(formatted.text, /will not accept "approved" or "changes_requested"/);
|
|
168
|
+
});
|
|
148
169
|
test("reviewer prompt includes the developer's conclusion, checks summary, and prior findings when given", () => {
|
|
149
170
|
const prompt = buildReviewerPrompt({
|
|
150
171
|
...reviewerBase,
|
|
@@ -29,7 +29,7 @@ import { getTaskSnapshot } from "./commands.js";
|
|
|
29
29
|
import { buildReviewerPrompt, formatDiffForPrompt, TOTAL_DIFF_LIMIT } from "./prompts.js";
|
|
30
30
|
import { guidanceForNextSession } from "./guidance.js";
|
|
31
31
|
import { realReviewerSpawn, reviewerSessionLogPath } from "./spawn.js";
|
|
32
|
-
import { parseReviewerResult, ReviewerResult as ReviewerResultSchema } from "./reviewer-result.js";
|
|
32
|
+
import { parseReviewerResult, normalizeReviewerResult, INCOMPLETE_REVIEW_FINGERPRINT, ReviewerResult as ReviewerResultSchema, } from "./reviewer-result.js";
|
|
33
33
|
import { createRoleWorktree, safeRemoveWorktree, isWorktreeClean, mergeBase, fetchRef, diffShas, } from "../adapters/git-worktree.js";
|
|
34
34
|
import { ensureIssueRepoCheckout, roleWorktreePathForResolution, resolveCheckoutBaseBranch, } from "../adapters/managed-repo.js";
|
|
35
35
|
import { prepareWorkerDeckConnection, releaseWorkerDeckConnection } from "../adapters/agent-deck-bind.js";
|
|
@@ -108,6 +108,47 @@ async function bestEffortRemove(repo, worktreePath) {
|
|
|
108
108
|
// leave it for crash-recovery inspection — cleanup is a courtesy, not part of the contract
|
|
109
109
|
}
|
|
110
110
|
}
|
|
111
|
+
/** Inject a stable blocking incomplete-review finding listing omitted paths (NOT-150). */
|
|
112
|
+
function withIncompleteReviewFinding(result, omittedPaths) {
|
|
113
|
+
if (result.findings.some((f) => f.fingerprint === INCOMPLETE_REVIEW_FINGERPRINT)) {
|
|
114
|
+
const { productScopeQuestion: _drop, ...rest } = result;
|
|
115
|
+
return { ...rest, verdict: "changes_requested" };
|
|
116
|
+
}
|
|
117
|
+
const paths = omittedPaths.length > 0 ? omittedPaths.join(", ") : "(unlisted omitted files)";
|
|
118
|
+
const { productScopeQuestion: _drop, ...rest } = result;
|
|
119
|
+
return {
|
|
120
|
+
...rest,
|
|
121
|
+
verdict: "changes_requested",
|
|
122
|
+
findings: [
|
|
123
|
+
...result.findings,
|
|
124
|
+
{
|
|
125
|
+
fingerprint: INCOMPLETE_REVIEW_FINGERPRINT,
|
|
126
|
+
severity: "blocking",
|
|
127
|
+
title: "Diff truncated — incomplete review",
|
|
128
|
+
rationale: `Reviewer prompt omitted path(s): ${paths}. Cannot fully certify acceptance criteria without them; shrink the change set or split the PR so the next review sees the full AC-critical surface.`,
|
|
129
|
+
},
|
|
130
|
+
],
|
|
131
|
+
};
|
|
132
|
+
}
|
|
133
|
+
/**
|
|
134
|
+
* Truncation remap (PRD §6.4 / design NOT-150): never blind escalate → Resume|Close.
|
|
135
|
+
* Shippable approved (no blocking) stays approved; otherwise ensure changes_requested with
|
|
136
|
+
* the incomplete-review finding listing omitted paths (in addition to any other blocking).
|
|
137
|
+
*/
|
|
138
|
+
function applyTruncationVerdictPolicy(result, omittedPaths) {
|
|
139
|
+
const normalized = normalizeReviewerResult(result);
|
|
140
|
+
const hasProductQ = !!normalized.productScopeQuestion?.trim();
|
|
141
|
+
const hasBlocking = normalized.findings.some((f) => f.severity === "blocking");
|
|
142
|
+
if (normalized.verdict === "escalated" && hasProductQ) {
|
|
143
|
+
return normalized;
|
|
144
|
+
}
|
|
145
|
+
if (normalized.verdict === "approved" && !hasBlocking) {
|
|
146
|
+
return normalized;
|
|
147
|
+
}
|
|
148
|
+
// Truncated + not shippable-approved → changes_requested with incomplete-review (omitted paths).
|
|
149
|
+
const { productScopeQuestion: _drop, ...rest } = normalized;
|
|
150
|
+
return withIncompleteReviewFinding({ ...rest, verdict: "changes_requested" }, omittedPaths);
|
|
151
|
+
}
|
|
111
152
|
function readImplementationConclusion(issueId) {
|
|
112
153
|
const artifact = latestIssueArtifact(issueId, "implementation_conclusion");
|
|
113
154
|
if (!artifact?.contentJson)
|
|
@@ -281,7 +322,7 @@ export async function runReviewerEffect(ctx, deps = defaultDeps) {
|
|
|
281
322
|
await fetchRef(worktreePath, baseBranch);
|
|
282
323
|
const baseSha = await mergeBase({ repo: worktreePath, base: `origin/${baseBranch}`, head: headSha });
|
|
283
324
|
const diff = await diffShas({ worktreePath, baseSha, headSha });
|
|
284
|
-
const { truncated: diffTruncated } = formatDiffForPrompt(diff);
|
|
325
|
+
const { truncated: diffTruncated, omittedPaths } = formatDiffForPrompt(diff);
|
|
285
326
|
const openFindings = listFindingsForIssue(issue.id).filter((f) => f.status === "open" || f.status === "recurring");
|
|
286
327
|
const guidance = guidanceForNextSession(issue.id, sessionId);
|
|
287
328
|
const prompt = buildReviewerPrompt({
|
|
@@ -402,20 +443,23 @@ export async function runReviewerEffect(ctx, deps = defaultDeps) {
|
|
|
402
443
|
return { kind: "session_failed" };
|
|
403
444
|
}
|
|
404
445
|
result = parsed;
|
|
405
|
-
//
|
|
406
|
-
//
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
if (diffTruncated && result.verdict !== "escalated") {
|
|
446
|
+
// Truncation policy (PRD §6.4 / design NOT-150): never blind-escalate to Resume|Close.
|
|
447
|
+
// Persist evidence; remap bare escalate / approved+blocking; keep shippable approved.
|
|
448
|
+
if (diffTruncated) {
|
|
449
|
+
const reportedVerdict = result.verdict;
|
|
450
|
+
result = applyTruncationVerdictPolicy(result, omittedPaths);
|
|
411
451
|
createIssueArtifact({
|
|
412
452
|
issueId: issue.id,
|
|
413
453
|
workerSessionId: sessionId,
|
|
414
454
|
kind: "diff_truncated_evidence",
|
|
415
455
|
author: "system",
|
|
416
|
-
content: {
|
|
456
|
+
content: {
|
|
457
|
+
reportedVerdict,
|
|
458
|
+
overriddenTo: result.verdict,
|
|
459
|
+
diffCharLimit: TOTAL_DIFF_LIMIT,
|
|
460
|
+
omittedPaths,
|
|
461
|
+
},
|
|
417
462
|
});
|
|
418
|
-
result = { ...result, verdict: "escalated" };
|
|
419
463
|
}
|
|
420
464
|
}
|
|
421
465
|
catch {
|
|
@@ -456,7 +500,8 @@ export async function runReviewerEffect(ctx, deps = defaultDeps) {
|
|
|
456
500
|
// diff). A `row` with no recorded result (still `claimed` after every wait
|
|
457
501
|
// attempt, or the winner itself failed) has nothing safe to report — escalate.
|
|
458
502
|
if (claim.row?.state === "published" && claim.row.resultJson) {
|
|
459
|
-
const
|
|
503
|
+
const winnerParsed = ReviewerResultSchema.parse(JSON.parse(claim.row.resultJson));
|
|
504
|
+
const winnerResult = normalizeReviewerResult(winnerParsed);
|
|
460
505
|
return { kind: "verdict", result: winnerResult };
|
|
461
506
|
}
|
|
462
507
|
return { kind: "publish_failed" };
|