agent-dealer 1.2.3 → 1.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/bundle/server/dist/adapters/agent-health.js +25 -2
  2. package/bundle/server/dist/adapters/agent-health.test.js +55 -2
  3. package/bundle/server/dist/adapters/github.js +110 -12
  4. package/bundle/server/dist/adapters/github.test.js +274 -3
  5. package/bundle/server/dist/adapters/muse-capability.js +330 -0
  6. package/bundle/server/dist/adapters/muse-capability.test.js +378 -0
  7. package/bundle/server/dist/capacity/claude-local-cache.js +222 -63
  8. package/bundle/server/dist/capacity/claude-local-cache.test.js +179 -30
  9. package/bundle/server/dist/capacity/muse-host.js +111 -23
  10. package/bundle/server/dist/capacity/muse-host.test.js +125 -1
  11. package/bundle/server/dist/capacity/muse-lifecycle.test.js +2 -3
  12. package/bundle/server/dist/capacity/muse-probe.js +250 -0
  13. package/bundle/server/dist/capacity/muse-probe.test.js +183 -0
  14. package/bundle/server/dist/capacity/muse.js +16 -6
  15. package/bundle/server/dist/coordinator/admission.test.js +85 -0
  16. package/bundle/server/dist/coordinator/commands.js +35 -8
  17. package/bundle/server/dist/coordinator/developer-effect.js +40 -6
  18. package/bundle/server/dist/coordinator/developer-effect.test.js +121 -12
  19. package/bundle/server/dist/coordinator/human-resolution.js +31 -1
  20. package/bundle/server/dist/coordinator/muse-spawn.js +12 -2
  21. package/bundle/server/dist/coordinator/prompts.js +15 -0
  22. package/bundle/server/dist/coordinator/prompts.test.js +20 -0
  23. package/bundle/server/dist/coordinator/spawn.js +7 -0
  24. package/bundle/server/dist/coordinator/worktree-cwd-guard.js +48 -0
  25. package/bundle/server/dist/coordinator/worktree-cwd-guard.test.js +83 -0
  26. package/bundle/server/dist/routes/human-actions.js +10 -1
  27. package/bundle/server/dist/routes/human-actions.test.js +117 -1
  28. package/bundle/server/dist/routes/index.js +17 -14
  29. package/bundle/server/dist/routes/runtime-capacity.test.js +11 -3
  30. package/bundle/server/dist/runners/muse-serve-session.js +7 -6
  31. package/bundle/server/package.json +2 -2
  32. package/bundle/server/static-ui/assets/{index-CYqRh_-S.css → index-BII-LgB8.css} +1 -1
  33. package/bundle/server/static-ui/assets/index-Bq8wWpZm.js +60 -0
  34. package/bundle/server/static-ui/index.html +2 -2
  35. package/bundle/shared/dist/agents.d.ts +15 -15
  36. package/bundle/shared/dist/agents.js +6 -0
  37. package/bundle/shared/dist/index.d.ts +7 -7
  38. package/bundle/shared/package.json +1 -1
  39. package/dist/doctor.d.ts +22 -3
  40. package/dist/doctor.js +46 -12
  41. package/dist/doctor.test.js +41 -10
  42. package/package.json +1 -1
  43. package/bundle/server/static-ui/assets/index-UO4lHZw4.js +0 -60
@@ -4,6 +4,7 @@ import { CODEX_AUTH_REMEDIATION, MUSE_AUTH_REMEDIATION, cursorAuthIssueFromOutpu
4
4
  import { claudeBinExists, cursorBinExists, resolveClaudeBin, cursorInvokeArgs, resolveCursorBin, resolveCodexBin, codexBinExists, resolveMuseBin, museBinExists, resolveMuseAuthFile, MUSE_CLI_ENV, } from "../cli-env.js";
5
5
  import { checkAgentDeckHealth, checkAgentDeckMcpRegistration, fetchDecks, } from "./agent-deck.js";
6
6
  import { runtimeAvailability } from "../repository/runtime-availability.js";
7
+ import { museCapabilityCheckInFlight, museCapabilityIssues, museCapabilitySettleCount, parseMuseVersion, resetMuseCapabilityStateForTests, } from "./muse-capability.js";
7
8
  const RUNTIME_LABEL = {
8
9
  claude_code: "Claude",
9
10
  cursor_local: "Cursor",
@@ -112,6 +113,7 @@ export function clearAgentHealthCaches() {
112
113
  githubIssueCache = null;
113
114
  cursorSoftFailStreak = 0;
114
115
  cursorLastHealthyAt = null;
116
+ resetMuseCapabilityStateForTests();
115
117
  }
116
118
  function isSoftCursorProbeIssue(issue) {
117
119
  return (issue.code === "runtime_auth" &&
@@ -195,6 +197,11 @@ async function cursorRuntimeIssues() {
195
197
  * billed `muse exec`, so a *present* but expired login is not detectable here — it surfaces at
196
198
  * the first run, which the same classifier reads from stderr. `auth.json` is tested for
197
199
  * existence only, never read.
200
+ *
201
+ * NOT-277: with CLI and credentials present, the reported version is handed to the capability
202
+ * check — a version not yet checked runs one real developer-shell probe (muse-capability.ts);
203
+ * `runtime_capability` blocks while it runs, when it finds shell/write missing, or when it could
204
+ * not complete.
198
205
  */
199
206
  async function museRuntimeIssues() {
200
207
  const bin = resolveMuseBin();
@@ -224,7 +231,17 @@ async function museRuntimeIssues() {
224
231
  if (!process.env.META_API_KEY && !fs.existsSync(resolveMuseAuthFile())) {
225
232
  return [{ code: "runtime_auth", message: MUSE_AUTH_REMEDIATION }];
226
233
  }
227
- return [];
234
+ const version = parseMuseVersion(ver.output);
235
+ if (!version) {
236
+ return [
237
+ {
238
+ code: "runtime_unknown",
239
+ message: "Could not determine Muse Code health — `muse --version` printed no version",
240
+ },
241
+ ];
242
+ }
243
+ // A settled check drops the cached result so admission sees it on its next health read.
244
+ return museCapabilityIssues(version, () => runtimeIssueCache.delete("muse_code"));
228
245
  }
229
246
  /** Exported for direct testing — bypasses the 60s cache in runtimeIssues(). */
230
247
  export async function runtimeIssuesUncached(runtime) {
@@ -427,12 +444,18 @@ async function runtimeIssues(runtime) {
427
444
  if (cached && Date.now() - cached.at < ttl) {
428
445
  return [...capIssues, ...cached.issues];
429
446
  }
447
+ const museSettles = museCapabilitySettleCount();
430
448
  const issues = await runtimeIssuesUncached(runtime);
431
449
  const nonCap = issues.filter((i) => i.code !== "usage_capped");
450
+ // NOT-277: a capability check that settled during this read already superseded what it returned.
451
+ if (runtime === "muse_code" && museCapabilitySettleCount() !== museSettles) {
452
+ return runtimeIssues(runtime);
453
+ }
432
454
  // Soft fail (published or grace-held) uses a short TTL so a wake retry can clear quickly;
433
455
  // a sticky 60s cache of "Could not confirm" is what parked the queue after sleep (NOT-157).
434
456
  const softProbeFailure = nonCap.some(isSoftCursorProbeIssue) ||
435
- (runtime === "cursor_local" && cursorSoftFailStreak > 0);
457
+ (runtime === "cursor_local" && cursorSoftFailStreak > 0) ||
458
+ (runtime === "muse_code" && museCapabilityCheckInFlight());
436
459
  runtimeIssueCache.set(runtime, { at: Date.now(), issues: nonCap, softProbeFailure });
437
460
  return [...capIssues, ...nonCap];
438
461
  }
@@ -16,7 +16,10 @@ process.env.AGENT_DEALER_HOME = fs.mkdtempSync(path.join(os.tmpdir(), "dealer-ag
16
16
  const { migrate, getDb } = await import("../db/index.js");
17
17
  const { createAgent, getAgent } = await import("../repository/agents.js");
18
18
  const { healthForAgent, runtimeIssuesUncached, githubIssuesUncached, githubIssuesSync, classifyGithubAuthStatus, clearAgentHealthCaches, setCursorProbeTimingForTests, setRunCommandForTests, } = await import("./agent-health.js");
19
+ const { setMuseCapabilityProbeForTests, settleMuseCapabilityCheckForTests } = await import("./muse-capability.js");
19
20
  migrate();
21
+ // NOT-277: health never runs the real (billed) Muse capability probe in tests.
22
+ setMuseCapabilityProbeForTests(async () => ({ status: "capable" }));
20
23
  const FAILURE = { ok: false, code: "DECK_UNAVAILABLE", message: "Agent Deck API error: 502" };
21
24
  /** Tests inject an empty github list so host `gh auth` does not pollute assertions. */
22
25
  const NO_GITHUB = [];
@@ -414,8 +417,14 @@ The keychain item is stuck. Delete it and sign in again:
414
417
  const withLogin = emptyConfigHome();
415
418
  fs.mkdirSync(path.join(withLogin, "muse"));
416
419
  fs.writeFileSync(path.join(withLogin, "muse", "auth.json"), "{}");
417
- assert.deepEqual(await withMuseEnv({ MUSE_CLI: stub.bin, configHome: withLogin }, () => runtimeIssuesUncached("muse_code")), []);
418
- assert.deepEqual(await withMuseEnv({ MUSE_CLI: stub.bin, META_API_KEY: "k", configHome: emptyConfigHome() }, () => runtimeIssuesUncached("muse_code")), []);
420
+ // NOT-277: the first read of a version starts its one-time capability check; healthy once settled.
421
+ const settledIssues = async () => {
422
+ await runtimeIssuesUncached("muse_code");
423
+ await settleMuseCapabilityCheckForTests();
424
+ return runtimeIssuesUncached("muse_code");
425
+ };
426
+ assert.deepEqual(await withMuseEnv({ MUSE_CLI: stub.bin, configHome: withLogin }, settledIssues), []);
427
+ assert.deepEqual(await withMuseEnv({ MUSE_CLI: stub.bin, META_API_KEY: "k", configHome: emptyConfigHome() }, settledIssues), []);
419
428
  });
420
429
  test("muse probes run with MUSE_NO_AUTO_UPDATE=1 so the launcher cannot swap the pinned version", async () => {
421
430
  const stub = stubMuse("echo 'Muse Code 1.3.0 (1.3.0-R3401.1)'");
@@ -442,6 +451,50 @@ The keychain item is stuck. Delete it and sign in again:
442
451
  setCursorProbeTimingForTests(null);
443
452
  }
444
453
  });
454
+ // NOT-277: through the cached health path the agent list / admission read — a Muse auto-update
455
+ // fires the capability check exactly once for the new version, and the result (not a generic
456
+ // symptom) is what the unhealthy agent shows.
457
+ test("a Muse version change is capability-checked once and its named result surfaces on the agent", async () => {
458
+ const OLD = "1.3.0-R3401.1";
459
+ const NEW = "1.4.0-R4161.1";
460
+ const versionFile = path.join(fs.mkdtempSync(path.join(os.tmpdir(), "dealer-muse-ver-")), "v");
461
+ fs.writeFileSync(versionFile, `Muse Code 1.3.0 (${OLD})\n`);
462
+ const stub = stubMuse(`cat ${JSON.stringify(versionFile)}`);
463
+ const calls = [];
464
+ setMuseCapabilityProbeForTests(async (version) => {
465
+ calls.push(version);
466
+ return version === OLD
467
+ ? { status: "capable" }
468
+ : { status: "missing", detail: "probe session completed without running its shell command" };
469
+ });
470
+ const agent = createAgent({ name: "muse-updated", runtime: "muse_code", deckId: randomUUID() });
471
+ const health = () => healthForAgent(agent, true, undefined, true, null, NO_GITHUB);
472
+ try {
473
+ await withMuseEnv({ MUSE_CLI: stub.bin, META_API_KEY: "k", configHome: emptyConfigHome() }, async () => {
474
+ // The stub probe settles within this read; the health read must not cache its stale
475
+ // "verifying" block (the in-flight block itself is covered in muse-capability.test.ts).
476
+ await health();
477
+ await settleMuseCapabilityCheckForTests();
478
+ for (let i = 0; i < 3; i++)
479
+ assert.deepEqual((await health()).issues, []);
480
+ assert.deepEqual(calls, [OLD]);
481
+ // Muse auto-updates. The 60s runtime cache notices on its next miss.
482
+ fs.writeFileSync(versionFile, `Muse Code 1.4.0 (${NEW})\n`);
483
+ await runtimeIssuesUncached("muse_code");
484
+ await settleMuseCapabilityCheckForTests();
485
+ for (let i = 0; i < 3; i++) {
486
+ const after = await health();
487
+ assert.equal(after.healthy, false);
488
+ assert.deepEqual(after.issues.map((x) => x.code), ["runtime_capability"]);
489
+ assert.match(after.issues[0].message, new RegExp(`Muse Code updated ${OLD} → ${NEW}: developer sessions no longer get shell/write access`));
490
+ }
491
+ assert.deepEqual(calls, [OLD, NEW]);
492
+ });
493
+ }
494
+ finally {
495
+ setMuseCapabilityProbeForTests(async () => ({ status: "capable" }));
496
+ }
497
+ });
445
498
  test("claude MCP endpoint mismatch surfaces a distinct message, not the generic setup hint", async () => {
446
499
  const agent = createAgent({
447
500
  name: "claude-mcp-mismatch",
@@ -70,8 +70,23 @@ export const CHECKS_FAILURE_GENERIC_REASON = "Developer's PR checks failed.";
70
70
  export const CHECKS_EVIDENCE_MAX_FAILED_CHECKS = 10;
71
71
  /** Max distinct Actions runs whose logs are fetched (one `gh run view` per run). */
72
72
  export const CHECKS_EVIDENCE_MAX_RUNS = 5;
73
- /** Per-run cap on fetched log text before excerpt focusing (tail kept — failures surface last). */
74
- export const CHECKS_EVIDENCE_MAX_LOG_CHARS_PER_RUN = 20_000;
73
+ /**
74
+ * NOT-276: raw per-run ceiling (2MB) on fetched log text actually held in
75
+ * memory — purely a guard against pathologically huge logs, NOT a "search only
76
+ * the tail" window. `fetchChecksFailureEvidence` passes everything under this
77
+ * ceiling untruncated into `buildFailureExcerpt`, whose failure-pattern search
78
+ * plus the per-excerpt line/char caps already bound the output.
79
+ */
80
+ export const CHECKS_EVIDENCE_MAX_RAW_LOG_CHARS_PER_RUN = 2_000_000;
81
+ /**
82
+ * NOT-276 round 3: explicit `maxBuffer` (bytes) for the `gh run view --log-failed`
83
+ * fetch. Node's `execFile` defaults to ~1 MiB, which would reject any log over
84
+ * that size with ERR_CHILD_PROCESS_STDIO_MAXBUFFER before the 2MB raw-log
85
+ * ceiling above ever applies. This buffer sits above the ceiling (plus headroom
86
+ * for stderr bytes and multi-byte chars) so the ceiling — not the process
87
+ * buffer — is what bounds memory.
88
+ */
89
+ export const CHECKS_EVIDENCE_GH_LOG_MAX_BUFFER = 4_000_000;
75
90
  /** Single global cap on the focused excerpt threaded into the retry prompt. */
76
91
  export const CHECKS_EVIDENCE_MAX_EXCERPT_CHARS = 4_000;
77
92
  /** Max lines in the focused excerpt; context lines kept around each failure line. */
@@ -148,11 +163,35 @@ export function sanitizeCiText(text) {
148
163
  return out;
149
164
  }
150
165
  const FAILURE_LINE_PATTERN = /error|err!|e404|fail|fatal|exception|traceback|assert|not found|cannot |can't |unable |conflict|reject|denied|panic|timed?\s*out|npm ERR!/i;
166
+ /**
167
+ * Explicit failure markers, not just failure-adjacent vocabulary: TAP's own `not ok`
168
+ * result line (the TAP spec's standard fail marker, used by many test harnesses beyond
169
+ * `node:test` — not a runner-specific parser) and GitHub Actions' own `##[error]`
170
+ * workflow-command annotation (emitted by the platform itself for a genuinely failed
171
+ * step, regardless of what tool ran in it). A coordinator/CI-tooling repo's own test
172
+ * suite is full of passing tests *about* failure handling — "conflict", "timeout",
173
+ * "denied", "escalates" — so `FAILURE_LINE_PATTERN` density alone is not reliable
174
+ * (NOT-276 round 3: verified against the real run-36330128633 log, a cluster of
175
+ * passing tests named around escalation/conflict/timeout out-ranked the actual
176
+ * `not ok 355`/`356` failure under pure density ranking). These two markers get a
177
+ * large ranking bonus below so a window that contains one always wins.
178
+ */
179
+ // Not anchored to line start: `gh run view --log-failed` prefixes every line with
180
+ // `<job>\t<step>\t<timestamp> ` before the actual tool output, so the TAP/annotation
181
+ // text never starts at column 0.
182
+ const STRONG_FAILURE_LINE_PATTERN = /\bnot ok\b|##\[error\]/i;
183
+ const STRONG_HIT_WEIGHT = 1000;
151
184
  /**
152
185
  * Focus a (sanitized) log around its useful failure/error lines: keep a small context
153
- * window around each matching line, merge overlapping windows, collapse long runs of
154
- * identical lines (CI setup spam), then enforce the global line/char caps. With no
155
- * matching line, the tail is the most likely failure site. Never returns unsanitized text.
186
+ * window around each matching line, merge overlapping windows, rank merged regions by
187
+ * weighted hit density (explicit failure markers far outweigh failure-adjacent
188
+ * vocabulary; ties break on density, then log order) so an early real failure cluster
189
+ * is not crowded out of the line budget by sparse isolated mentions in passing-test
190
+ * names, or by a *dense* but merely topical cluster of passing tests about failure
191
+ * handling itself (NOT-276 round 2 found the former, round 3 the latter — see
192
+ * `STRONG_FAILURE_LINE_PATTERN`'s comment), then collapse long runs of identical lines
193
+ * (CI setup spam) and enforce the global line/char caps. With no matching line, the
194
+ * tail is the most likely failure site. Never returns unsanitized text.
156
195
  */
157
196
  export function buildFailureExcerpt(combinedLog) {
158
197
  const sanitized = sanitizeCiText(combinedLog);
@@ -160,9 +199,14 @@ export function buildFailureExcerpt(combinedLog) {
160
199
  if (lines.every((l) => !l.trim()))
161
200
  return { excerpt: "", truncated: false };
162
201
  const hits = [];
202
+ const strongHits = new Set();
163
203
  lines.forEach((line, i) => {
164
- if (FAILURE_LINE_PATTERN.test(line))
204
+ const strong = STRONG_FAILURE_LINE_PATTERN.test(line);
205
+ if (strong || FAILURE_LINE_PATTERN.test(line)) {
165
206
  hits.push(i);
207
+ if (strong)
208
+ strongHits.add(i);
209
+ }
166
210
  });
167
211
  let selected;
168
212
  let truncated = false;
@@ -180,8 +224,54 @@ export function buildFailureExcerpt(combinedLog) {
180
224
  else
181
225
  merged.push([w[0], w[1]]);
182
226
  }
227
+ // Rank merged regions by weighted hit density so the excerpt budget goes to the
228
+ // most failure-indicative clusters first: an explicit failure marker (`not ok`,
229
+ // `##[error]`) counts for STRONG_HIT_WEIGHT, everything else for 1 — a window
230
+ // with one real marker always outranks a window with many topical-only hits.
231
+ // Ties keep log order, and output is re-sorted to log order for readability.
232
+ // `hits` is ascending (built in line order) and `merged` is ascending by `from`,
233
+ // so a single linear pass sums weight per region.
234
+ let hi = 0;
235
+ const ranked = merged.map(([from, to]) => {
236
+ while (hi < hits.length && hits[hi] < from)
237
+ hi++;
238
+ let weight = 0;
239
+ let k = hi;
240
+ while (k < hits.length && hits[k] <= to) {
241
+ weight += strongHits.has(hits[k]) ? STRONG_HIT_WEIGHT : 1;
242
+ k++;
243
+ }
244
+ return { from, to, weight };
245
+ });
246
+ ranked.sort((a, b) => b.weight - a.weight || a.from - b.from);
247
+ const taken = [];
248
+ let used = 0;
249
+ let usedChars = 0;
250
+ for (const w of ranked) {
251
+ const size = w.to - w.from + 1 + (taken.length > 0 ? 1 : 0);
252
+ let chars = taken.length > 0 ? 3 : 0; // "...\n" separator
253
+ for (let i = w.from; i <= w.to; i++)
254
+ chars += lines[i].length + 1;
255
+ // Skip regions that no longer fit — by line count or by character count, since
256
+ // the final excerpt is char-capped too (NOT-276 round 3: a lower-ranked but
257
+ // verbose region taken first would otherwise still crowd a higher-ranked
258
+ // region's content out of the *character* budget after chronological
259
+ // reassembly below, even though ranking correctly gave the real failure
260
+ // priority for line-budget inclusion). A smaller later region can still fill
261
+ // whatever budget remains — but always take the top-ranked region even if it
262
+ // alone exceeds either budget (the caps below truncate it, as before).
263
+ if (taken.length > 0 &&
264
+ (used + size > CHECKS_EVIDENCE_MAX_EXCERPT_LINES || usedChars + chars > CHECKS_EVIDENCE_MAX_EXCERPT_CHARS)) {
265
+ truncated = true;
266
+ continue;
267
+ }
268
+ taken.push([w.from, w.to]);
269
+ used += size;
270
+ usedChars += chars;
271
+ }
272
+ taken.sort((a, b) => a[0] - b[0]);
183
273
  const picked = [];
184
- merged.forEach(([from, to], idx) => {
274
+ taken.forEach(([from, to], idx) => {
185
275
  if (idx > 0)
186
276
  picked.push("...");
187
277
  for (let i = from; i <= to; i++)
@@ -241,7 +331,7 @@ export function formatChecksFailureDetails(opts) {
241
331
  return parts.join("\n");
242
332
  }
243
333
  const NO_COMMITS_PATTERN = /no commits between/i;
244
- const defaultExec = (args, opts) => run("gh", args, opts);
334
+ const defaultExec = (args, opts) => run("gh", args, { cwd: opts.cwd, maxBuffer: opts.maxBuffer });
245
335
  async function ghPrView(exec, cwd, fields, selector) {
246
336
  const args = ["pr", "view", ...(selector != null ? [selector] : []), "--json", fields];
247
337
  try {
@@ -354,11 +444,19 @@ export async function fetchChecksFailureEvidence(exec, opts) {
354
444
  let logsUnavailable = false;
355
445
  for (const runId of runIds) {
356
446
  try {
357
- const { stdout } = await exec(["run", "view", runId, "--log-failed"], { cwd: opts.cwd });
358
- const tail = stdout.length > CHECKS_EVIDENCE_MAX_LOG_CHARS_PER_RUN
359
- ? stdout.slice(-CHECKS_EVIDENCE_MAX_LOG_CHARS_PER_RUN)
447
+ const { stdout } = await exec(["run", "view", runId, "--log-failed"], {
448
+ cwd: opts.cwd,
449
+ maxBuffer: CHECKS_EVIDENCE_GH_LOG_MAX_BUFFER,
450
+ });
451
+ // NOT-276: search before truncating — the full fetched log feeds
452
+ // `buildFailureExcerpt`'s failure-pattern search, so an early failure is
453
+ // never discarded by a small tail window. Only logs beyond the raw memory
454
+ // ceiling are cut at all (tail kept), and the excerpt caps still bound
455
+ // the final output.
456
+ const full = stdout.length > CHECKS_EVIDENCE_MAX_RAW_LOG_CHARS_PER_RUN
457
+ ? stdout.slice(-CHECKS_EVIDENCE_MAX_RAW_LOG_CHARS_PER_RUN)
360
458
  : stdout;
361
- logsByRun.set(runId, tail);
459
+ logsByRun.set(runId, full);
362
460
  }
363
461
  catch {
364
462
  logsUnavailable = true;
@@ -1,7 +1,7 @@
1
1
  // packages/server/src/adapters/github.test.ts
2
2
  import { test } from "node:test";
3
3
  import assert from "node:assert/strict";
4
- import { parsePrView, summarizeChecks, pollPrChecks, createGithubAdapter, fetchChecksFailureEvidence, sanitizeCiText, sanitizeUrl, extractActionsRunId, buildFailureExcerpt, formatChecksFailureDetails, CHECKS_EVIDENCE_MAX_EXCERPT_CHARS, CHECKS_FAILURE_GENERIC_REASON, PR_VIEW_FIELDS, } from "./github.js";
4
+ import { parsePrView, summarizeChecks, pollPrChecks, createGithubAdapter, fetchChecksFailureEvidence, sanitizeCiText, sanitizeUrl, extractActionsRunId, buildFailureExcerpt, formatChecksFailureDetails, CHECKS_EVIDENCE_MAX_EXCERPT_CHARS, CHECKS_EVIDENCE_MAX_EXCERPT_LINES, CHECKS_EVIDENCE_MAX_RAW_LOG_CHARS_PER_RUN, CHECKS_EVIDENCE_GH_LOG_MAX_BUFFER, CHECKS_FAILURE_GENERIC_REASON, PR_VIEW_FIELDS, } from "./github.js";
5
5
  test("parsePrView extracts the ground-truth handoff fields, including draft status", () => {
6
6
  const view = parsePrView(JSON.stringify({
7
7
  number: 7,
@@ -34,15 +34,17 @@ test("summarizeChecks accepts neutral/skipped as success but fails closed on an
34
34
  /** Records every `gh` invocation and returns responses off a queue — never calls real `gh`. */
35
35
  function queuedExec(responses) {
36
36
  const calls = [];
37
+ const execOpts = [];
37
38
  const queue = [...responses];
38
- const exec = async (args) => {
39
+ const exec = async (args, opts) => {
39
40
  calls.push(args);
41
+ execOpts.push(opts);
40
42
  const next = queue.shift() ?? { stdout: "" };
41
43
  if (next.error != null)
42
44
  throw Object.assign(new Error(next.error), { stderr: next.error });
43
45
  return { stdout: next.stdout ?? "" };
44
46
  };
45
- return { exec, calls };
47
+ return { exec, calls, execOpts };
46
48
  }
47
49
  test("viewPr looks up the PR explicitly by branch, never a bare `gh pr view`", async () => {
48
50
  const { exec, calls } = queuedExec([
@@ -426,6 +428,275 @@ test("NOT-252: extractActionsRunId and sanitizeUrl helpers", async () => {
426
428
  assert.equal(sanitizeUrl("https://example.com/a?b=1#c"), "https://example.com/a");
427
429
  assert.equal(sanitizeUrl("https://example.com/a"), "https://example.com/a");
428
430
  });
431
+ // --- NOT-276: search-before-truncate — an early failure must survive excerpt building ---
432
+ // Replays the NOT-273 incident shape (Actions run 36330128633 "Unit tests" step):
433
+ // 1667 TAP lines, `not ok 355/356` + `cancelledByParent` at ~21% through the log,
434
+ // followed by ~1300 passing-test lines. Under the old last-20K-chars pre-truncation
435
+ // the failure sat outside the kept tail and the excerpt showed only passing
436
+ // tail noise; with search-before-truncate it must surface the failure instead.
437
+ // Faithful-scale substitute for the real `gh run view 36330128633 --log-failed`
438
+ // replay (no network/gh in this sandbox — the PR description must still record a
439
+ // replay against the saved real "Unit tests" log showing `not ok 355`/`356`).
440
+ // Crucially, ~21 passing lines BEFORE the failure carry realistic
441
+ // failure-pattern words in their test names (real TAP `ok` lines do this — e.g.
442
+ // "handles error ..."), spaced >13 lines apart so each is an isolated hit region:
443
+ // without hit-density ranking, those early weak hits fill the 80-line budget
444
+ // ahead of the real failure cluster (round-2 blocking finding). The dense
445
+ // failure block (failureType x2 + error: x1 within 9 lines) must outrank them.
446
+ test("NOT-276: failure line before the last 20K chars still reaches the excerpt", async () => {
447
+ const early = [];
448
+ for (let n = 1; n <= 354; n++) {
449
+ // Isolated pattern hits every 16 lines (> 2x the 6-line context radius, so
450
+ // windows never merge): ~22 weak hits precede the real failure.
451
+ if (n % 16 === 0)
452
+ early.push(`ok ${n} - handles error output for test ${n}`);
453
+ else if (n === 353)
454
+ early.push("ok 353 - reports error when child fails to spawn");
455
+ else if (n === 354)
456
+ early.push("ok 354 - cleans up after failure");
457
+ else
458
+ early.push(`ok ${n} - passing test number ${n}`);
459
+ }
460
+ const failureBlock = [
461
+ "not ok 355 - coordinator spawns child with explicit cwd",
462
+ " ---",
463
+ " failureType: 'cancelledByParent'",
464
+ " error: test cancelled by parent",
465
+ " ---",
466
+ "not ok 356 - coordinator spawns child with explicit cwd (2)",
467
+ " ---",
468
+ " failureType: 'cancelledByParent'",
469
+ " ---",
470
+ ].join("\n");
471
+ const filler = Array.from({ length: 1667 - 356 }, (_, i) => {
472
+ const n = 357 + i;
473
+ if (i % 200 === 0)
474
+ return `ok ${n} - handles error output for test ${n}`;
475
+ if (i % 150 === 0)
476
+ return `ok ${n} - cleans up after failure ${n}`;
477
+ return `ok ${n} - passing test number ${n}`;
478
+ }).join("\n");
479
+ const log = `${early.join("\n")}\n${failureBlock}\n${filler}`;
480
+ // Guard the test's premise: the failure really does sit outside the old 20K tail window.
481
+ assert.ok(log.indexOf("not ok 355") < log.length - 20_000);
482
+ const { exec } = queuedExec([
483
+ { stdout: prViewWithRollup([actionsCheck("verify", "FAILURE", "36330128633")]) },
484
+ { stdout: log },
485
+ ]);
486
+ const evidence = await fetchChecksFailureEvidence(exec, { cwd: "/repo", number: 160, expectedHeadSha: HEAD_SHA });
487
+ assert.ok(evidence);
488
+ assert.match(evidence.excerpt, /not ok 355/);
489
+ assert.match(evidence.excerpt, /not ok 356/);
490
+ assert.match(evidence.excerpt, /cancelledByParent/);
491
+ assert.doesNotMatch(evidence.excerpt, /passing test number 1667/);
492
+ assert.ok(evidence.excerpt.length <= CHECKS_EVIDENCE_MAX_EXCERPT_CHARS);
493
+ assert.ok(evidence.excerpt.split("\n").length <= CHECKS_EVIDENCE_MAX_EXCERPT_LINES);
494
+ });
495
+ // NOT-276 round-2 replay stand-in (blocking finding: the real `gh run view
496
+ // 36330128633 --log-failed` replay needs network/gh, unavailable in CI/sandbox,
497
+ // so this mirrors its exact byte shape at faithful scale instead): every line
498
+ // carries the real `<job>\t<step>\t<timestamp> ` prefix, 1667 TAP results with
499
+ // `not ok 355`/`356` + `cancelledByParent` at ~21% through, ~24 passing lines
500
+ // before the failure carrying failure-pattern words in their names, >20K
501
+ // chars of passing-test tail after it, and the node:test TAP summary block at
502
+ // the true end of the log (the root-cause mechanism: node:test runs to
503
+ // completion, so `# fail 2` — a weak tail hit — sits after thousands of passing
504
+ // lines). The excerpt must surface the early strong failure, not the tail.
505
+ test("NOT-276 round-2 replay: full-scale Actions-prefixed log with an early `not ok` failure", async () => {
506
+ const bodies = [];
507
+ for (let n = 1; n <= 354; n++) {
508
+ if (n % 16 === 0)
509
+ bodies.push(`ok ${n} - handles error output for test ${n}`);
510
+ else if (n === 353)
511
+ bodies.push("ok 353 - reports error when child fails to spawn");
512
+ else if (n === 354)
513
+ bodies.push("ok 354 - cleans up after failure");
514
+ else
515
+ bodies.push(`ok ${n} - passing test number ${n}`);
516
+ }
517
+ bodies.push("not ok 355 - coordinator spawns child with explicit cwd", " ---", " failureType: 'cancelledByParent'", " error: test cancelled by parent", " ---", "not ok 356 - coordinator spawns child with explicit cwd (2)", " ---", " failureType: 'cancelledByParent'", " ---");
518
+ for (let n = 357; n <= 1667; n++) {
519
+ const i = n - 357;
520
+ if (i % 200 === 0)
521
+ bodies.push(`ok ${n} - handles error output for test ${n}`);
522
+ else if (i % 150 === 0)
523
+ bodies.push(`ok ${n} - cleans up after failure ${n}`);
524
+ else
525
+ bodies.push(`ok ${n} - passing test number ${n}`);
526
+ }
527
+ // node:test's own end-of-run TAP summary, as in the real incident log — its
528
+ // `# fail 2` line is a weak failure-pattern hit in the tail that must not
529
+ // outrank the early strong `not ok` failure (round-2 blocking finding).
530
+ bodies.push("# tests 1667", "# suites 4", "# pass 1663", "# fail 2", "# cancelled 2", "# skipped 0", "# todo 0", "# duration_ms 184213.015");
531
+ const log = bodies
532
+ .map((b, i) => `verify\tUnit tests\t2026-09-27T16:${String(Math.floor(i / 60)).padStart(2, "0")}:${String(i % 60).padStart(2, "0")}Z ${b}`)
533
+ .join("\n");
534
+ // Guard the test's premise: the failure really does sit outside the old 20K tail window.
535
+ assert.ok(log.indexOf("not ok 355") < log.length - 20_000);
536
+ const { exec } = queuedExec([
537
+ { stdout: prViewWithRollup([actionsCheck("verify", "FAILURE", "36330128633")]) },
538
+ { stdout: log },
539
+ ]);
540
+ const evidence = await fetchChecksFailureEvidence(exec, { cwd: "/repo", number: 160, expectedHeadSha: HEAD_SHA });
541
+ assert.ok(evidence);
542
+ assert.match(evidence.excerpt, /not ok 355/);
543
+ assert.match(evidence.excerpt, /not ok 356/);
544
+ assert.match(evidence.excerpt, /cancelledByParent/);
545
+ assert.doesNotMatch(evidence.excerpt, /passing test number 1667/);
546
+ assert.ok(evidence.excerpt.length <= CHECKS_EVIDENCE_MAX_EXCERPT_CHARS);
547
+ assert.ok(evidence.excerpt.split("\n").length <= CHECKS_EVIDENCE_MAX_EXCERPT_LINES);
548
+ });
549
+ test("NOT-276: dense failure cluster outranks sparse earlier weak hits (unit level)", async () => {
550
+ // Minimal direct proof of the round-2 ranking: 30 isolated weak hits precede one
551
+ // dense 3-hit failure cluster; the excerpt must contain the cluster.
552
+ const lines = [];
553
+ for (let g = 0; g < 30; g++) {
554
+ for (let k = 0; k < 15; k++)
555
+ lines.push(`ok ${g * 16 + k + 1} - passing test number ${g * 16 + k + 1}`);
556
+ lines.push(`ok ${g * 16 + 16} - handles error output for test ${g * 16 + 16}`);
557
+ }
558
+ lines.push("not ok 999 - real failure here", " failureType: 'cancelledByParent'", " error: test cancelled by parent");
559
+ for (let n = 1000; n < 1100; n++)
560
+ lines.push(`ok ${n} - passing test number ${n}`);
561
+ const { excerpt } = buildFailureExcerpt(lines.join("\n"));
562
+ assert.match(excerpt, /not ok 999/);
563
+ assert.match(excerpt, /cancelledByParent/);
564
+ });
565
+ // NOT-276 round 3 (real PR #163 review, verified against the actual saved
566
+ // `gh run view 36330128633 --log-failed` output): plain hit-density ranking still
567
+ // missed the real failure two different ways. (1) `gh run view --log-failed`
568
+ // prefixes every line with `<job>\t<step>\t<timestamp> ` before the tool's own
569
+ // output, so an anchored `^\s*not ok\b` pattern never matches the real thing — only
570
+ // an unanchored one does. (2) This repo's own coordinator tests are *about*
571
+ // escalation/conflict/timeout handling, so a cluster of unrelated passing tests can
572
+ // out-rank the real (sparse) failure under density alone; an explicit marker
573
+ // (`not ok`, `##[error]`) must dominate ranking regardless of density.
574
+ test("NOT-276 round 3: a GitHub Actions log-line prefix must not hide the `not ok` marker", async () => {
575
+ const prefix = (ts) => `verify\tUnit tests\t${ts}Z `;
576
+ const early = Array.from({ length: 60 }, (_, i) => `${prefix(`2026-09-27T16:07:${String(i).padStart(2, "0")}.0000000`)}ok ${i + 1} - passing test number ${i + 1}`).join("\n");
577
+ const failureBlock = [
578
+ `${prefix("2026-09-27T16:07:31.3990644")}not ok 355 - NOT-273: the serve child spawns with an explicit cwd`,
579
+ `${prefix("2026-09-27T16:07:31.3991406")} failureType: 'cancelledByParent'`,
580
+ `${prefix("2026-09-27T16:07:31.3992743")}not ok 356 - no credential short-circuits to missing without spawning`,
581
+ `${prefix("2026-09-27T16:07:31.3993384")} failureType: 'cancelledByParent'`,
582
+ ].join("\n");
583
+ const late = Array.from({ length: 60 }, (_, i) => `${prefix(`2026-09-27T16:08:${String(i).padStart(2, "0")}.0000000`)}ok ${357 + i} - passing test number ${357 + i}`).join("\n");
584
+ const log = `${early}\n${failureBlock}\n${late}\n${prefix("2026-09-27T16:09:02.6611414")}##[error]Process completed with exit code 1.`;
585
+ const { excerpt } = buildFailureExcerpt(log);
586
+ assert.match(excerpt, /not ok 355/);
587
+ assert.match(excerpt, /not ok 356/);
588
+ assert.match(excerpt, /cancelledByParent/);
589
+ });
590
+ // NOT-276 round 3: a lower-ranked region taken first can still be chronologically
591
+ // *earlier* than a higher-ranked one; if the excerpt were char-capped by slicing
592
+ // the reassembled string's tail (ignoring rank), the earlier low-priority region
593
+ // could crowd the later high-priority region's content out of the character
594
+ // budget even though line-budget ranking correctly preferred the real failure.
595
+ test("NOT-276 round 3: an earlier low-priority region must not crowd the real failure out of the character budget", async () => {
596
+ // ~12 weak hits (one per line, "failed"/"error"), long lines: ~12 * 460 ≈ 5500 chars.
597
+ const earlyWeakCluster = Array.from({ length: 12 }, (_, i) => `ok ${i + 1} - reports a handled failure/error path for scenario ${i + 1} ` + "x".repeat(400)).join("\n");
598
+ // The real failure: 2 strong `not ok` hits, short lines (~450 chars total).
599
+ const realFailure = [
600
+ "not ok 999 - the real failure",
601
+ " failureType: 'cancelledByParent'",
602
+ "not ok 1000 - a second real failure",
603
+ " failureType: 'cancelledByParent'",
604
+ ].join("\n");
605
+ // A gap of ordinary passing lines keeps the two clusters as separate merged
606
+ // windows (matching the real incident: the weak cluster and the real failure
607
+ // sat ~2300 lines apart) rather than merging into one oversized window, which
608
+ // would exercise a different, narrower edge case than the one under test here.
609
+ const gap = Array.from({ length: 20 }, (_, i) => `ok ${900 + i} - passing test number ${900 + i}`).join("\n");
610
+ const log = `${earlyWeakCluster}\n${gap}\n${realFailure}`;
611
+ assert.ok(earlyWeakCluster.length + realFailure.length > CHECKS_EVIDENCE_MAX_EXCERPT_CHARS, "test premise: both regions together exceed the char budget");
612
+ const { excerpt, truncated } = buildFailureExcerpt(log);
613
+ assert.match(excerpt, /not ok 999/);
614
+ assert.match(excerpt, /not ok 1000/);
615
+ assert.match(excerpt, /cancelledByParent/);
616
+ assert.ok(excerpt.length <= CHECKS_EVIDENCE_MAX_EXCERPT_CHARS);
617
+ assert.equal(truncated, true, "dropping the lower-ranked region must be reported as truncation");
618
+ });
619
+ test("NOT-276: an exact line-budget fill still reports a later failure region as truncated", async () => {
620
+ // The A and B windows contain 39 and 40 lines respectively. With the separator
621
+ // between them, taking both fills the 80-line budget exactly. Region C is a real
622
+ // failure that must be omitted, and that omission must set truncated=true.
623
+ const ordinary = (label, count) => Array.from({ length: count }, (_, i) => `ok ${label}-${i + 1} - passing test`);
624
+ const regionA = Array.from({ length: 27 }, (_, i) => `not ok A-${i + 1} - strong failure A`);
625
+ const regionB = Array.from({ length: 28 }, (_, i) => `not ok B-${i + 1} - strong failure B`);
626
+ const lines = [
627
+ ...ordinary("prefix", 6),
628
+ ...regionA,
629
+ ...ordinary("gap-a-b", 13),
630
+ ...regionB,
631
+ ...ordinary("gap-b-c", 13),
632
+ "not ok C-1 - later real failure",
633
+ ...ordinary("suffix", 6),
634
+ ];
635
+ const { excerpt, truncated } = buildFailureExcerpt(lines.join("\n"));
636
+ assert.equal(excerpt.split("\n").length, CHECKS_EVIDENCE_MAX_EXCERPT_LINES);
637
+ assert.match(excerpt, /not ok A-1/);
638
+ assert.match(excerpt, /not ok B-1/);
639
+ assert.doesNotMatch(excerpt, /not ok C-1/);
640
+ assert.equal(truncated, true, "the omitted third failure region must be reported as truncation");
641
+ });
642
+ test("NOT-276: log with no failure-pattern hits still falls back to the tail, unchanged", async () => {
643
+ // NB: this filler must stay free of FAILURE_LINE_PATTERN words — that absence is the no-hit premise.
644
+ const log = Array.from({ length: 200 }, (_, i) => `ok ${i + 1} - passing test number ${i + 1}`).join("\n");
645
+ const { exec } = queuedExec([
646
+ { stdout: prViewWithRollup([actionsCheck("verify", "FAILURE", "111")]) },
647
+ { stdout: log },
648
+ ]);
649
+ const evidence = await fetchChecksFailureEvidence(exec, { cwd: "/repo", number: 42, expectedHeadSha: HEAD_SHA });
650
+ assert.ok(evidence);
651
+ // The fetch path must feed the full log through exactly as buildFailureExcerpt sees it.
652
+ assert.equal(evidence.excerpt, buildFailureExcerpt(log).excerpt);
653
+ const excerptLines = evidence.excerpt.split("\n");
654
+ assert.equal(excerptLines.length, CHECKS_EVIDENCE_MAX_EXCERPT_LINES);
655
+ assert.match(excerptLines[0] ?? "", /passing test number 121/);
656
+ assert.match(excerptLines[excerptLines.length - 1] ?? "", /passing test number 200/);
657
+ assert.equal(evidence.excerptTruncated, true);
658
+ assert.ok(evidence.excerpt.length <= CHECKS_EVIDENCE_MAX_EXCERPT_CHARS);
659
+ });
660
+ test("NOT-276: log larger than the raw memory ceiling stays bounded without crashing", async () => {
661
+ // Short lines on purpose: the 80-line tail must fit under the 4,000-char
662
+ // excerpt cap, otherwise the char cap (not the memory ceiling) would cut the
663
+ // asserted last line and the test would prove nothing about the tail.
664
+ const line = (i) => `ok ${i} - pad pad ${i}`;
665
+ const targetLen = CHECKS_EVIDENCE_MAX_RAW_LOG_CHARS_PER_RUN + 500_000;
666
+ const count = Math.ceil(targetLen / 20);
667
+ const parts = new Array(count);
668
+ for (let i = 0; i < count; i++)
669
+ parts[i] = line(i);
670
+ const log = parts.join("\n");
671
+ assert.ok(log.length > CHECKS_EVIDENCE_MAX_RAW_LOG_CHARS_PER_RUN);
672
+ const { exec } = queuedExec([
673
+ { stdout: prViewWithRollup([actionsCheck("verify", "FAILURE", "111")]) },
674
+ { stdout: log },
675
+ ]);
676
+ const evidence = await fetchChecksFailureEvidence(exec, { cwd: "/repo", number: 42, expectedHeadSha: HEAD_SHA });
677
+ assert.ok(evidence);
678
+ assert.ok(evidence.excerpt.length <= CHECKS_EVIDENCE_MAX_EXCERPT_CHARS);
679
+ // The excerpt is built from the kept tail portion, never the discarded head.
680
+ assert.equal(evidence.excerpt, buildFailureExcerpt(log.slice(-CHECKS_EVIDENCE_MAX_RAW_LOG_CHARS_PER_RUN)).excerpt);
681
+ assert.match(evidence.excerpt, new RegExp(`pad pad ${count - 1}`));
682
+ });
683
+ test("NOT-276 round 3: run-log fetch carries an explicit maxBuffer above the raw ceiling", async () => {
684
+ // Node's execFile defaults to ~1 MiB maxBuffer, which would reject any larger
685
+ // --log-failed output with ERR_CHILD_PROCESS_STDIO_MAXBUFFER before the 2MB
686
+ // raw-log ceiling applies. The fetch must pass an explicit buffer above the
687
+ // ceiling so the ceiling — not the process buffer — bounds memory.
688
+ assert.ok(CHECKS_EVIDENCE_GH_LOG_MAX_BUFFER > CHECKS_EVIDENCE_MAX_RAW_LOG_CHARS_PER_RUN, "maxBuffer must sit above the raw-log ceiling");
689
+ const { exec, calls, execOpts } = queuedExec([
690
+ { stdout: prViewWithRollup([actionsCheck("verify", "FAILURE", "111")]) },
691
+ { stdout: "verify log\nError: boom\n" },
692
+ ]);
693
+ const evidence = await fetchChecksFailureEvidence(exec, { cwd: "/repo", number: 42, expectedHeadSha: HEAD_SHA });
694
+ assert.ok(evidence);
695
+ assert.deepEqual(calls[1], ["run", "view", "111", "--log-failed"]);
696
+ assert.equal(execOpts[1]?.maxBuffer, CHECKS_EVIDENCE_GH_LOG_MAX_BUFFER);
697
+ // The small pr-view lookup needs no oversized buffer.
698
+ assert.ok(execOpts[0]?.maxBuffer == null, "pr view must not carry the log buffer");
699
+ });
429
700
  test("NOT-252: formatChecksFailureDetails labels the excerpt as untrusted, not instructions", async () => {
430
701
  const details = formatChecksFailureDetails({
431
702
  headSha: HEAD_SHA,