agent-dealer 1.2.3 → 1.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bundle/server/dist/adapters/agent-health.js +25 -2
- package/bundle/server/dist/adapters/agent-health.test.js +55 -2
- package/bundle/server/dist/adapters/github.js +110 -12
- package/bundle/server/dist/adapters/github.test.js +274 -3
- package/bundle/server/dist/adapters/muse-capability.js +330 -0
- package/bundle/server/dist/adapters/muse-capability.test.js +378 -0
- package/bundle/server/dist/capacity/claude-local-cache.js +222 -63
- package/bundle/server/dist/capacity/claude-local-cache.test.js +179 -30
- package/bundle/server/dist/capacity/muse-host.js +111 -23
- package/bundle/server/dist/capacity/muse-host.test.js +125 -1
- package/bundle/server/dist/capacity/muse-lifecycle.test.js +2 -3
- package/bundle/server/dist/capacity/muse-probe.js +250 -0
- package/bundle/server/dist/capacity/muse-probe.test.js +183 -0
- package/bundle/server/dist/capacity/muse.js +16 -6
- package/bundle/server/dist/coordinator/admission.test.js +85 -0
- package/bundle/server/dist/coordinator/commands.js +35 -8
- package/bundle/server/dist/coordinator/developer-effect.js +40 -6
- package/bundle/server/dist/coordinator/developer-effect.test.js +121 -12
- package/bundle/server/dist/coordinator/human-resolution.js +31 -1
- package/bundle/server/dist/coordinator/muse-spawn.js +12 -2
- package/bundle/server/dist/coordinator/prompts.js +15 -0
- package/bundle/server/dist/coordinator/prompts.test.js +20 -0
- package/bundle/server/dist/coordinator/spawn.js +7 -0
- package/bundle/server/dist/coordinator/worktree-cwd-guard.js +48 -0
- package/bundle/server/dist/coordinator/worktree-cwd-guard.test.js +83 -0
- package/bundle/server/dist/routes/human-actions.js +10 -1
- package/bundle/server/dist/routes/human-actions.test.js +117 -1
- package/bundle/server/dist/routes/index.js +17 -14
- package/bundle/server/dist/routes/runtime-capacity.test.js +11 -3
- package/bundle/server/dist/runners/muse-serve-session.js +7 -6
- package/bundle/server/package.json +2 -2
- package/bundle/server/static-ui/assets/{index-CYqRh_-S.css → index-BII-LgB8.css} +1 -1
- package/bundle/server/static-ui/assets/index-Bq8wWpZm.js +60 -0
- package/bundle/server/static-ui/index.html +2 -2
- package/bundle/shared/dist/agents.d.ts +15 -15
- package/bundle/shared/dist/agents.js +6 -0
- package/bundle/shared/dist/index.d.ts +7 -7
- package/bundle/shared/package.json +1 -1
- package/dist/doctor.d.ts +22 -3
- package/dist/doctor.js +46 -12
- package/dist/doctor.test.js +41 -10
- package/package.json +1 -1
- package/bundle/server/static-ui/assets/index-UO4lHZw4.js +0 -60
|
@@ -4,6 +4,7 @@ import { CODEX_AUTH_REMEDIATION, MUSE_AUTH_REMEDIATION, cursorAuthIssueFromOutpu
|
|
|
4
4
|
import { claudeBinExists, cursorBinExists, resolveClaudeBin, cursorInvokeArgs, resolveCursorBin, resolveCodexBin, codexBinExists, resolveMuseBin, museBinExists, resolveMuseAuthFile, MUSE_CLI_ENV, } from "../cli-env.js";
|
|
5
5
|
import { checkAgentDeckHealth, checkAgentDeckMcpRegistration, fetchDecks, } from "./agent-deck.js";
|
|
6
6
|
import { runtimeAvailability } from "../repository/runtime-availability.js";
|
|
7
|
+
import { museCapabilityCheckInFlight, museCapabilityIssues, museCapabilitySettleCount, parseMuseVersion, resetMuseCapabilityStateForTests, } from "./muse-capability.js";
|
|
7
8
|
const RUNTIME_LABEL = {
|
|
8
9
|
claude_code: "Claude",
|
|
9
10
|
cursor_local: "Cursor",
|
|
@@ -112,6 +113,7 @@ export function clearAgentHealthCaches() {
|
|
|
112
113
|
githubIssueCache = null;
|
|
113
114
|
cursorSoftFailStreak = 0;
|
|
114
115
|
cursorLastHealthyAt = null;
|
|
116
|
+
resetMuseCapabilityStateForTests();
|
|
115
117
|
}
|
|
116
118
|
function isSoftCursorProbeIssue(issue) {
|
|
117
119
|
return (issue.code === "runtime_auth" &&
|
|
@@ -195,6 +197,11 @@ async function cursorRuntimeIssues() {
|
|
|
195
197
|
* billed `muse exec`, so a *present* but expired login is not detectable here — it surfaces at
|
|
196
198
|
* the first run, which the same classifier reads from stderr. `auth.json` is tested for
|
|
197
199
|
* existence only, never read.
|
|
200
|
+
*
|
|
201
|
+
* NOT-277: with CLI and credentials present, the reported version is handed to the capability
|
|
202
|
+
* check — a version not yet checked runs one real developer-shell probe (muse-capability.ts);
|
|
203
|
+
* `runtime_capability` blocks while it runs, when it finds shell/write missing, or when it could
|
|
204
|
+
* not complete.
|
|
198
205
|
*/
|
|
199
206
|
async function museRuntimeIssues() {
|
|
200
207
|
const bin = resolveMuseBin();
|
|
@@ -224,7 +231,17 @@ async function museRuntimeIssues() {
|
|
|
224
231
|
if (!process.env.META_API_KEY && !fs.existsSync(resolveMuseAuthFile())) {
|
|
225
232
|
return [{ code: "runtime_auth", message: MUSE_AUTH_REMEDIATION }];
|
|
226
233
|
}
|
|
227
|
-
|
|
234
|
+
const version = parseMuseVersion(ver.output);
|
|
235
|
+
if (!version) {
|
|
236
|
+
return [
|
|
237
|
+
{
|
|
238
|
+
code: "runtime_unknown",
|
|
239
|
+
message: "Could not determine Muse Code health — `muse --version` printed no version",
|
|
240
|
+
},
|
|
241
|
+
];
|
|
242
|
+
}
|
|
243
|
+
// A settled check drops the cached result so admission sees it on its next health read.
|
|
244
|
+
return museCapabilityIssues(version, () => runtimeIssueCache.delete("muse_code"));
|
|
228
245
|
}
|
|
229
246
|
/** Exported for direct testing — bypasses the 60s cache in runtimeIssues(). */
|
|
230
247
|
export async function runtimeIssuesUncached(runtime) {
|
|
@@ -427,12 +444,18 @@ async function runtimeIssues(runtime) {
|
|
|
427
444
|
if (cached && Date.now() - cached.at < ttl) {
|
|
428
445
|
return [...capIssues, ...cached.issues];
|
|
429
446
|
}
|
|
447
|
+
const museSettles = museCapabilitySettleCount();
|
|
430
448
|
const issues = await runtimeIssuesUncached(runtime);
|
|
431
449
|
const nonCap = issues.filter((i) => i.code !== "usage_capped");
|
|
450
|
+
// NOT-277: a capability check that settled during this read already superseded what it returned.
|
|
451
|
+
if (runtime === "muse_code" && museCapabilitySettleCount() !== museSettles) {
|
|
452
|
+
return runtimeIssues(runtime);
|
|
453
|
+
}
|
|
432
454
|
// Soft fail (published or grace-held) uses a short TTL so a wake retry can clear quickly;
|
|
433
455
|
// a sticky 60s cache of "Could not confirm" is what parked the queue after sleep (NOT-157).
|
|
434
456
|
const softProbeFailure = nonCap.some(isSoftCursorProbeIssue) ||
|
|
435
|
-
(runtime === "cursor_local" && cursorSoftFailStreak > 0)
|
|
457
|
+
(runtime === "cursor_local" && cursorSoftFailStreak > 0) ||
|
|
458
|
+
(runtime === "muse_code" && museCapabilityCheckInFlight());
|
|
436
459
|
runtimeIssueCache.set(runtime, { at: Date.now(), issues: nonCap, softProbeFailure });
|
|
437
460
|
return [...capIssues, ...nonCap];
|
|
438
461
|
}
|
|
@@ -16,7 +16,10 @@ process.env.AGENT_DEALER_HOME = fs.mkdtempSync(path.join(os.tmpdir(), "dealer-ag
|
|
|
16
16
|
const { migrate, getDb } = await import("../db/index.js");
|
|
17
17
|
const { createAgent, getAgent } = await import("../repository/agents.js");
|
|
18
18
|
const { healthForAgent, runtimeIssuesUncached, githubIssuesUncached, githubIssuesSync, classifyGithubAuthStatus, clearAgentHealthCaches, setCursorProbeTimingForTests, setRunCommandForTests, } = await import("./agent-health.js");
|
|
19
|
+
const { setMuseCapabilityProbeForTests, settleMuseCapabilityCheckForTests } = await import("./muse-capability.js");
|
|
19
20
|
migrate();
|
|
21
|
+
// NOT-277: health never runs the real (billed) Muse capability probe in tests.
|
|
22
|
+
setMuseCapabilityProbeForTests(async () => ({ status: "capable" }));
|
|
20
23
|
const FAILURE = { ok: false, code: "DECK_UNAVAILABLE", message: "Agent Deck API error: 502" };
|
|
21
24
|
/** Tests inject an empty github list so host `gh auth` does not pollute assertions. */
|
|
22
25
|
const NO_GITHUB = [];
|
|
@@ -414,8 +417,14 @@ The keychain item is stuck. Delete it and sign in again:
|
|
|
414
417
|
const withLogin = emptyConfigHome();
|
|
415
418
|
fs.mkdirSync(path.join(withLogin, "muse"));
|
|
416
419
|
fs.writeFileSync(path.join(withLogin, "muse", "auth.json"), "{}");
|
|
417
|
-
|
|
418
|
-
|
|
420
|
+
// NOT-277: the first read of a version starts its one-time capability check; healthy once settled.
|
|
421
|
+
const settledIssues = async () => {
|
|
422
|
+
await runtimeIssuesUncached("muse_code");
|
|
423
|
+
await settleMuseCapabilityCheckForTests();
|
|
424
|
+
return runtimeIssuesUncached("muse_code");
|
|
425
|
+
};
|
|
426
|
+
assert.deepEqual(await withMuseEnv({ MUSE_CLI: stub.bin, configHome: withLogin }, settledIssues), []);
|
|
427
|
+
assert.deepEqual(await withMuseEnv({ MUSE_CLI: stub.bin, META_API_KEY: "k", configHome: emptyConfigHome() }, settledIssues), []);
|
|
419
428
|
});
|
|
420
429
|
test("muse probes run with MUSE_NO_AUTO_UPDATE=1 so the launcher cannot swap the pinned version", async () => {
|
|
421
430
|
const stub = stubMuse("echo 'Muse Code 1.3.0 (1.3.0-R3401.1)'");
|
|
@@ -442,6 +451,50 @@ The keychain item is stuck. Delete it and sign in again:
|
|
|
442
451
|
setCursorProbeTimingForTests(null);
|
|
443
452
|
}
|
|
444
453
|
});
|
|
454
|
+
// NOT-277: through the cached health path the agent list / admission read — a Muse auto-update
|
|
455
|
+
// fires the capability check exactly once for the new version, and the result (not a generic
|
|
456
|
+
// symptom) is what the unhealthy agent shows.
|
|
457
|
+
test("a Muse version change is capability-checked once and its named result surfaces on the agent", async () => {
|
|
458
|
+
const OLD = "1.3.0-R3401.1";
|
|
459
|
+
const NEW = "1.4.0-R4161.1";
|
|
460
|
+
const versionFile = path.join(fs.mkdtempSync(path.join(os.tmpdir(), "dealer-muse-ver-")), "v");
|
|
461
|
+
fs.writeFileSync(versionFile, `Muse Code 1.3.0 (${OLD})\n`);
|
|
462
|
+
const stub = stubMuse(`cat ${JSON.stringify(versionFile)}`);
|
|
463
|
+
const calls = [];
|
|
464
|
+
setMuseCapabilityProbeForTests(async (version) => {
|
|
465
|
+
calls.push(version);
|
|
466
|
+
return version === OLD
|
|
467
|
+
? { status: "capable" }
|
|
468
|
+
: { status: "missing", detail: "probe session completed without running its shell command" };
|
|
469
|
+
});
|
|
470
|
+
const agent = createAgent({ name: "muse-updated", runtime: "muse_code", deckId: randomUUID() });
|
|
471
|
+
const health = () => healthForAgent(agent, true, undefined, true, null, NO_GITHUB);
|
|
472
|
+
try {
|
|
473
|
+
await withMuseEnv({ MUSE_CLI: stub.bin, META_API_KEY: "k", configHome: emptyConfigHome() }, async () => {
|
|
474
|
+
// The stub probe settles within this read; the health read must not cache its stale
|
|
475
|
+
// "verifying" block (the in-flight block itself is covered in muse-capability.test.ts).
|
|
476
|
+
await health();
|
|
477
|
+
await settleMuseCapabilityCheckForTests();
|
|
478
|
+
for (let i = 0; i < 3; i++)
|
|
479
|
+
assert.deepEqual((await health()).issues, []);
|
|
480
|
+
assert.deepEqual(calls, [OLD]);
|
|
481
|
+
// Muse auto-updates. The 60s runtime cache notices on its next miss.
|
|
482
|
+
fs.writeFileSync(versionFile, `Muse Code 1.4.0 (${NEW})\n`);
|
|
483
|
+
await runtimeIssuesUncached("muse_code");
|
|
484
|
+
await settleMuseCapabilityCheckForTests();
|
|
485
|
+
for (let i = 0; i < 3; i++) {
|
|
486
|
+
const after = await health();
|
|
487
|
+
assert.equal(after.healthy, false);
|
|
488
|
+
assert.deepEqual(after.issues.map((x) => x.code), ["runtime_capability"]);
|
|
489
|
+
assert.match(after.issues[0].message, new RegExp(`Muse Code updated ${OLD} → ${NEW}: developer sessions no longer get shell/write access`));
|
|
490
|
+
}
|
|
491
|
+
assert.deepEqual(calls, [OLD, NEW]);
|
|
492
|
+
});
|
|
493
|
+
}
|
|
494
|
+
finally {
|
|
495
|
+
setMuseCapabilityProbeForTests(async () => ({ status: "capable" }));
|
|
496
|
+
}
|
|
497
|
+
});
|
|
445
498
|
test("claude MCP endpoint mismatch surfaces a distinct message, not the generic setup hint", async () => {
|
|
446
499
|
const agent = createAgent({
|
|
447
500
|
name: "claude-mcp-mismatch",
|
|
@@ -70,8 +70,23 @@ export const CHECKS_FAILURE_GENERIC_REASON = "Developer's PR checks failed.";
|
|
|
70
70
|
export const CHECKS_EVIDENCE_MAX_FAILED_CHECKS = 10;
|
|
71
71
|
/** Max distinct Actions runs whose logs are fetched (one `gh run view` per run). */
|
|
72
72
|
export const CHECKS_EVIDENCE_MAX_RUNS = 5;
|
|
73
|
-
/**
|
|
74
|
-
|
|
73
|
+
/**
|
|
74
|
+
* NOT-276: raw per-run ceiling (2MB) on fetched log text actually held in
|
|
75
|
+
* memory — purely a guard against pathologically huge logs, NOT a "search only
|
|
76
|
+
* the tail" window. `fetchChecksFailureEvidence` passes everything under this
|
|
77
|
+
* ceiling untruncated into `buildFailureExcerpt`, whose failure-pattern search
|
|
78
|
+
* plus the per-excerpt line/char caps already bound the output.
|
|
79
|
+
*/
|
|
80
|
+
export const CHECKS_EVIDENCE_MAX_RAW_LOG_CHARS_PER_RUN = 2_000_000;
|
|
81
|
+
/**
|
|
82
|
+
* NOT-276 round 3: explicit `maxBuffer` (bytes) for the `gh run view --log-failed`
|
|
83
|
+
* fetch. Node's `execFile` defaults to ~1 MiB, which would reject any log over
|
|
84
|
+
* that size with ERR_CHILD_PROCESS_STDIO_MAXBUFFER before the 2MB raw-log
|
|
85
|
+
* ceiling above ever applies. This buffer sits above the ceiling (plus headroom
|
|
86
|
+
* for stderr bytes and multi-byte chars) so the ceiling — not the process
|
|
87
|
+
* buffer — is what bounds memory.
|
|
88
|
+
*/
|
|
89
|
+
export const CHECKS_EVIDENCE_GH_LOG_MAX_BUFFER = 4_000_000;
|
|
75
90
|
/** Single global cap on the focused excerpt threaded into the retry prompt. */
|
|
76
91
|
export const CHECKS_EVIDENCE_MAX_EXCERPT_CHARS = 4_000;
|
|
77
92
|
/** Max lines in the focused excerpt; context lines kept around each failure line. */
|
|
@@ -148,11 +163,35 @@ export function sanitizeCiText(text) {
|
|
|
148
163
|
return out;
|
|
149
164
|
}
|
|
150
165
|
const FAILURE_LINE_PATTERN = /error|err!|e404|fail|fatal|exception|traceback|assert|not found|cannot |can't |unable |conflict|reject|denied|panic|timed?\s*out|npm ERR!/i;
|
|
166
|
+
/**
|
|
167
|
+
* Explicit failure markers, not just failure-adjacent vocabulary: TAP's own `not ok`
|
|
168
|
+
* result line (the TAP spec's standard fail marker, used by many test harnesses beyond
|
|
169
|
+
* `node:test` — not a runner-specific parser) and GitHub Actions' own `##[error]`
|
|
170
|
+
* workflow-command annotation (emitted by the platform itself for a genuinely failed
|
|
171
|
+
* step, regardless of what tool ran in it). A coordinator/CI-tooling repo's own test
|
|
172
|
+
* suite is full of passing tests *about* failure handling — "conflict", "timeout",
|
|
173
|
+
* "denied", "escalates" — so `FAILURE_LINE_PATTERN` density alone is not reliable
|
|
174
|
+
* (NOT-276 round 3: verified against the real run-36330128633 log, a cluster of
|
|
175
|
+
* passing tests named around escalation/conflict/timeout out-ranked the actual
|
|
176
|
+
* `not ok 355`/`356` failure under pure density ranking). These two markers get a
|
|
177
|
+
* large ranking bonus below so a window that contains one always wins.
|
|
178
|
+
*/
|
|
179
|
+
// Not anchored to line start: `gh run view --log-failed` prefixes every line with
|
|
180
|
+
// `<job>\t<step>\t<timestamp> ` before the actual tool output, so the TAP/annotation
|
|
181
|
+
// text never starts at column 0.
|
|
182
|
+
const STRONG_FAILURE_LINE_PATTERN = /\bnot ok\b|##\[error\]/i;
|
|
183
|
+
const STRONG_HIT_WEIGHT = 1000;
|
|
151
184
|
/**
|
|
152
185
|
* Focus a (sanitized) log around its useful failure/error lines: keep a small context
|
|
153
|
-
* window around each matching line, merge overlapping windows,
|
|
154
|
-
*
|
|
155
|
-
*
|
|
186
|
+
* window around each matching line, merge overlapping windows, rank merged regions by
|
|
187
|
+
* weighted hit density (explicit failure markers far outweigh failure-adjacent
|
|
188
|
+
* vocabulary; ties break on density, then log order) so an early real failure cluster
|
|
189
|
+
* is not crowded out of the line budget by sparse isolated mentions in passing-test
|
|
190
|
+
* names, or by a *dense* but merely topical cluster of passing tests about failure
|
|
191
|
+
* handling itself (NOT-276 round 2 found the former, round 3 the latter — see
|
|
192
|
+
* `STRONG_FAILURE_LINE_PATTERN`'s comment), then collapse long runs of identical lines
|
|
193
|
+
* (CI setup spam) and enforce the global line/char caps. With no matching line, the
|
|
194
|
+
* tail is the most likely failure site. Never returns unsanitized text.
|
|
156
195
|
*/
|
|
157
196
|
export function buildFailureExcerpt(combinedLog) {
|
|
158
197
|
const sanitized = sanitizeCiText(combinedLog);
|
|
@@ -160,9 +199,14 @@ export function buildFailureExcerpt(combinedLog) {
|
|
|
160
199
|
if (lines.every((l) => !l.trim()))
|
|
161
200
|
return { excerpt: "", truncated: false };
|
|
162
201
|
const hits = [];
|
|
202
|
+
const strongHits = new Set();
|
|
163
203
|
lines.forEach((line, i) => {
|
|
164
|
-
|
|
204
|
+
const strong = STRONG_FAILURE_LINE_PATTERN.test(line);
|
|
205
|
+
if (strong || FAILURE_LINE_PATTERN.test(line)) {
|
|
165
206
|
hits.push(i);
|
|
207
|
+
if (strong)
|
|
208
|
+
strongHits.add(i);
|
|
209
|
+
}
|
|
166
210
|
});
|
|
167
211
|
let selected;
|
|
168
212
|
let truncated = false;
|
|
@@ -180,8 +224,54 @@ export function buildFailureExcerpt(combinedLog) {
|
|
|
180
224
|
else
|
|
181
225
|
merged.push([w[0], w[1]]);
|
|
182
226
|
}
|
|
227
|
+
// Rank merged regions by weighted hit density so the excerpt budget goes to the
|
|
228
|
+
// most failure-indicative clusters first: an explicit failure marker (`not ok`,
|
|
229
|
+
// `##[error]`) counts for STRONG_HIT_WEIGHT, everything else for 1 — a window
|
|
230
|
+
// with one real marker always outranks a window with many topical-only hits.
|
|
231
|
+
// Ties keep log order, and output is re-sorted to log order for readability.
|
|
232
|
+
// `hits` is ascending (built in line order) and `merged` is ascending by `from`,
|
|
233
|
+
// so a single linear pass sums weight per region.
|
|
234
|
+
let hi = 0;
|
|
235
|
+
const ranked = merged.map(([from, to]) => {
|
|
236
|
+
while (hi < hits.length && hits[hi] < from)
|
|
237
|
+
hi++;
|
|
238
|
+
let weight = 0;
|
|
239
|
+
let k = hi;
|
|
240
|
+
while (k < hits.length && hits[k] <= to) {
|
|
241
|
+
weight += strongHits.has(hits[k]) ? STRONG_HIT_WEIGHT : 1;
|
|
242
|
+
k++;
|
|
243
|
+
}
|
|
244
|
+
return { from, to, weight };
|
|
245
|
+
});
|
|
246
|
+
ranked.sort((a, b) => b.weight - a.weight || a.from - b.from);
|
|
247
|
+
const taken = [];
|
|
248
|
+
let used = 0;
|
|
249
|
+
let usedChars = 0;
|
|
250
|
+
for (const w of ranked) {
|
|
251
|
+
const size = w.to - w.from + 1 + (taken.length > 0 ? 1 : 0);
|
|
252
|
+
let chars = taken.length > 0 ? 3 : 0; // "...\n" separator
|
|
253
|
+
for (let i = w.from; i <= w.to; i++)
|
|
254
|
+
chars += lines[i].length + 1;
|
|
255
|
+
// Skip regions that no longer fit — by line count or by character count, since
|
|
256
|
+
// the final excerpt is char-capped too (NOT-276 round 3: a lower-ranked but
|
|
257
|
+
// verbose region taken first would otherwise still crowd a higher-ranked
|
|
258
|
+
// region's content out of the *character* budget after chronological
|
|
259
|
+
// reassembly below, even though ranking correctly gave the real failure
|
|
260
|
+
// priority for line-budget inclusion). A smaller later region can still fill
|
|
261
|
+
// whatever budget remains — but always take the top-ranked region even if it
|
|
262
|
+
// alone exceeds either budget (the caps below truncate it, as before).
|
|
263
|
+
if (taken.length > 0 &&
|
|
264
|
+
(used + size > CHECKS_EVIDENCE_MAX_EXCERPT_LINES || usedChars + chars > CHECKS_EVIDENCE_MAX_EXCERPT_CHARS)) {
|
|
265
|
+
truncated = true;
|
|
266
|
+
continue;
|
|
267
|
+
}
|
|
268
|
+
taken.push([w.from, w.to]);
|
|
269
|
+
used += size;
|
|
270
|
+
usedChars += chars;
|
|
271
|
+
}
|
|
272
|
+
taken.sort((a, b) => a[0] - b[0]);
|
|
183
273
|
const picked = [];
|
|
184
|
-
|
|
274
|
+
taken.forEach(([from, to], idx) => {
|
|
185
275
|
if (idx > 0)
|
|
186
276
|
picked.push("...");
|
|
187
277
|
for (let i = from; i <= to; i++)
|
|
@@ -241,7 +331,7 @@ export function formatChecksFailureDetails(opts) {
|
|
|
241
331
|
return parts.join("\n");
|
|
242
332
|
}
|
|
243
333
|
const NO_COMMITS_PATTERN = /no commits between/i;
|
|
244
|
-
const defaultExec = (args, opts) => run("gh", args, opts);
|
|
334
|
+
const defaultExec = (args, opts) => run("gh", args, { cwd: opts.cwd, maxBuffer: opts.maxBuffer });
|
|
245
335
|
async function ghPrView(exec, cwd, fields, selector) {
|
|
246
336
|
const args = ["pr", "view", ...(selector != null ? [selector] : []), "--json", fields];
|
|
247
337
|
try {
|
|
@@ -354,11 +444,19 @@ export async function fetchChecksFailureEvidence(exec, opts) {
|
|
|
354
444
|
let logsUnavailable = false;
|
|
355
445
|
for (const runId of runIds) {
|
|
356
446
|
try {
|
|
357
|
-
const { stdout } = await exec(["run", "view", runId, "--log-failed"], {
|
|
358
|
-
|
|
359
|
-
|
|
447
|
+
const { stdout } = await exec(["run", "view", runId, "--log-failed"], {
|
|
448
|
+
cwd: opts.cwd,
|
|
449
|
+
maxBuffer: CHECKS_EVIDENCE_GH_LOG_MAX_BUFFER,
|
|
450
|
+
});
|
|
451
|
+
// NOT-276: search before truncating — the full fetched log feeds
|
|
452
|
+
// `buildFailureExcerpt`'s failure-pattern search, so an early failure is
|
|
453
|
+
// never discarded by a small tail window. Only logs beyond the raw memory
|
|
454
|
+
// ceiling are cut at all (tail kept), and the excerpt caps still bound
|
|
455
|
+
// the final output.
|
|
456
|
+
const full = stdout.length > CHECKS_EVIDENCE_MAX_RAW_LOG_CHARS_PER_RUN
|
|
457
|
+
? stdout.slice(-CHECKS_EVIDENCE_MAX_RAW_LOG_CHARS_PER_RUN)
|
|
360
458
|
: stdout;
|
|
361
|
-
logsByRun.set(runId,
|
|
459
|
+
logsByRun.set(runId, full);
|
|
362
460
|
}
|
|
363
461
|
catch {
|
|
364
462
|
logsUnavailable = true;
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
// packages/server/src/adapters/github.test.ts
|
|
2
2
|
import { test } from "node:test";
|
|
3
3
|
import assert from "node:assert/strict";
|
|
4
|
-
import { parsePrView, summarizeChecks, pollPrChecks, createGithubAdapter, fetchChecksFailureEvidence, sanitizeCiText, sanitizeUrl, extractActionsRunId, buildFailureExcerpt, formatChecksFailureDetails, CHECKS_EVIDENCE_MAX_EXCERPT_CHARS, CHECKS_FAILURE_GENERIC_REASON, PR_VIEW_FIELDS, } from "./github.js";
|
|
4
|
+
import { parsePrView, summarizeChecks, pollPrChecks, createGithubAdapter, fetchChecksFailureEvidence, sanitizeCiText, sanitizeUrl, extractActionsRunId, buildFailureExcerpt, formatChecksFailureDetails, CHECKS_EVIDENCE_MAX_EXCERPT_CHARS, CHECKS_EVIDENCE_MAX_EXCERPT_LINES, CHECKS_EVIDENCE_MAX_RAW_LOG_CHARS_PER_RUN, CHECKS_EVIDENCE_GH_LOG_MAX_BUFFER, CHECKS_FAILURE_GENERIC_REASON, PR_VIEW_FIELDS, } from "./github.js";
|
|
5
5
|
test("parsePrView extracts the ground-truth handoff fields, including draft status", () => {
|
|
6
6
|
const view = parsePrView(JSON.stringify({
|
|
7
7
|
number: 7,
|
|
@@ -34,15 +34,17 @@ test("summarizeChecks accepts neutral/skipped as success but fails closed on an
|
|
|
34
34
|
/** Records every `gh` invocation and returns responses off a queue — never calls real `gh`. */
|
|
35
35
|
function queuedExec(responses) {
|
|
36
36
|
const calls = [];
|
|
37
|
+
const execOpts = [];
|
|
37
38
|
const queue = [...responses];
|
|
38
|
-
const exec = async (args) => {
|
|
39
|
+
const exec = async (args, opts) => {
|
|
39
40
|
calls.push(args);
|
|
41
|
+
execOpts.push(opts);
|
|
40
42
|
const next = queue.shift() ?? { stdout: "" };
|
|
41
43
|
if (next.error != null)
|
|
42
44
|
throw Object.assign(new Error(next.error), { stderr: next.error });
|
|
43
45
|
return { stdout: next.stdout ?? "" };
|
|
44
46
|
};
|
|
45
|
-
return { exec, calls };
|
|
47
|
+
return { exec, calls, execOpts };
|
|
46
48
|
}
|
|
47
49
|
test("viewPr looks up the PR explicitly by branch, never a bare `gh pr view`", async () => {
|
|
48
50
|
const { exec, calls } = queuedExec([
|
|
@@ -426,6 +428,275 @@ test("NOT-252: extractActionsRunId and sanitizeUrl helpers", async () => {
|
|
|
426
428
|
assert.equal(sanitizeUrl("https://example.com/a?b=1#c"), "https://example.com/a");
|
|
427
429
|
assert.equal(sanitizeUrl("https://example.com/a"), "https://example.com/a");
|
|
428
430
|
});
|
|
431
|
+
// --- NOT-276: search-before-truncate — an early failure must survive excerpt building ---
|
|
432
|
+
// Replays the NOT-273 incident shape (Actions run 36330128633 "Unit tests" step):
|
|
433
|
+
// 1667 TAP lines, `not ok 355/356` + `cancelledByParent` at ~21% through the log,
|
|
434
|
+
// followed by ~1300 passing-test lines. Under the old last-20K-chars pre-truncation
|
|
435
|
+
// the failure sat outside the kept tail and the excerpt showed only passing
|
|
436
|
+
// tail noise; with search-before-truncate it must surface the failure instead.
|
|
437
|
+
// Faithful-scale substitute for the real `gh run view 36330128633 --log-failed`
|
|
438
|
+
// replay (no network/gh in this sandbox — the PR description must still record a
|
|
439
|
+
// replay against the saved real "Unit tests" log showing `not ok 355`/`356`).
|
|
440
|
+
// Crucially, ~21 passing lines BEFORE the failure carry realistic
|
|
441
|
+
// failure-pattern words in their test names (real TAP `ok` lines do this — e.g.
|
|
442
|
+
// "handles error ..."), spaced >13 lines apart so each is an isolated hit region:
|
|
443
|
+
// without hit-density ranking, those early weak hits fill the 80-line budget
|
|
444
|
+
// ahead of the real failure cluster (round-2 blocking finding). The dense
|
|
445
|
+
// failure block (failureType x2 + error: x1 within 9 lines) must outrank them.
|
|
446
|
+
test("NOT-276: failure line before the last 20K chars still reaches the excerpt", async () => {
|
|
447
|
+
const early = [];
|
|
448
|
+
for (let n = 1; n <= 354; n++) {
|
|
449
|
+
// Isolated pattern hits every 16 lines (> 2x the 6-line context radius, so
|
|
450
|
+
// windows never merge): ~22 weak hits precede the real failure.
|
|
451
|
+
if (n % 16 === 0)
|
|
452
|
+
early.push(`ok ${n} - handles error output for test ${n}`);
|
|
453
|
+
else if (n === 353)
|
|
454
|
+
early.push("ok 353 - reports error when child fails to spawn");
|
|
455
|
+
else if (n === 354)
|
|
456
|
+
early.push("ok 354 - cleans up after failure");
|
|
457
|
+
else
|
|
458
|
+
early.push(`ok ${n} - passing test number ${n}`);
|
|
459
|
+
}
|
|
460
|
+
const failureBlock = [
|
|
461
|
+
"not ok 355 - coordinator spawns child with explicit cwd",
|
|
462
|
+
" ---",
|
|
463
|
+
" failureType: 'cancelledByParent'",
|
|
464
|
+
" error: test cancelled by parent",
|
|
465
|
+
" ---",
|
|
466
|
+
"not ok 356 - coordinator spawns child with explicit cwd (2)",
|
|
467
|
+
" ---",
|
|
468
|
+
" failureType: 'cancelledByParent'",
|
|
469
|
+
" ---",
|
|
470
|
+
].join("\n");
|
|
471
|
+
const filler = Array.from({ length: 1667 - 356 }, (_, i) => {
|
|
472
|
+
const n = 357 + i;
|
|
473
|
+
if (i % 200 === 0)
|
|
474
|
+
return `ok ${n} - handles error output for test ${n}`;
|
|
475
|
+
if (i % 150 === 0)
|
|
476
|
+
return `ok ${n} - cleans up after failure ${n}`;
|
|
477
|
+
return `ok ${n} - passing test number ${n}`;
|
|
478
|
+
}).join("\n");
|
|
479
|
+
const log = `${early.join("\n")}\n${failureBlock}\n${filler}`;
|
|
480
|
+
// Guard the test's premise: the failure really does sit outside the old 20K tail window.
|
|
481
|
+
assert.ok(log.indexOf("not ok 355") < log.length - 20_000);
|
|
482
|
+
const { exec } = queuedExec([
|
|
483
|
+
{ stdout: prViewWithRollup([actionsCheck("verify", "FAILURE", "36330128633")]) },
|
|
484
|
+
{ stdout: log },
|
|
485
|
+
]);
|
|
486
|
+
const evidence = await fetchChecksFailureEvidence(exec, { cwd: "/repo", number: 160, expectedHeadSha: HEAD_SHA });
|
|
487
|
+
assert.ok(evidence);
|
|
488
|
+
assert.match(evidence.excerpt, /not ok 355/);
|
|
489
|
+
assert.match(evidence.excerpt, /not ok 356/);
|
|
490
|
+
assert.match(evidence.excerpt, /cancelledByParent/);
|
|
491
|
+
assert.doesNotMatch(evidence.excerpt, /passing test number 1667/);
|
|
492
|
+
assert.ok(evidence.excerpt.length <= CHECKS_EVIDENCE_MAX_EXCERPT_CHARS);
|
|
493
|
+
assert.ok(evidence.excerpt.split("\n").length <= CHECKS_EVIDENCE_MAX_EXCERPT_LINES);
|
|
494
|
+
});
|
|
495
|
+
// NOT-276 round-2 replay stand-in (blocking finding: the real `gh run view
|
|
496
|
+
// 36330128633 --log-failed` replay needs network/gh, unavailable in CI/sandbox,
|
|
497
|
+
// so this mirrors its exact byte shape at faithful scale instead): every line
|
|
498
|
+
// carries the real `<job>\t<step>\t<timestamp> ` prefix, 1667 TAP results with
|
|
499
|
+
// `not ok 355`/`356` + `cancelledByParent` at ~21% through, ~24 passing lines
|
|
500
|
+
// before the failure carrying failure-pattern words in their names, >20K
|
|
501
|
+
// chars of passing-test tail after it, and the node:test TAP summary block at
|
|
502
|
+
// the true end of the log (the root-cause mechanism: node:test runs to
|
|
503
|
+
// completion, so `# fail 2` — a weak tail hit — sits after thousands of passing
|
|
504
|
+
// lines). The excerpt must surface the early strong failure, not the tail.
|
|
505
|
+
test("NOT-276 round-2 replay: full-scale Actions-prefixed log with an early `not ok` failure", async () => {
|
|
506
|
+
const bodies = [];
|
|
507
|
+
for (let n = 1; n <= 354; n++) {
|
|
508
|
+
if (n % 16 === 0)
|
|
509
|
+
bodies.push(`ok ${n} - handles error output for test ${n}`);
|
|
510
|
+
else if (n === 353)
|
|
511
|
+
bodies.push("ok 353 - reports error when child fails to spawn");
|
|
512
|
+
else if (n === 354)
|
|
513
|
+
bodies.push("ok 354 - cleans up after failure");
|
|
514
|
+
else
|
|
515
|
+
bodies.push(`ok ${n} - passing test number ${n}`);
|
|
516
|
+
}
|
|
517
|
+
bodies.push("not ok 355 - coordinator spawns child with explicit cwd", " ---", " failureType: 'cancelledByParent'", " error: test cancelled by parent", " ---", "not ok 356 - coordinator spawns child with explicit cwd (2)", " ---", " failureType: 'cancelledByParent'", " ---");
|
|
518
|
+
for (let n = 357; n <= 1667; n++) {
|
|
519
|
+
const i = n - 357;
|
|
520
|
+
if (i % 200 === 0)
|
|
521
|
+
bodies.push(`ok ${n} - handles error output for test ${n}`);
|
|
522
|
+
else if (i % 150 === 0)
|
|
523
|
+
bodies.push(`ok ${n} - cleans up after failure ${n}`);
|
|
524
|
+
else
|
|
525
|
+
bodies.push(`ok ${n} - passing test number ${n}`);
|
|
526
|
+
}
|
|
527
|
+
// node:test's own end-of-run TAP summary, as in the real incident log — its
|
|
528
|
+
// `# fail 2` line is a weak failure-pattern hit in the tail that must not
|
|
529
|
+
// outrank the early strong `not ok` failure (round-2 blocking finding).
|
|
530
|
+
bodies.push("# tests 1667", "# suites 4", "# pass 1663", "# fail 2", "# cancelled 2", "# skipped 0", "# todo 0", "# duration_ms 184213.015");
|
|
531
|
+
const log = bodies
|
|
532
|
+
.map((b, i) => `verify\tUnit tests\t2026-09-27T16:${String(Math.floor(i / 60)).padStart(2, "0")}:${String(i % 60).padStart(2, "0")}Z ${b}`)
|
|
533
|
+
.join("\n");
|
|
534
|
+
// Guard the test's premise: the failure really does sit outside the old 20K tail window.
|
|
535
|
+
assert.ok(log.indexOf("not ok 355") < log.length - 20_000);
|
|
536
|
+
const { exec } = queuedExec([
|
|
537
|
+
{ stdout: prViewWithRollup([actionsCheck("verify", "FAILURE", "36330128633")]) },
|
|
538
|
+
{ stdout: log },
|
|
539
|
+
]);
|
|
540
|
+
const evidence = await fetchChecksFailureEvidence(exec, { cwd: "/repo", number: 160, expectedHeadSha: HEAD_SHA });
|
|
541
|
+
assert.ok(evidence);
|
|
542
|
+
assert.match(evidence.excerpt, /not ok 355/);
|
|
543
|
+
assert.match(evidence.excerpt, /not ok 356/);
|
|
544
|
+
assert.match(evidence.excerpt, /cancelledByParent/);
|
|
545
|
+
assert.doesNotMatch(evidence.excerpt, /passing test number 1667/);
|
|
546
|
+
assert.ok(evidence.excerpt.length <= CHECKS_EVIDENCE_MAX_EXCERPT_CHARS);
|
|
547
|
+
assert.ok(evidence.excerpt.split("\n").length <= CHECKS_EVIDENCE_MAX_EXCERPT_LINES);
|
|
548
|
+
});
|
|
549
|
+
test("NOT-276: dense failure cluster outranks sparse earlier weak hits (unit level)", async () => {
|
|
550
|
+
// Minimal direct proof of the round-2 ranking: 30 isolated weak hits precede one
|
|
551
|
+
// dense 3-hit failure cluster; the excerpt must contain the cluster.
|
|
552
|
+
const lines = [];
|
|
553
|
+
for (let g = 0; g < 30; g++) {
|
|
554
|
+
for (let k = 0; k < 15; k++)
|
|
555
|
+
lines.push(`ok ${g * 16 + k + 1} - passing test number ${g * 16 + k + 1}`);
|
|
556
|
+
lines.push(`ok ${g * 16 + 16} - handles error output for test ${g * 16 + 16}`);
|
|
557
|
+
}
|
|
558
|
+
lines.push("not ok 999 - real failure here", " failureType: 'cancelledByParent'", " error: test cancelled by parent");
|
|
559
|
+
for (let n = 1000; n < 1100; n++)
|
|
560
|
+
lines.push(`ok ${n} - passing test number ${n}`);
|
|
561
|
+
const { excerpt } = buildFailureExcerpt(lines.join("\n"));
|
|
562
|
+
assert.match(excerpt, /not ok 999/);
|
|
563
|
+
assert.match(excerpt, /cancelledByParent/);
|
|
564
|
+
});
|
|
565
|
+
// NOT-276 round 3 (real PR #163 review, verified against the actual saved
|
|
566
|
+
// `gh run view 36330128633 --log-failed` output): plain hit-density ranking still
|
|
567
|
+
// missed the real failure two different ways. (1) `gh run view --log-failed`
|
|
568
|
+
// prefixes every line with `<job>\t<step>\t<timestamp> ` before the tool's own
|
|
569
|
+
// output, so an anchored `^\s*not ok\b` pattern never matches the real thing — only
|
|
570
|
+
// an unanchored one does. (2) This repo's own coordinator tests are *about*
|
|
571
|
+
// escalation/conflict/timeout handling, so a cluster of unrelated passing tests can
|
|
572
|
+
// out-rank the real (sparse) failure under density alone; an explicit marker
|
|
573
|
+
// (`not ok`, `##[error]`) must dominate ranking regardless of density.
|
|
574
|
+
test("NOT-276 round 3: a GitHub Actions log-line prefix must not hide the `not ok` marker", async () => {
|
|
575
|
+
const prefix = (ts) => `verify\tUnit tests\t${ts}Z `;
|
|
576
|
+
const early = Array.from({ length: 60 }, (_, i) => `${prefix(`2026-09-27T16:07:${String(i).padStart(2, "0")}.0000000`)}ok ${i + 1} - passing test number ${i + 1}`).join("\n");
|
|
577
|
+
const failureBlock = [
|
|
578
|
+
`${prefix("2026-09-27T16:07:31.3990644")}not ok 355 - NOT-273: the serve child spawns with an explicit cwd`,
|
|
579
|
+
`${prefix("2026-09-27T16:07:31.3991406")} failureType: 'cancelledByParent'`,
|
|
580
|
+
`${prefix("2026-09-27T16:07:31.3992743")}not ok 356 - no credential short-circuits to missing without spawning`,
|
|
581
|
+
`${prefix("2026-09-27T16:07:31.3993384")} failureType: 'cancelledByParent'`,
|
|
582
|
+
].join("\n");
|
|
583
|
+
const late = Array.from({ length: 60 }, (_, i) => `${prefix(`2026-09-27T16:08:${String(i).padStart(2, "0")}.0000000`)}ok ${357 + i} - passing test number ${357 + i}`).join("\n");
|
|
584
|
+
const log = `${early}\n${failureBlock}\n${late}\n${prefix("2026-09-27T16:09:02.6611414")}##[error]Process completed with exit code 1.`;
|
|
585
|
+
const { excerpt } = buildFailureExcerpt(log);
|
|
586
|
+
assert.match(excerpt, /not ok 355/);
|
|
587
|
+
assert.match(excerpt, /not ok 356/);
|
|
588
|
+
assert.match(excerpt, /cancelledByParent/);
|
|
589
|
+
});
|
|
590
|
+
// NOT-276 round 3: a lower-ranked region taken first can still be chronologically
|
|
591
|
+
// *earlier* than a higher-ranked one; if the excerpt were char-capped by slicing
|
|
592
|
+
// the reassembled string's tail (ignoring rank), the earlier low-priority region
|
|
593
|
+
// could crowd the later high-priority region's content out of the character
|
|
594
|
+
// budget even though line-budget ranking correctly preferred the real failure.
|
|
595
|
+
test("NOT-276 round 3: an earlier low-priority region must not crowd the real failure out of the character budget", async () => {
|
|
596
|
+
// ~12 weak hits (one per line, "failed"/"error"), long lines: ~12 * 460 ≈ 5500 chars.
|
|
597
|
+
const earlyWeakCluster = Array.from({ length: 12 }, (_, i) => `ok ${i + 1} - reports a handled failure/error path for scenario ${i + 1} ` + "x".repeat(400)).join("\n");
|
|
598
|
+
// The real failure: 2 strong `not ok` hits, short lines (~450 chars total).
|
|
599
|
+
const realFailure = [
|
|
600
|
+
"not ok 999 - the real failure",
|
|
601
|
+
" failureType: 'cancelledByParent'",
|
|
602
|
+
"not ok 1000 - a second real failure",
|
|
603
|
+
" failureType: 'cancelledByParent'",
|
|
604
|
+
].join("\n");
|
|
605
|
+
// A gap of ordinary passing lines keeps the two clusters as separate merged
|
|
606
|
+
// windows (matching the real incident: the weak cluster and the real failure
|
|
607
|
+
// sat ~2300 lines apart) rather than merging into one oversized window, which
|
|
608
|
+
// would exercise a different, narrower edge case than the one under test here.
|
|
609
|
+
const gap = Array.from({ length: 20 }, (_, i) => `ok ${900 + i} - passing test number ${900 + i}`).join("\n");
|
|
610
|
+
const log = `${earlyWeakCluster}\n${gap}\n${realFailure}`;
|
|
611
|
+
assert.ok(earlyWeakCluster.length + realFailure.length > CHECKS_EVIDENCE_MAX_EXCERPT_CHARS, "test premise: both regions together exceed the char budget");
|
|
612
|
+
const { excerpt, truncated } = buildFailureExcerpt(log);
|
|
613
|
+
assert.match(excerpt, /not ok 999/);
|
|
614
|
+
assert.match(excerpt, /not ok 1000/);
|
|
615
|
+
assert.match(excerpt, /cancelledByParent/);
|
|
616
|
+
assert.ok(excerpt.length <= CHECKS_EVIDENCE_MAX_EXCERPT_CHARS);
|
|
617
|
+
assert.equal(truncated, true, "dropping the lower-ranked region must be reported as truncation");
|
|
618
|
+
});
|
|
619
|
+
test("NOT-276: an exact line-budget fill still reports a later failure region as truncated", async () => {
|
|
620
|
+
// The A and B windows contain 39 and 40 lines respectively. With the separator
|
|
621
|
+
// between them, taking both fills the 80-line budget exactly. Region C is a real
|
|
622
|
+
// failure that must be omitted, and that omission must set truncated=true.
|
|
623
|
+
const ordinary = (label, count) => Array.from({ length: count }, (_, i) => `ok ${label}-${i + 1} - passing test`);
|
|
624
|
+
const regionA = Array.from({ length: 27 }, (_, i) => `not ok A-${i + 1} - strong failure A`);
|
|
625
|
+
const regionB = Array.from({ length: 28 }, (_, i) => `not ok B-${i + 1} - strong failure B`);
|
|
626
|
+
const lines = [
|
|
627
|
+
...ordinary("prefix", 6),
|
|
628
|
+
...regionA,
|
|
629
|
+
...ordinary("gap-a-b", 13),
|
|
630
|
+
...regionB,
|
|
631
|
+
...ordinary("gap-b-c", 13),
|
|
632
|
+
"not ok C-1 - later real failure",
|
|
633
|
+
...ordinary("suffix", 6),
|
|
634
|
+
];
|
|
635
|
+
const { excerpt, truncated } = buildFailureExcerpt(lines.join("\n"));
|
|
636
|
+
assert.equal(excerpt.split("\n").length, CHECKS_EVIDENCE_MAX_EXCERPT_LINES);
|
|
637
|
+
assert.match(excerpt, /not ok A-1/);
|
|
638
|
+
assert.match(excerpt, /not ok B-1/);
|
|
639
|
+
assert.doesNotMatch(excerpt, /not ok C-1/);
|
|
640
|
+
assert.equal(truncated, true, "the omitted third failure region must be reported as truncation");
|
|
641
|
+
});
|
|
642
|
+
test("NOT-276: log with no failure-pattern hits still falls back to the tail, unchanged", async () => {
|
|
643
|
+
// NB: this filler must stay free of FAILURE_LINE_PATTERN words — that absence is the no-hit premise.
|
|
644
|
+
const log = Array.from({ length: 200 }, (_, i) => `ok ${i + 1} - passing test number ${i + 1}`).join("\n");
|
|
645
|
+
const { exec } = queuedExec([
|
|
646
|
+
{ stdout: prViewWithRollup([actionsCheck("verify", "FAILURE", "111")]) },
|
|
647
|
+
{ stdout: log },
|
|
648
|
+
]);
|
|
649
|
+
const evidence = await fetchChecksFailureEvidence(exec, { cwd: "/repo", number: 42, expectedHeadSha: HEAD_SHA });
|
|
650
|
+
assert.ok(evidence);
|
|
651
|
+
// The fetch path must feed the full log through exactly as buildFailureExcerpt sees it.
|
|
652
|
+
assert.equal(evidence.excerpt, buildFailureExcerpt(log).excerpt);
|
|
653
|
+
const excerptLines = evidence.excerpt.split("\n");
|
|
654
|
+
assert.equal(excerptLines.length, CHECKS_EVIDENCE_MAX_EXCERPT_LINES);
|
|
655
|
+
assert.match(excerptLines[0] ?? "", /passing test number 121/);
|
|
656
|
+
assert.match(excerptLines[excerptLines.length - 1] ?? "", /passing test number 200/);
|
|
657
|
+
assert.equal(evidence.excerptTruncated, true);
|
|
658
|
+
assert.ok(evidence.excerpt.length <= CHECKS_EVIDENCE_MAX_EXCERPT_CHARS);
|
|
659
|
+
});
|
|
660
|
+
test("NOT-276: log larger than the raw memory ceiling stays bounded without crashing", async () => {
|
|
661
|
+
// Short lines on purpose: the 80-line tail must fit under the 4,000-char
|
|
662
|
+
// excerpt cap, otherwise the char cap (not the memory ceiling) would cut the
|
|
663
|
+
// asserted last line and the test would prove nothing about the tail.
|
|
664
|
+
const line = (i) => `ok ${i} - pad pad ${i}`;
|
|
665
|
+
const targetLen = CHECKS_EVIDENCE_MAX_RAW_LOG_CHARS_PER_RUN + 500_000;
|
|
666
|
+
const count = Math.ceil(targetLen / 20);
|
|
667
|
+
const parts = new Array(count);
|
|
668
|
+
for (let i = 0; i < count; i++)
|
|
669
|
+
parts[i] = line(i);
|
|
670
|
+
const log = parts.join("\n");
|
|
671
|
+
assert.ok(log.length > CHECKS_EVIDENCE_MAX_RAW_LOG_CHARS_PER_RUN);
|
|
672
|
+
const { exec } = queuedExec([
|
|
673
|
+
{ stdout: prViewWithRollup([actionsCheck("verify", "FAILURE", "111")]) },
|
|
674
|
+
{ stdout: log },
|
|
675
|
+
]);
|
|
676
|
+
const evidence = await fetchChecksFailureEvidence(exec, { cwd: "/repo", number: 42, expectedHeadSha: HEAD_SHA });
|
|
677
|
+
assert.ok(evidence);
|
|
678
|
+
assert.ok(evidence.excerpt.length <= CHECKS_EVIDENCE_MAX_EXCERPT_CHARS);
|
|
679
|
+
// The excerpt is built from the kept tail portion, never the discarded head.
|
|
680
|
+
assert.equal(evidence.excerpt, buildFailureExcerpt(log.slice(-CHECKS_EVIDENCE_MAX_RAW_LOG_CHARS_PER_RUN)).excerpt);
|
|
681
|
+
assert.match(evidence.excerpt, new RegExp(`pad pad ${count - 1}`));
|
|
682
|
+
});
|
|
683
|
+
test("NOT-276 round 3: run-log fetch carries an explicit maxBuffer above the raw ceiling", async () => {
|
|
684
|
+
// Node's execFile defaults to ~1 MiB maxBuffer, which would reject any larger
|
|
685
|
+
// --log-failed output with ERR_CHILD_PROCESS_STDIO_MAXBUFFER before the 2MB
|
|
686
|
+
// raw-log ceiling applies. The fetch must pass an explicit buffer above the
|
|
687
|
+
// ceiling so the ceiling — not the process buffer — bounds memory.
|
|
688
|
+
assert.ok(CHECKS_EVIDENCE_GH_LOG_MAX_BUFFER > CHECKS_EVIDENCE_MAX_RAW_LOG_CHARS_PER_RUN, "maxBuffer must sit above the raw-log ceiling");
|
|
689
|
+
const { exec, calls, execOpts } = queuedExec([
|
|
690
|
+
{ stdout: prViewWithRollup([actionsCheck("verify", "FAILURE", "111")]) },
|
|
691
|
+
{ stdout: "verify log\nError: boom\n" },
|
|
692
|
+
]);
|
|
693
|
+
const evidence = await fetchChecksFailureEvidence(exec, { cwd: "/repo", number: 42, expectedHeadSha: HEAD_SHA });
|
|
694
|
+
assert.ok(evidence);
|
|
695
|
+
assert.deepEqual(calls[1], ["run", "view", "111", "--log-failed"]);
|
|
696
|
+
assert.equal(execOpts[1]?.maxBuffer, CHECKS_EVIDENCE_GH_LOG_MAX_BUFFER);
|
|
697
|
+
// The small pr-view lookup needs no oversized buffer.
|
|
698
|
+
assert.ok(execOpts[0]?.maxBuffer == null, "pr view must not carry the log buffer");
|
|
699
|
+
});
|
|
429
700
|
test("NOT-252: formatChecksFailureDetails labels the excerpt as untrusted, not instructions", async () => {
|
|
430
701
|
const details = formatChecksFailureDetails({
|
|
431
702
|
headSha: HEAD_SHA,
|