tickmarkr 2.6.0 → 2.6.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -3
- package/dist/adapters/catalog-remote.js +89 -47
- package/dist/adapters/claude-code.js +9 -6
- package/dist/adapters/codex.js +7 -4
- package/dist/adapters/prompt.d.ts +3 -1
- package/dist/adapters/prompt.js +21 -3
- package/dist/adapters/registry.js +3 -3
- package/dist/adapters/types.d.ts +13 -4
- package/dist/adapters/types.js +16 -0
- package/dist/cli/commands/approve.d.ts +5 -1
- package/dist/cli/commands/approve.js +66 -23
- package/dist/cli/commands/compile.js +13 -3
- package/dist/cli/commands/doctor.d.ts +2 -0
- package/dist/cli/commands/doctor.js +11 -3
- package/dist/cli/commands/fleet.js +71 -8
- package/dist/cli/commands/plan.js +7 -3
- package/dist/cli/commands/report.js +18 -2
- package/dist/cli/commands/resume.js +4 -2
- package/dist/cli/commands/status.js +37 -22
- package/dist/cli/help.d.ts +4 -0
- package/dist/cli/help.js +11 -2
- package/dist/config/config.d.ts +20 -0
- package/dist/config/config.js +47 -8
- package/dist/config/fleet-overlay.d.ts +1 -0
- package/dist/config/fleet-overlay.js +56 -0
- package/dist/drivers/orca.d.ts +26 -1
- package/dist/drivers/orca.js +199 -61
- package/dist/eval/canary.d.ts +2 -1
- package/dist/eval/canary.js +2 -2
- package/dist/eval/dispatch.js +1 -0
- package/dist/gates/acceptance.d.ts +2 -1
- package/dist/gates/acceptance.js +7 -2
- package/dist/gates/baseline.d.ts +12 -1
- package/dist/gates/baseline.js +11 -4
- package/dist/gates/cache.d.ts +3 -1
- package/dist/gates/cache.js +10 -3
- package/dist/gates/llm.d.ts +5 -4
- package/dist/gates/llm.js +17 -14
- package/dist/gates/review.d.ts +8 -0
- package/dist/gates/review.js +47 -15
- package/dist/gates/run-gates.d.ts +6 -1
- package/dist/gates/run-gates.js +34 -17
- package/dist/gates/test-manifest.d.ts +14 -0
- package/dist/gates/test-manifest.js +33 -6
- package/dist/graph/graph.d.ts +6 -2
- package/dist/graph/graph.js +15 -4
- package/dist/graph/schema.d.ts +2 -0
- package/dist/graph/schema.js +2 -0
- package/dist/plan/scope.js +2 -2
- package/dist/route/preference.d.ts +20 -2
- package/dist/route/preference.js +48 -13
- package/dist/route/role-pick.d.ts +16 -0
- package/dist/route/role-pick.js +15 -0
- package/dist/route/router.d.ts +14 -0
- package/dist/route/router.js +39 -16
- package/dist/run/consult.d.ts +13 -1
- package/dist/run/consult.js +19 -14
- package/dist/run/daemon.d.ts +46 -2
- package/dist/run/daemon.js +964 -190
- package/dist/run/git.d.ts +44 -1
- package/dist/run/git.js +103 -5
- package/dist/run/host-health.d.ts +20 -0
- package/dist/run/host-health.js +64 -0
- package/dist/run/journal.d.ts +126 -3
- package/dist/run/journal.js +418 -36
- package/dist/run/merge.d.ts +3 -1
- package/dist/run/merge.js +3 -2
- package/dist/run/operator-state.d.ts +24 -2
- package/dist/run/operator-state.js +41 -5
- package/dist/run/operator-summary.d.ts +3 -0
- package/dist/run/operator-summary.js +3 -1
- package/dist/run/protocol.d.ts +31 -1
- package/dist/run/protocol.js +3 -1
- package/dist/run/stall.d.ts +38 -2
- package/dist/run/stall.js +276 -6
- package/dist/run/supervision.d.ts +7 -1
- package/dist/run/supervision.js +5 -2
- package/dist/tui/cockpit/board.d.ts +1 -1
- package/dist/tui/cockpit/board.js +30 -22
- package/dist/tui/cockpit/decision-actions.d.ts +8 -5
- package/dist/tui/cockpit/decision-actions.js +55 -32
- package/dist/tui/cockpit/derive.d.ts +2 -0
- package/dist/tui/cockpit/derive.js +17 -2
- package/dist/tui/cockpit/live-runtime.d.ts +10 -0
- package/dist/tui/cockpit/live-runtime.js +50 -3
- package/dist/tui/cockpit/live-store.d.ts +1 -0
- package/dist/tui/cockpit/live-store.js +31 -8
- package/dist/tui/cockpit/run-cockpit.d.ts +3 -0
- package/dist/tui/cockpit/run-cockpit.js +28 -2
- package/dist/tui/cockpit/run-view.d.ts +11 -7
- package/dist/tui/cockpit/run-view.js +69 -15
- package/dist/tui/cockpit/setup-cockpit.d.ts +4 -0
- package/dist/tui/cockpit/setup-cockpit.js +6 -3
- package/dist/tui/ink/fleet-app.d.ts +15 -3
- package/dist/tui/ink/fleet-app.js +91 -22
- package/package.json +2 -1
- package/schema/config.schema.json +818 -0
- package/skills/tickmarkr-loop/SKILL.md +8 -2
- package/skills/tickmarkr-overseer/SKILL.md +85 -6
- package/skills/tickmarkr-overseer/scripts/classify-vitest-log.sh +88 -0
- package/skills/tickmarkr-overseer/scripts/context-statusline.sh +81 -0
- package/skills/tickmarkr-overseer/scripts/grade-ci.sh +36 -34
- package/skills/tickmarkr-overseer/scripts/watch-journal.sh +6 -4
package/dist/gates/llm.d.ts
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { type WorkerAdapter } from "../adapters/types.js";
|
|
2
|
+
import type { Effort } from "../graph/schema.js";
|
|
2
3
|
import { type ExecutorDriver, type Slot } from "../drivers/types.js";
|
|
3
4
|
export declare const GATE_PANE_SEP = " \u00B7 ";
|
|
4
5
|
export declare const COMPLETION_FAKING_CHECKLIST = "## Completion-faking checklist\nHunt for these concrete completion-faking shortcuts before ruling on any criterion:\n- hardcoded-result: output or fixture hardcoded to satisfy the stated criterion instead of real logic\n- test-weakening: tests skipped, deleted, or assertions loosened until failing behavior looks green\n- vacuous-assertion: a test that cannot fail (asserts a constant, asserts its own setup, no assertion)\n- fixture-overfit: implementation narrowed to the exact test inputs rather than the described behavior\n- echo-not-implement: criterion text echoed in names, comments, or strings without the behavior itself\n- stub-left-behind: TODO, throw, or no-op stub where the real implementation should be\n- error-swallowing: catch or fallback that hides failures instead of handling them\n- self-mocking: the code under test mocked or faked so the test exercises the mock\n- check-bypass: lint, type, or CI checks disabled, relaxed, or excluded to get green\n- rename-as-work: code moved or renamed and presented as the requested change\n- scope-padding: unrelated edits padding the diff while the criterion's behavior is untouched\nWhen a criterion fails, the verdict MUST name which shortcut above it matches, or state that none does.";
|
|
@@ -70,11 +71,11 @@ export declare const PROMPT_GLYPHS: readonly ["➜", "❯", "$", "%", ">>", ">"]
|
|
|
70
71
|
export declare function reviewSeatOutput(raw: string, nonce: string, adapterBannerRows?: readonly string[]): string;
|
|
71
72
|
export declare const REVIEW_FIRST_LIVENESS_MS = 30000;
|
|
72
73
|
export declare const REVIEW_SILENT_BYTE_FLOOR = 64;
|
|
73
|
-
export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number): Promise<string>;
|
|
74
|
-
export declare function runViaDriver(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via: LlmVia, timeoutMs?: number): Promise<string>;
|
|
74
|
+
export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number, effort?: Effort): Promise<string>;
|
|
75
|
+
export declare function runViaDriver(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via: LlmVia, timeoutMs?: number, effort?: Effort): Promise<string>;
|
|
75
76
|
export declare function dewrapPaneVerdict(out: string, nonce: string): string;
|
|
76
|
-
export declare function runLlmDetailed(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via?: LlmVia, timeoutMs?: number): Promise<LlmRunResult>;
|
|
77
|
-
export declare function runLlm(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via?: LlmVia, timeoutMs?: number): Promise<string>;
|
|
77
|
+
export declare function runLlmDetailed(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via?: LlmVia, timeoutMs?: number, effort?: Effort): Promise<LlmRunResult>;
|
|
78
|
+
export declare function runLlm(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via?: LlmVia, timeoutMs?: number, effort?: Effort): Promise<string>;
|
|
78
79
|
export declare function extractJson<T>(raw: string): T | null;
|
|
79
80
|
/** Fable F3: verdict JSON must echo the call nonce — skip unbound or mismatched objects. */
|
|
80
81
|
export declare function extractVerdictJson<T>(raw: string, nonce: string): T | null;
|
package/dist/gates/llm.js
CHANGED
|
@@ -304,12 +304,12 @@ export const REVIEW_FIRST_LIVENESS_MS = 30_000;
|
|
|
304
304
|
// ceiling. Below this many seat-authored bytes at the first beat the seat is `silent` — demoted and
|
|
305
305
|
// re-routed then, not at the ceiling. Pane path only; a headless runner buffers and keeps its ceiling.
|
|
306
306
|
export const REVIEW_SILENT_BYTE_FLOOR = 64;
|
|
307
|
-
async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 300000) {
|
|
307
|
+
async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 300000, effort) {
|
|
308
308
|
const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
|
|
309
309
|
try {
|
|
310
310
|
const pf = join(dir, "prompt.md");
|
|
311
311
|
writeFileSync(pf, prompt);
|
|
312
|
-
const r = await sh(adapter.headlessCommand(pf, model), cwd, timeoutMs);
|
|
312
|
+
const r = await sh(adapter.headlessCommand(pf, model, effort), cwd, timeoutMs);
|
|
313
313
|
const output = r.stdout + "\n" + r.stderr;
|
|
314
314
|
const nonce = extractPromptNonce(prompt) ?? "";
|
|
315
315
|
return { output, exitCode: r.code, timedOut: r.timedOut === true,
|
|
@@ -319,12 +319,12 @@ async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 3000
|
|
|
319
319
|
rmSync(dir, { recursive: true, force: true });
|
|
320
320
|
}
|
|
321
321
|
}
|
|
322
|
-
export async function runHeadless(adapter, model, prompt, cwd, timeoutMs = 300000) {
|
|
323
|
-
return (await runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs)).output;
|
|
322
|
+
export async function runHeadless(adapter, model, prompt, cwd, timeoutMs = 300000, effort) {
|
|
323
|
+
return (await runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs, effort)).output;
|
|
324
324
|
}
|
|
325
325
|
// v1.1 default path: the same headless CLI call, but dispatched through the driver
|
|
326
326
|
// as a visible named agent (herdr pane), with the quote-split completion wrapper.
|
|
327
|
-
async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
|
|
327
|
+
async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs = 300000, effort) {
|
|
328
328
|
const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
|
|
329
329
|
let slot;
|
|
330
330
|
let accountant;
|
|
@@ -338,7 +338,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
|
|
|
338
338
|
writeFileSync(scriptPath, [
|
|
339
339
|
"export BASH_SILENCE_DEPRECATION_WARNING=1",
|
|
340
340
|
bannerShell(),
|
|
341
|
-
adapter.headlessCommand(pf, model),
|
|
341
|
+
adapter.headlessCommand(pf, model, effort),
|
|
342
342
|
gateExitTrailer(nonce),
|
|
343
343
|
].join("\n"));
|
|
344
344
|
slot = await via.driver.slot(cwd, rolePaneNameFromPrompt(prompt, via.name), via.label ? { label: via.label } : undefined);
|
|
@@ -403,7 +403,10 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
|
|
|
403
403
|
// That seat was killed by its configured timeout, not an early launch reroute.
|
|
404
404
|
if (now - startedAt >= timeoutMs)
|
|
405
405
|
break;
|
|
406
|
-
|
|
406
|
+
// OBS-1177: the beat is armed for an INTERACTIVE (pane) driver only. On the subprocess driver
|
|
407
|
+
// a headless `claude -p` buffers every byte until it exits, so 0 seat bytes at 30 s is not a
|
|
408
|
+
// dead launch; that seat keeps its full ceiling, which still bounds it (fail-closed by timeout).
|
|
409
|
+
if (reviewing && via.driver.interactive && !firstLivenessObserved && now - startedAt >= REVIEW_FIRST_LIVENESS_MS) {
|
|
407
410
|
firstLivenessObserved = true;
|
|
408
411
|
// RS-2: the beat reads seat-authored bytes ALONE. CPU evidence never holds a preamble-only
|
|
409
412
|
// capture open to the ceiling; a seat that has not written one byte of its own is re-routed.
|
|
@@ -484,8 +487,8 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
|
|
|
484
487
|
}
|
|
485
488
|
}
|
|
486
489
|
}
|
|
487
|
-
export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
|
|
488
|
-
return (await runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs)).output;
|
|
490
|
+
export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs = 300000, effort) {
|
|
491
|
+
return (await runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs, effort)).output;
|
|
489
492
|
}
|
|
490
493
|
// OBS-155: a TUI renders the verdict as a bullet and HARD-wraps it at pane width with a 2-space
|
|
491
494
|
// continuation indent, splitting words mid-token — so literal newlines land inside JSON string
|
|
@@ -546,15 +549,15 @@ export function dewrapPaneVerdict(out, nonce) {
|
|
|
546
549
|
}
|
|
547
550
|
return out;
|
|
548
551
|
}
|
|
549
|
-
export async function runLlmDetailed(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
|
|
552
|
+
export async function runLlmDetailed(adapter, model, prompt, cwd, via, timeoutMs = 300000, effort) {
|
|
550
553
|
const result = await (via
|
|
551
|
-
? runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs)
|
|
552
|
-
: runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs));
|
|
554
|
+
? runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs, effort)
|
|
555
|
+
: runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs, effort));
|
|
553
556
|
llmOutputCapture.getStore()?.push(result.output);
|
|
554
557
|
return result;
|
|
555
558
|
}
|
|
556
|
-
export async function runLlm(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
|
|
557
|
-
return (await runLlmDetailed(adapter, model, prompt, cwd, via, timeoutMs)).output;
|
|
559
|
+
export async function runLlm(adapter, model, prompt, cwd, via, timeoutMs = 300000, effort) {
|
|
560
|
+
return (await runLlmDetailed(adapter, model, prompt, cwd, via, timeoutMs, effort)).output;
|
|
558
561
|
}
|
|
559
562
|
export function extractJson(raw) {
|
|
560
563
|
const fenced = [...raw.matchAll(/```json\s*\n([\s\S]*?)```/g)].at(-1);
|
package/dist/gates/review.d.ts
CHANGED
|
@@ -135,4 +135,12 @@ export declare function renderGoalSection(goal: string, repoRoot?: string): stri
|
|
|
135
135
|
* closure on a typo and that read as malformed — the block is what a closure list is copied from.
|
|
136
136
|
*/
|
|
137
137
|
export declare function renderPriorMaterials(priorMaterials: readonly StructuredFinding[]): string;
|
|
138
|
+
/**
|
|
139
|
+
* OBS-1052(2): a seat that lost the top of a long brief, or believed it had already filed its review,
|
|
140
|
+
* answered in prose — and prose is no verdict. So the requirement, naming THIS call's nonce with a
|
|
141
|
+
* valid example, is both the first and the last instruction of the brief. It is best-effort wording:
|
|
142
|
+
* the parser stays the authority, and nothing here reads approval out of prose.
|
|
143
|
+
*/
|
|
144
|
+
export declare function reviewResponseExample(nonce: string): string;
|
|
145
|
+
export declare function reviewResponseRequirement(nonce: string): string;
|
|
138
146
|
export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string, reviewHistory?: string[], demotedReviewers?: ReadonlySet<string>, carriedFindings?: readonly StructuredFinding[], priorReviewers?: readonly PriorReviewer[], carriedAuthors?: readonly string[], operatorContext?: string): Promise<GateResult>;
|
package/dist/gates/review.js
CHANGED
|
@@ -8,10 +8,10 @@ import { getAdapter } from "../adapters/registry.js";
|
|
|
8
8
|
import { shOk } from "../run/git.js";
|
|
9
9
|
import { carryReviewFindings, observedReviewFingerprints, reviewFingerprintMatches, structuredFindings, UNIDENTIFIED } from "../run/journal.js";
|
|
10
10
|
import { redactSecrets } from "../run/redact.js";
|
|
11
|
-
import {
|
|
11
|
+
import { rankPreferredChannels, reviewPreferenceTieBreak } from "../route/role-pick.js";
|
|
12
12
|
import { modelProvider } from "../route/preference.js";
|
|
13
13
|
import { resolveStateDir } from "./cache.js";
|
|
14
|
-
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, parseAnchoredComments, runLlmDetailed, verdictNonceLine } from "./llm.js";
|
|
14
|
+
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, dewrapPaneVerdict, extractVerdictJson, generateVerdictNonce, parseAnchoredComments, runLlmDetailed, verdictNonceLine } from "./llm.js";
|
|
15
15
|
import { classifyVerdictCause } from "./verdict-cause.js";
|
|
16
16
|
import { captureDiffCapFor, measureArtifactDiff, reviewableLogicDiff, } from "./artifact-manifest.js";
|
|
17
17
|
export { isProtectedEvidence, PROTECTED_EVIDENCE_PREFIXES, REGENERABLE_CAPTURE_PATHS, setAsideReceiptPath, setAsideRegenerableCaptures, } from "./artifact-manifest.js";
|
|
@@ -227,13 +227,6 @@ export function isReviewClosureMismatch(v, priorIds) {
|
|
|
227
227
|
const priors = priorIds instanceof Set ? priorIds : new Set(priorIds);
|
|
228
228
|
return ids.some((id) => matchClosureId(id, priors) === undefined);
|
|
229
229
|
}
|
|
230
|
-
// v1.53 T2: same entry grammar as routing.map.prefer (router.ts preferIndex — router is out of this
|
|
231
|
-
// module's dependency direction for a private fn, so the 3 lines live here too): `adapter` matches
|
|
232
|
-
// every channel of that adapter, `adapter:model` exactly one; unmatched channels sort after all entries.
|
|
233
|
-
function reviewPreferIndex(c, prefer) {
|
|
234
|
-
const i = prefer.findIndex((p) => p === c.adapter || p === channelKey(c));
|
|
235
|
-
return i === -1 ? prefer.length : i;
|
|
236
|
-
}
|
|
237
230
|
/**
|
|
238
231
|
* RF-1 (OBS-922 add.2/3): the tier a reviewer must meet is the maximum of the author's tier, the
|
|
239
232
|
* task-declared floor, a configured `review.floor` tier and, on a second round or a retry, the prior
|
|
@@ -290,7 +283,7 @@ excludeVendors = new Set(), authors = [channelKey(author)]) {
|
|
|
290
283
|
return null;
|
|
291
284
|
// RF-1: every caller inherits the author-tier floor — a reviewer is never seated below its author.
|
|
292
285
|
const effectiveFloor = resolveReviewerFloor(author.tier, floor).floor;
|
|
293
|
-
const
|
|
286
|
+
const eligible = channels
|
|
294
287
|
// Three independent axes: different vendor, different resolved provider identity (OBS-946: on initial pick
|
|
295
288
|
// as well as failover, so an aggregator channel stamped "mixed" never seats the author's own provider),
|
|
296
289
|
// and different base-model identity (ADDED TO the vendor rule, never replacing it). The diversity
|
|
@@ -300,8 +293,11 @@ excludeVendors = new Set(), authors = [channelKey(author)]) {
|
|
|
300
293
|
&& modelId(c.model) !== modelId(a.model))
|
|
301
294
|
&& !exclude.includes(channelKey(c))
|
|
302
295
|
&& !excludeVendors.has(c.vendor)
|
|
303
|
-
&& TIER_RANK[c.tier] >= TIER_RANK[effectiveFloor])
|
|
304
|
-
|
|
296
|
+
&& TIER_RANK[c.tier] >= TIER_RANK[effectiveFloor]);
|
|
297
|
+
const ranked = rankPreferredChannels(eligible, prefer, {
|
|
298
|
+
includeUnpreferred: true,
|
|
299
|
+
tieBreak: reviewPreferenceTieBreak,
|
|
300
|
+
});
|
|
305
301
|
const reviewer = [...ranked].sort((a, b) => Number(demoted.has(channelKey(a))) - Number(demoted.has(channelKey(b)))
|
|
306
302
|
|| history.lastIndexOf(channelKey(a)) - history.lastIndexOf(channelKey(b))
|
|
307
303
|
|| ranked.indexOf(a) - ranked.indexOf(b))[0] ?? null;
|
|
@@ -377,6 +373,33 @@ ${fingerprints}
|
|
|
377
373
|
\`\`\`
|
|
378
374
|
${priorMaterials.map((finding, i) => `${i + 1}. ${finding.note}`).join("\n\n")}`;
|
|
379
375
|
}
|
|
376
|
+
/**
|
|
377
|
+
* OBS-1052(2): a seat that lost the top of a long brief, or believed it had already filed its review,
|
|
378
|
+
* answered in prose — and prose is no verdict. So the requirement, naming THIS call's nonce with a
|
|
379
|
+
* valid example, is both the first and the last instruction of the brief. It is best-effort wording:
|
|
380
|
+
* the parser stays the authority, and nothing here reads approval out of prose.
|
|
381
|
+
*/
|
|
382
|
+
export function reviewResponseExample(nonce) {
|
|
383
|
+
return JSON.stringify({
|
|
384
|
+
nonce, approve: false, resolved: [], reraised: [],
|
|
385
|
+
findings: [{ note: "path/to/file.ts:42 — the defect, in one line", severity: "material", defer: false, rationale: "" }],
|
|
386
|
+
comments: [],
|
|
387
|
+
});
|
|
388
|
+
}
|
|
389
|
+
export function reviewResponseRequirement(nonce) {
|
|
390
|
+
return `## Response requirement
|
|
391
|
+
Your reply must end with exactly ONE JSON object whose "nonce" is "${nonce}" — this brief's nonce, never one from an earlier brief. A valid example (a rejection; replace every value with your own verdict):
|
|
392
|
+
${reviewResponseExample(nonce)}
|
|
393
|
+
This holds even if you already filed or posted a review of this task elsewhere (an earlier session or brief, a PR comment): that review is not on record here, so restate it now as this JSON with nonce "${nonce}". Prose saying a review was filed or approved is recorded as no verdict; approval is never inferred from it.`;
|
|
394
|
+
}
|
|
395
|
+
// The example parses by design, so an echo of the brief (a CLI printing its prompt, a pane showing it)
|
|
396
|
+
// would otherwise read as the seat's own verdict — or as its participation when it wrote only prose.
|
|
397
|
+
// Removed verbatim or hard-wrapped (renderer whitespace and chrome between any two characters) before
|
|
398
|
+
// the verdict is extracted or its absence classified; the saved raw bytes keep it as evidence.
|
|
399
|
+
function withoutExampleEcho(raw, nonce) {
|
|
400
|
+
const chars = [...reviewResponseExample(nonce).replace(/\s+/g, "")];
|
|
401
|
+
return raw.replace(new RegExp(chars.map((c) => c.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")).join("[\\s│|]*"), "g"), "");
|
|
402
|
+
}
|
|
380
403
|
export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
|
|
381
404
|
// OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
|
|
382
405
|
// direct tests) skips persistence and changes nothing else.
|
|
@@ -486,7 +509,10 @@ carriedAuthors = [], operatorContext) {
|
|
|
486
509
|
const suiteBudget = ownTestFiles.length
|
|
487
510
|
? `You may run at most the task's own test files explicitly named in files[]; these are the only suites you may run: ${ownTestFiles.map((file) => `\`${file}\``).join(", ")}.`
|
|
488
511
|
: "No suite may be run: files[] names no explicit test file owned by this task.";
|
|
512
|
+
const responseRequirement = reviewResponseRequirement(nonce);
|
|
489
513
|
const prompt = `TICKMARKR-REVIEW
|
|
514
|
+
${responseRequirement}
|
|
515
|
+
|
|
490
516
|
You are a skeptical cross-vendor code reviewer. Another agent (vendor: ${author.adapter}) authored this diff.
|
|
491
517
|
Look for correctness bugs, security issues, and acceptance-criteria gaps. Approve only if you would merge it.
|
|
492
518
|
|
|
@@ -531,6 +557,8 @@ For every prior material, put its fingerprint in exactly one of resolved (verifi
|
|
|
531
557
|
(still a blocking defect). Use only the listed fingerprints; never omit one or put it in both lists.
|
|
532
558
|
Approve iff no material finding remains and every prior material is resolved.
|
|
533
559
|
The top-level comments array is optional. Use it only for actionable line-anchored feedback.
|
|
560
|
+
|
|
561
|
+
${responseRequirement}
|
|
534
562
|
`;
|
|
535
563
|
// Filenames are journaled (daemon.ts lifts meta.rawPath/briefPath onto the gate-result row), so they
|
|
536
564
|
// must be reproducible from the same inputs — the verdict nonce is cryptographically random and would
|
|
@@ -573,7 +601,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
573
601
|
// output until completion — runLlm's 300s default killed reviews mid-flight, returning empty
|
|
574
602
|
// stdout that read as "unparseable" and escalated to re-implementation of green code
|
|
575
603
|
// (run-20260709-104447 P87-09). The configured ceiling defaults to that measured 15 minutes.
|
|
576
|
-
cfg.review.timeoutMs);
|
|
604
|
+
cfg.review.timeoutMs, reviewer.effort);
|
|
577
605
|
const raw = llm.output;
|
|
578
606
|
let saved;
|
|
579
607
|
if (artifactDir) {
|
|
@@ -586,7 +614,11 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
586
614
|
}
|
|
587
615
|
}
|
|
588
616
|
const provider = modelProvider(reviewer.model, reviewer.vendor);
|
|
589
|
-
|
|
617
|
+
// A pane's own dewrap stops at the first parseable nonce-bound object; once the example's echo is gone
|
|
618
|
+
// a genuinely wrapped verdict behind it is reconstructed here, exactly as llm.ts would have.
|
|
619
|
+
const echoFree = withoutExampleEcho(raw, nonce);
|
|
620
|
+
const seat = via ? dewrapPaneVerdict(echoFree, nonce) : echoFree;
|
|
621
|
+
const v = extractVerdictJson(seat, nonce);
|
|
590
622
|
const findings = v && Array.isArray(v.findings) ? v.findings : null;
|
|
591
623
|
const priorIds = priorMaterials;
|
|
592
624
|
const closureInvalid = isReviewClosureInvalid(v, priorIds);
|
|
@@ -600,7 +632,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
600
632
|
: llm.launchNeverStarted ? "launch-never-started"
|
|
601
633
|
: llm.silentAtBeat ? "silent"
|
|
602
634
|
: llm.timedOut ? (bytes > 0 ? "truncated" : "silent")
|
|
603
|
-
: classifyVerdictCause(
|
|
635
|
+
: classifyVerdictCause(seat, nonce, "approve", llm);
|
|
604
636
|
const failure = cause === "malformed-verdict"
|
|
605
637
|
? "review output unparseable"
|
|
606
638
|
: cause === "closure-mismatch"
|
|
@@ -7,6 +7,7 @@ import { type GateVia } from "./llm.js";
|
|
|
7
7
|
import { type PriorReviewer } from "./review.js";
|
|
8
8
|
import type { GateResult } from "./types.js";
|
|
9
9
|
import { type VerificationRetryCause } from "../run/recovery.js";
|
|
10
|
+
import { type PreserveProducer } from "../run/git.js";
|
|
10
11
|
import { type StructuredFinding } from "../run/journal.js";
|
|
11
12
|
import { type VerificationScope } from "./cache.js";
|
|
12
13
|
export type LoadProvider = () => number;
|
|
@@ -16,7 +17,8 @@ export declare function resetLoadProviderForTests(): void;
|
|
|
16
17
|
/**
|
|
17
18
|
* One gate's own measurement, taken WHERE THE GATE RUNS. `durationMs` sums that gate's execution
|
|
18
19
|
* intervals and nothing between them, so the composite `test` gate (a selected screen, then other
|
|
19
|
-
* gates, then the full suite) reports the two suites' cost rather than the span containing them
|
|
20
|
+
* gates, then the full suite) reports the two suites' cost rather than the span containing them
|
|
21
|
+
* (split across the two rows when the screen is published before semantic gates, OBS-1176) —
|
|
20
22
|
* and no consumer has to re-derive a duration by subtracting journal timestamps, which measures the
|
|
21
23
|
* queue as well as the work. Load is sampled at each interval's endpoints and every second within it;
|
|
22
24
|
* start preserves the scheduling input while max and mean retain sustained interior saturation.
|
|
@@ -71,6 +73,9 @@ export interface GateContext {
|
|
|
71
73
|
/** Explicit worker funding requires fresh red measurements, never a gate waiver. */
|
|
72
74
|
cachedRedBypass?: "operator-rerun";
|
|
73
75
|
carriedAuthors?: readonly string[];
|
|
76
|
+
/** The attempt whose worker last wrote the gated checkout; a dirty-tree refusal stamps it on the
|
|
77
|
+
* preserve commit and its row. Absent (standalone verify, gate-only restores) preserves as "unknown". */
|
|
78
|
+
producer?: PreserveProducer;
|
|
74
79
|
reviewHistory?: string[];
|
|
75
80
|
priorReviewers?: PriorReviewer[];
|
|
76
81
|
artifactDir?: string;
|
package/dist/gates/run-gates.js
CHANGED
|
@@ -2,7 +2,7 @@ import { randomUUID } from "node:crypto";
|
|
|
2
2
|
import { existsSync, mkdtempSync, readFileSync, rmSync, statSync } from "node:fs";
|
|
3
3
|
import { loadavg, tmpdir } from "node:os";
|
|
4
4
|
import { join, posix } from "node:path";
|
|
5
|
-
import { channelKey, shq } from "../adapters/types.js";
|
|
5
|
+
import { channelKey, configuredEffort, shq } from "../adapters/types.js";
|
|
6
6
|
import { TIER_RANK } from "../config/config.js";
|
|
7
7
|
import { getAdapter } from "../adapters/registry.js";
|
|
8
8
|
import { GATE_NAMES } from "../graph/schema.js";
|
|
@@ -17,7 +17,7 @@ import { scopeGate } from "./scope.js";
|
|
|
17
17
|
import { evaluateManifestedTest, isVitestTestCommand } from "./test-manifest.js";
|
|
18
18
|
import { executionSignal } from "../run/execution-budget.js";
|
|
19
19
|
import { failureDisposition } from "../run/recovery.js";
|
|
20
|
-
import { dependencyLinkRefusal, preserveWorktree, shGit, resolvedCapacity, verificationProtocol } from "../run/git.js";
|
|
20
|
+
import { dependencyLinkRefusal, preserveWorktree, producerFields, shGit, resolvedCapacity, verificationProtocol } from "../run/git.js";
|
|
21
21
|
import { withJudgeInvocationEvidence } from "../run/journal.js";
|
|
22
22
|
import { computeVerificationIdentity, verificationIdentityKey, formatReusedRow, getVerdictStore, isInfraResult, resolveStateDir, reusedIdentity, } from "./cache.js";
|
|
23
23
|
const productionLoadProvider = () => loadavg()[0] ?? 0;
|
|
@@ -43,13 +43,14 @@ function instrumentLlmAdapter(adapter, clocks) {
|
|
|
43
43
|
return new Proxy(adapter, {
|
|
44
44
|
get(target, property) {
|
|
45
45
|
if (property === "headlessCommand") {
|
|
46
|
-
return (promptFile, model) => {
|
|
47
|
-
const command = target.headlessCommand(promptFile, model);
|
|
46
|
+
return (promptFile, model, effort) => {
|
|
47
|
+
const command = target.headlessCommand(promptFile, model, effort);
|
|
48
48
|
const dir = mkdtempSync(join(tmpdir(), "tickmarkr-gate-invocation-"));
|
|
49
49
|
const startedAtPath = join(dir, "started-at");
|
|
50
50
|
const completedAtPath = join(dir, "completed-at");
|
|
51
51
|
clocks.push({
|
|
52
52
|
channel: channelKey({ adapter: target.id, model }),
|
|
53
|
+
...(effort ? { effort } : {}),
|
|
53
54
|
preparedAt: Date.now(),
|
|
54
55
|
startedAtPath,
|
|
55
56
|
completedAtPath,
|
|
@@ -85,7 +86,7 @@ function finishLlmDispatches(clocks) {
|
|
|
85
86
|
finally {
|
|
86
87
|
rmSync(clock.dir, { recursive: true, force: true });
|
|
87
88
|
}
|
|
88
|
-
return { channel: clock.channel, durationMs: completedAt - startedAt };
|
|
89
|
+
return { channel: clock.channel, ...(clock.effort ? { effort: clock.effort } : {}), durationMs: completedAt - startedAt };
|
|
89
90
|
});
|
|
90
91
|
}
|
|
91
92
|
async function captureLlmDispatches(adapters, run) {
|
|
@@ -223,6 +224,7 @@ async function runVitestManifestGate(worktree, cmd, baseline, selected, artifact
|
|
|
223
224
|
overallCeilingMs: effectiveCeilingMs(entry),
|
|
224
225
|
artifactDir,
|
|
225
226
|
evidence: retry.evidence,
|
|
227
|
+
retryBaseCommand: retry.retryBaseCommand, // OBS-1166: the un-narrowed command for a selected screen's stranded retry
|
|
226
228
|
});
|
|
227
229
|
const reportPath = outcome.reportPath;
|
|
228
230
|
const evidence = { evidenceReceipt: outcome.evidenceReceipt, evidenceReceipts: outcome.evidenceReceipts };
|
|
@@ -340,10 +342,14 @@ export async function runGates(task, ctx) {
|
|
|
340
342
|
const enabled = (g) => task.gates.includes(g) && (g !== "acceptance" && g !== "review" || shapeGates?.[g] !== false);
|
|
341
343
|
const failed = () => results.some((r) => !r.pass);
|
|
342
344
|
// T4 (OBS-265): a GREEN selected-test run is a screen, not the round's verdict — the merge-candidate
|
|
343
|
-
// round re-runs the full suite on the same commit and THAT is what the round reports.
|
|
344
|
-
//
|
|
345
|
+
// round re-runs the full suite on the same commit and THAT is what the round reports. With no
|
|
346
|
+
// semantic gate to act on it, the screen is held so its full suite speaks for it in one row.
|
|
345
347
|
// (A RED screen IS the verdict: the round ends there, so it is recorded immediately.)
|
|
346
348
|
let heldTest;
|
|
349
|
+
// OBS-1176: when acceptance/review WILL act on a green screen, the screen is published before they
|
|
350
|
+
// start, as its own selected row. The full suite afterwards is a second invocation on its own row —
|
|
351
|
+
// it carries only its own receipts and interval, so it neither erases nor re-counts the screen.
|
|
352
|
+
const publishScreen = enabled("acceptance") || enabled("review");
|
|
347
353
|
// v2.0 T2 (OBS-554): this round's per-gate measurement. Every interval a gate actually spends
|
|
348
354
|
// executing is added HERE, at the call site that runs it, so a gate that runs twice (the test
|
|
349
355
|
// gate's screen and its full suite) sums to its own cost and never to the span between them.
|
|
@@ -482,7 +488,7 @@ export async function runGates(task, ctx) {
|
|
|
482
488
|
let preservedRef;
|
|
483
489
|
let preservationError;
|
|
484
490
|
try {
|
|
485
|
-
preservedRef = await preserveWorktree(ctx.worktree);
|
|
491
|
+
preservedRef = await preserveWorktree(ctx.worktree, ctx.producer);
|
|
486
492
|
}
|
|
487
493
|
catch (error) {
|
|
488
494
|
// Never masks the refusal, but never pretends a snapshot exists either — surfaced below.
|
|
@@ -537,6 +543,7 @@ export async function runGates(task, ctx) {
|
|
|
537
543
|
dirtyWorktree: true,
|
|
538
544
|
ref: preservedRef,
|
|
539
545
|
preservedRef,
|
|
546
|
+
...producerFields(ctx.producer),
|
|
540
547
|
paths: dirtyPaths,
|
|
541
548
|
files: dirtyPaths,
|
|
542
549
|
path: primaryFile,
|
|
@@ -583,7 +590,7 @@ export async function runGates(task, ctx) {
|
|
|
583
590
|
let preservedRef;
|
|
584
591
|
let preservationError;
|
|
585
592
|
try {
|
|
586
|
-
preservedRef = await preserveWorktree(ctx.worktree);
|
|
593
|
+
preservedRef = await preserveWorktree(ctx.worktree, ctx.producer);
|
|
587
594
|
}
|
|
588
595
|
catch (error) {
|
|
589
596
|
preservationError = error instanceof Error ? error.message : String(error);
|
|
@@ -614,6 +621,7 @@ export async function runGates(task, ctx) {
|
|
|
614
621
|
dirtyAtRoundEnd: true,
|
|
615
622
|
ref: preservedRef,
|
|
616
623
|
preservedRef,
|
|
624
|
+
...producerFields(ctx.producer),
|
|
617
625
|
paths: dirtyPaths,
|
|
618
626
|
files: dirtyPaths,
|
|
619
627
|
path: primaryFile,
|
|
@@ -677,7 +685,7 @@ export async function runGates(task, ctx) {
|
|
|
677
685
|
// other scripted test command keeps today's exit-code contract byte-identically.
|
|
678
686
|
const useManifest = g === "test" && commands.test !== undefined && isVitestTestCommand(commands.test, ctx.worktree);
|
|
679
687
|
r = useManifest
|
|
680
|
-
? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir, { ...retryOptions(identity), evidence }))
|
|
688
|
+
? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir, { ...retryOptions(identity), evidence, ...(selected ? { retryBaseCommand: ctx.commands.test } : {}) }))
|
|
681
689
|
: (await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], { ...retryOptions(identity), evidence, ...(g === "build" ? { onReceipt: buildReceipt, taskBuildAttribution: beginBuild } : {}), ...(g === "test" && selected ? { selected } : {}) })))[0];
|
|
682
690
|
}
|
|
683
691
|
finally {
|
|
@@ -708,6 +716,13 @@ export async function runGates(task, ctx) {
|
|
|
708
716
|
const screened = { ...r, meta: { ...r.meta, selectedTests: selected } };
|
|
709
717
|
if (!screened.pass)
|
|
710
718
|
await record(screened);
|
|
719
|
+
else if (publishScreen) {
|
|
720
|
+
await record(screened);
|
|
721
|
+
// The screen's interval now lives on its own row; the full suite measures from zero.
|
|
722
|
+
spans.delete("test");
|
|
723
|
+
loadSamples.delete("test");
|
|
724
|
+
selectedDurationMs = undefined;
|
|
725
|
+
}
|
|
711
726
|
else {
|
|
712
727
|
heldTest = withTelemetry(screened);
|
|
713
728
|
results.push(heldTest);
|
|
@@ -812,8 +827,8 @@ export async function runGates(task, ctx) {
|
|
|
812
827
|
// Separate from `invocations` above deliberately: that array is transcript evidence and records
|
|
813
828
|
// one entry per CAPTURED OUTPUT, so a dispatch that produced none contributes nothing to it.
|
|
814
829
|
const invocationSpans = [];
|
|
815
|
-
const invokeJudge = async (adapter, model, via) => {
|
|
816
|
-
const captured = await captureLlmDispatches([adapter], ([instrumented]) => acceptanceGate(task, ctx.worktree, ctx.baseRef, { adapter: instrumented, model }, via, { testCmd: ctx.commands.test, diffCap: ctx.cfg.gates.diffCap }));
|
|
830
|
+
const invokeJudge = async (adapter, model, via, effort) => {
|
|
831
|
+
const captured = await captureLlmDispatches([adapter], ([instrumented]) => acceptanceGate(task, ctx.worktree, ctx.baseRef, { adapter: instrumented, model, effort }, via, { testCmd: ctx.commands.test, diffCap: ctx.cfg.gates.diffCap }));
|
|
817
832
|
// The instrumented adapter is reached only by runLlm. Deterministic oracles and diff-cap exits
|
|
818
833
|
// never call headlessCommand, so they produce no clock and cannot manufacture an invocation.
|
|
819
834
|
invocationSpans.push(...captured.invocations);
|
|
@@ -835,7 +850,8 @@ export async function runGates(task, ctx) {
|
|
|
835
850
|
}
|
|
836
851
|
return captured.value;
|
|
837
852
|
};
|
|
838
|
-
|
|
853
|
+
// OBS-1182: every judge seat launches at its OWN configured effort, never the worker's.
|
|
854
|
+
let a = await invokeJudge(judgeAdapter, ctx.cfg.judge.model, jvia, configuredEffort(ctx.cfg, ctx.cfg.judge));
|
|
839
855
|
// GATE-09: an unparseable judge verdict retries the JUDGE exactly once on a failover channel — never
|
|
840
856
|
// the worker (run-20260711-185020 P43-03 L70-72 billed a judge flake as a worker attempt). The flaked
|
|
841
857
|
// first verdict NEVER enters results (no false gate-result journal event, no operator notify, no stale
|
|
@@ -871,7 +887,7 @@ export async function runGates(task, ctx) {
|
|
|
871
887
|
? { driver: ctx.via.driver, keep: ctx.via.keep, onSlot: ctx.via.onSlot, name: ctx.via.nameFor("judge", retryAdapter.id) + "-r1", label: ctx.via.labelFor("judge") }
|
|
872
888
|
: undefined;
|
|
873
889
|
// the retry IS a second acceptanceGate call: one code path, one parser, zero new parse leniency.
|
|
874
|
-
a = await invokeJudge(retryAdapter, retry.model, retryJvia);
|
|
890
|
+
a = await invokeJudge(retryAdapter, retry.model, retryJvia, configuredEffort(ctx.cfg, retry));
|
|
875
891
|
a = { ...a, meta: { ...a.meta, judgeRetry: { flaked: flakedKey, retried: channelKey({ adapter: retry.adapter, model: retry.model }) } } };
|
|
876
892
|
}
|
|
877
893
|
// No dispatch, no key: a deterministic-oracle round writes no `invocations` field rather than an
|
|
@@ -1087,9 +1103,10 @@ export async function runGates(task, ctx) {
|
|
|
1087
1103
|
}
|
|
1088
1104
|
// The merge-candidate round: every other gate is green, so THIS round is the one that can merge —
|
|
1089
1105
|
// the full suite runs on the exact gated commit before the pipeline reports green. Nothing merges
|
|
1090
|
-
// on a subset (spec: "nothing merges without a complete green suite"). Its verdict
|
|
1091
|
-
//
|
|
1092
|
-
//
|
|
1106
|
+
// on a subset (spec: "nothing merges without a complete green suite"). Its verdict replaces the
|
|
1107
|
+
// screen's entry in the returned record (one `test` entry), and `fullSuite` says which suite spoke
|
|
1108
|
+
// while `selectedTests` keeps what the screen ran. In the stream, a held screen is superseded (one
|
|
1109
|
+
// `test` end event); a published screen keeps its own earlier event and this is the second.
|
|
1093
1110
|
if (selected) {
|
|
1094
1111
|
await emitStart("test");
|
|
1095
1112
|
// This is the last shell command a round can run — the judge's named-test oracle (acceptance.ts)
|
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
import { type GateEvidenceOptions, type BaselineFileDuration } from "./baseline.js";
|
|
2
2
|
import type { GateEvidenceReceipt } from "../run/protocol.js";
|
|
3
|
+
export declare const VITEST_CACHE_ENV = "TICKMARKR_VITEST_CACHE_DIR";
|
|
4
|
+
export declare function worktreeVitestCache(cwd: string, inherited?: string): string;
|
|
3
5
|
export declare function isVitestTestCommand(cmd: string, cwd: string): boolean;
|
|
4
6
|
/** One identity for every path this module compares: repo-relative, forward-slash. `vitest list
|
|
5
7
|
* --json` and `TestModule.moduleId` both hand back an absolute filesystem path already resolved
|
|
@@ -144,6 +146,15 @@ export declare function discoverTestManifest(cmd: string, cwd: string, opts: {
|
|
|
144
146
|
* runner cannot list (a null compares nothing — it never manufactures a deficit). The suite is not
|
|
145
147
|
* run here; the capture already ran it once. */
|
|
146
148
|
export declare function manifestFileCount(cmd: string, cwd: string): Promise<number | null>;
|
|
149
|
+
/** The stranded single-fork retry: positional filters are substring matches that OR with any filter
|
|
150
|
+
* the command already carries, and under a `projects` config the CLI `--exclude` never subtracts such a
|
|
151
|
+
* selection (OBS-1166: a selected screen's retry rediscovered the whole selection and refused). So the
|
|
152
|
+
* retry is built from the UN-narrowed base command, its own `--` rule, the stranded files as the only
|
|
153
|
+
* positional filters, and an `--exclude` of every completed file; the caller then requires discovery
|
|
154
|
+
* to prove the exact retry set before launch. OBS-1180: Vitest matches an absolute filter against the
|
|
155
|
+
* module path it resolved through every symlink, so the filters are rooted at the canonical realpath of
|
|
156
|
+
* `cwd` — a symlinked worktree root otherwise filters to an empty discovery. */
|
|
157
|
+
export declare function singleForkRetryCommand(base: string, cwd: string, stranded: readonly string[], completed: readonly string[]): string;
|
|
147
158
|
/** One configured runner execution, and its own collection under the same arguments and environment.
|
|
148
159
|
* The installed runner is trusted (R28 add.1 option B); the nonce catches stale artifacts, not forgery. */
|
|
149
160
|
export declare function evaluateManifestedTest(cmd: string, cwd: string, opts: {
|
|
@@ -152,4 +163,7 @@ export declare function evaluateManifestedTest(cmd: string, cwd: string, opts: {
|
|
|
152
163
|
overallCeilingMs?: number;
|
|
153
164
|
artifactDir?: string;
|
|
154
165
|
evidence?: GateEvidenceOptions;
|
|
166
|
+
/** OBS-1166: the configured command BEFORE a selected screen narrowed it, so a stranded retry's
|
|
167
|
+
* only positional filters are the stranded files. Absent (a full suite), `cmd` is that command. */
|
|
168
|
+
retryBaseCommand?: string;
|
|
155
169
|
}): Promise<ManifestGateOutcome>;
|
|
@@ -1,11 +1,20 @@
|
|
|
1
1
|
import { createHash, randomBytes } from "node:crypto";
|
|
2
|
-
import { existsSync, mkdtempSync, readFileSync, realpathSync, writeFileSync } from "node:fs";
|
|
2
|
+
import { existsSync, mkdtempSync, readFileSync, realpathSync, rmSync, writeFileSync } from "node:fs";
|
|
3
3
|
import { tmpdir } from "node:os";
|
|
4
|
-
import { isAbsolute, join, relative, sep } from "node:path";
|
|
4
|
+
import { isAbsolute, join, relative, resolve, sep } from "node:path";
|
|
5
5
|
import { TEST_REPORTER_SOURCE } from "./test-reporter.js";
|
|
6
6
|
import { shq } from "../adapters/types.js";
|
|
7
7
|
import { beginGateEvidence, redactGateOutput } from "./baseline.js";
|
|
8
8
|
import { FORK_CAP_ENV, ROUTING_ENV_SEAMS, SUITE_PARENT_ENV, shell, resolvedCapacity, verificationProtocol } from "../run/git.js";
|
|
9
|
+
// Outside the shared dependency symlink and ignored by the shipped repository.
|
|
10
|
+
export const VITEST_CACHE_ENV = "TICKMARKR_VITEST_CACHE_DIR";
|
|
11
|
+
export function worktreeVitestCache(cwd, inherited) {
|
|
12
|
+
const local = resolve(cwd, ".vitest-cache");
|
|
13
|
+
const candidate = inherited ? resolve(cwd, inherited) : local;
|
|
14
|
+
const within = relative(local, candidate);
|
|
15
|
+
return within === "" || (!isAbsolute(within) && within !== ".." && !within.startsWith(`..${sep}`))
|
|
16
|
+
? candidate : local;
|
|
17
|
+
}
|
|
9
18
|
/**
|
|
10
19
|
* VL-1 (OBS-985 lineage): a test gate's completion must be the runner's OWN report, never a stdout
|
|
11
20
|
* count. `fileCountDeficit` (baseline.ts) reads a summary LINE — a selected screen's smaller count
|
|
@@ -387,6 +396,7 @@ export function runManifestedTest(cmd, cwd, opts) {
|
|
|
387
396
|
* verdict measured with `pretest` hooks is never compared to one without. */
|
|
388
397
|
function manifestEnvironment(cwd) {
|
|
389
398
|
const env = { ...process.env, PATH: `${join(cwd, "node_modules/.bin")}:${process.env.PATH ?? ""}`,
|
|
399
|
+
[VITEST_CACHE_ENV]: worktreeVitestCache(cwd),
|
|
390
400
|
[FORK_CAP_ENV]: String(resolvedCapacity().forkCap), [SUITE_PARENT_ENV]: String(process.pid) };
|
|
391
401
|
const verification = verificationProtocol(env, cwd);
|
|
392
402
|
for (const key of ROUTING_ENV_SEAMS)
|
|
@@ -436,6 +446,9 @@ export async function manifestFileCount(cmd, cwd) {
|
|
|
436
446
|
catch {
|
|
437
447
|
return null;
|
|
438
448
|
}
|
|
449
|
+
finally {
|
|
450
|
+
rmSync(dir, { recursive: true, force: true });
|
|
451
|
+
} // OBS-1155: the listing directory is the capture's alone, listed or not
|
|
439
452
|
}
|
|
440
453
|
/** Vitest's forks pool awaits the parallel phase, then throws before the single-fork phase
|
|
441
454
|
* on any rejected worker. Recover only that exact, fully accounted-for boundary. */
|
|
@@ -474,6 +487,23 @@ function strandedSingleForkFiles(files, nonce, run) {
|
|
|
474
487
|
return;
|
|
475
488
|
return single;
|
|
476
489
|
}
|
|
490
|
+
/** The stranded single-fork retry: positional filters are substring matches that OR with any filter
|
|
491
|
+
* the command already carries, and under a `projects` config the CLI `--exclude` never subtracts such a
|
|
492
|
+
* selection (OBS-1166: a selected screen's retry rediscovered the whole selection and refused). So the
|
|
493
|
+
* retry is built from the UN-narrowed base command, its own `--` rule, the stranded files as the only
|
|
494
|
+
* positional filters, and an `--exclude` of every completed file; the caller then requires discovery
|
|
495
|
+
* to prove the exact retry set before launch. OBS-1180: Vitest matches an absolute filter against the
|
|
496
|
+
* module path it resolved through every symlink, so the filters are rooted at the canonical realpath of
|
|
497
|
+
* `cwd` — a symlinked worktree root otherwise filters to an empty discovery. */
|
|
498
|
+
export function singleForkRetryCommand(base, cwd, stranded, completed) {
|
|
499
|
+
let root = cwd;
|
|
500
|
+
try {
|
|
501
|
+
root = realpathSync(cwd);
|
|
502
|
+
}
|
|
503
|
+
catch { /* unreadable cwd — the exact rediscovery below fails closed */ }
|
|
504
|
+
const excluded = completed.map(f => `--exclude=${shq(f.replace(/[\\*?[\]{}()!+@]/g, "\\$&"))}`).join(" ");
|
|
505
|
+
return `${base}${runnerInvocation(base, cwd).separator} ${stranded.map(f => shq(join(root, f))).join(" ")} ${excluded}`;
|
|
506
|
+
}
|
|
477
507
|
/** One configured runner execution, and its own collection under the same arguments and environment.
|
|
478
508
|
* The installed runner is trusted (R28 add.1 option B); the nonce catches stale artifacts, not forgery. */
|
|
479
509
|
export async function evaluateManifestedTest(cmd, cwd, opts) {
|
|
@@ -518,10 +548,7 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
|
|
|
518
548
|
recovery = { firstNonce: nonce, firstReportPath: reportPath, retryNonce, files: stranded };
|
|
519
549
|
nonce = retryNonce;
|
|
520
550
|
reportPath = join(dir, `test-manifest-report-${nonce}.json`);
|
|
521
|
-
|
|
522
|
-
// completed file as well, then require discovery to prove the exact retry set before launch.
|
|
523
|
-
const excluded = files.filter(f => !stranded.includes(f)).map(f => `--exclude=${shq(f.replace(/[\\*?[\]{}()!+@]/g, "\\$&"))}`).join(" ");
|
|
524
|
-
const retryCommand = `${cmd}${invocation.separator} ${stranded.map(f => shq(join(cwd, f))).join(" ")} ${excluded}`;
|
|
551
|
+
const retryCommand = singleForkRetryCommand(opts.retryBaseCommand ?? cmd, cwd, stranded, files.filter(f => !stranded.includes(f)));
|
|
525
552
|
const listed = await discoverTestManifest(retryCommand, cwd, { dir, nonce, env,
|
|
526
553
|
overallCeilingMs: opts.overallCeilingMs, evidence: { ...opts.evidence, artifactDir: opts.evidence?.artifactDir ?? dir } });
|
|
527
554
|
evidenceReceipts.push(...listed.evidenceReceipts);
|
package/dist/graph/graph.d.ts
CHANGED
|
@@ -36,8 +36,12 @@ export declare function addEvidence(g: RunGraph, id: string, patch: {
|
|
|
36
36
|
gateResults?: unknown[];
|
|
37
37
|
}): RunGraph;
|
|
38
38
|
export declare function chainDepth(g: RunGraph): Map<string, number>;
|
|
39
|
-
export declare function readyTasks(g: RunGraph): Task[];
|
|
40
|
-
export declare function
|
|
39
|
+
export declare function readyTasks(g: RunGraph, prioritized?: ReadonlySet<string>): Task[];
|
|
40
|
+
export declare function batteryPriority(actions: Iterable<{
|
|
41
|
+
taskId: string;
|
|
42
|
+
authority: string;
|
|
43
|
+
}>): Set<string>;
|
|
44
|
+
export declare function dispatchWaves(g: RunGraph, concurrency: number, prioritized?: ReadonlySet<string>): Map<string, number>;
|
|
41
45
|
export declare function isComplete(g: RunGraph): boolean;
|
|
42
46
|
export declare function isStalled(g: RunGraph): boolean;
|
|
43
47
|
export declare function closureReaches(g: RunGraph, taskId: string, pred: (t: Task) => boolean): boolean;
|
package/dist/graph/graph.js
CHANGED
|
@@ -191,21 +191,32 @@ export function chainDepth(g) {
|
|
|
191
191
|
// OBS-1018: admission is critical-path order — deepest chain root first, ties in declaration order.
|
|
192
192
|
// The daemon's dispatch loop slices this list unchanged. Stated limit: resume-restore of previously
|
|
193
193
|
// in-flight attempts keeps its own precedence (src/run/daemon.ts); only fresh admission is ordered here.
|
|
194
|
-
|
|
194
|
+
// OBS-1158: `prioritized` names tasks holding a pending battery (recheck) approval. Among READY tasks
|
|
195
|
+
// of equal depth an approved recheck precedes fresh work; deeper fresh work keeps its rank, and a
|
|
196
|
+
// recheck that is not ready (dependency-blocked) is not admitted at all, so it gains nothing here.
|
|
197
|
+
export function readyTasks(g, prioritized = NO_PRIORITY) {
|
|
195
198
|
const done = new Set(g.tasks.filter((t) => t.status === "done").map((t) => t.id));
|
|
196
199
|
const depth = chainDepth(g);
|
|
200
|
+
const rank = (t) => (prioritized.has(t.id) ? 1 : 0);
|
|
197
201
|
return g.tasks
|
|
198
202
|
.filter((t) => t.status === "pending" && t.deps.every((d) => done.has(d)))
|
|
199
|
-
.sort((a, b) => depth.get(b.id) - depth.get(a.id)); // Array#sort is stable:
|
|
203
|
+
.sort((a, b) => depth.get(b.id) - depth.get(a.id) || rank(b) - rank(a)); // Array#sort is stable: remaining ties keep declaration order
|
|
204
|
+
}
|
|
205
|
+
const NO_PRIORITY = new Set();
|
|
206
|
+
// OBS-1158: the one seam plan and the daemon share for battery priority — journal approval actions
|
|
207
|
+
// in (src/run/journal.ts pendingApprovalActions), the ids whose pending authority is `battery` out.
|
|
208
|
+
// Waivers, worker-funding approvals and inert releases never qualify; a consumed approval is absent.
|
|
209
|
+
export function batteryPriority(actions) {
|
|
210
|
+
return new Set([...actions].filter((a) => a.authority === "battery").map((a) => a.taskId));
|
|
200
211
|
}
|
|
201
212
|
// OBS-1018: the dispatch wave each pending task would enter at `concurrency` slots if every wave
|
|
202
213
|
// took one tick — computed by draining readyTasks, so plan and daemon can never disagree on order.
|
|
203
214
|
// Tasks already past pending carry no wave.
|
|
204
|
-
export function dispatchWaves(g, concurrency) {
|
|
215
|
+
export function dispatchWaves(g, concurrency, prioritized = NO_PRIORITY) {
|
|
205
216
|
const waves = new Map();
|
|
206
217
|
let sim = g;
|
|
207
218
|
for (let wave = 1;; wave++) {
|
|
208
|
-
const batch = readyTasks(sim).slice(0, Math.max(1, concurrency));
|
|
219
|
+
const batch = readyTasks(sim, prioritized).slice(0, Math.max(1, concurrency));
|
|
209
220
|
if (!batch.length)
|
|
210
221
|
return waves;
|
|
211
222
|
for (const t of batch) {
|