tickmarkr 2.5.0 → 2.5.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/fake.d.ts +1 -0
- package/dist/adapters/fake.js +2 -1
- package/dist/adapters/opencode.d.ts +1 -0
- package/dist/adapters/opencode.js +4 -1
- package/dist/adapters/types.d.ts +14 -0
- package/dist/adapters/types.js +25 -0
- package/dist/cli/commands/approve.js +1 -1
- package/dist/cli/commands/doctor.d.ts +2 -0
- package/dist/cli/commands/init.d.ts +2 -0
- package/dist/cli/commands/init.js +46 -6
- package/dist/cli/commands/plan.js +58 -6
- package/dist/cli/commands/report.js +3 -1
- package/dist/cli/commands/run.d.ts +5 -0
- package/dist/cli/commands/run.js +17 -1
- package/dist/compile/native.js +3 -0
- package/dist/config/config.d.ts +6 -0
- package/dist/config/config.js +6 -0
- package/dist/drivers/subprocess.d.ts +2 -3
- package/dist/drivers/subprocess.js +2 -3
- package/dist/gates/baseline.d.ts +9 -0
- package/dist/gates/baseline.js +85 -29
- package/dist/gates/llm.d.ts +5 -0
- package/dist/gates/llm.js +188 -3
- package/dist/gates/review.d.ts +6 -3
- package/dist/gates/review.js +95 -42
- package/dist/gates/run-gates.d.ts +3 -0
- package/dist/gates/run-gates.js +14 -4
- package/dist/run/daemon.d.ts +8 -4
- package/dist/run/daemon.js +3150 -2839
- package/dist/run/git.d.ts +4 -2
- package/dist/run/git.js +11 -23
- package/dist/run/merge.d.ts +1 -1
- package/dist/run/merge.js +126 -18
- package/dist/tui/ink/init-app.d.ts +4 -0
- package/dist/tui/ink/init-app.js +18 -7
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +16 -2
package/dist/gates/review.js
CHANGED
|
@@ -6,6 +6,7 @@ import { filesGlob } from "../graph/files-glob.js";
|
|
|
6
6
|
import { renderAcceptanceItem } from "../graph/schema.js";
|
|
7
7
|
import { getAdapter } from "../adapters/registry.js";
|
|
8
8
|
import { shOk } from "../run/git.js";
|
|
9
|
+
import { structuredFindings } from "../run/journal.js";
|
|
9
10
|
import { redactSecrets } from "../run/redact.js";
|
|
10
11
|
import { marginalCostRank } from "../route/router.js";
|
|
11
12
|
import { modelProvider } from "../route/preference.js";
|
|
@@ -178,7 +179,7 @@ export function pickReviewer(author, channels, exclude = [], // v1.1 failover: r
|
|
|
178
179
|
prefer = [], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
|
|
179
180
|
floor, // task-declared only; config floors govern workers and must not silently move review seats
|
|
180
181
|
history = [], // run-scoped picks, oldest to newest; empty preserves the established ranking
|
|
181
|
-
onSeat) {
|
|
182
|
+
onSeat, demoted = new Set()) {
|
|
182
183
|
// FLEET-05 success criterion 2: an author not resolvable in the channel list yields NO reviewer.
|
|
183
184
|
// The old `?? author.adapter` fallback compared an adapter id to vendor names, matched nothing, and
|
|
184
185
|
// admitted every reviewer — including the author's own channel (fail-OPEN). null lands on reviewGate's
|
|
@@ -188,20 +189,21 @@ onSeat) {
|
|
|
188
189
|
return null;
|
|
189
190
|
const authorProvider = modelProvider(author.model, authorChannel.vendor);
|
|
190
191
|
const ranked = channels
|
|
191
|
-
//
|
|
192
|
-
//
|
|
193
|
-
//
|
|
192
|
+
// Three independent axes: different vendor, different resolved provider identity (OBS-946: on initial pick
|
|
193
|
+
// as well as failover, so an aggregator channel stamped "mixed" never seats the author's own provider),
|
|
194
|
+
// and different base-model identity (ADDED TO the vendor rule, never replacing it). The diversity
|
|
194
195
|
// filter runs BEFORE preference ranking, so prefer cannot resurrect an excluded channel.
|
|
195
196
|
.filter((c) => c.vendor !== authorChannel.vendor
|
|
196
|
-
&&
|
|
197
|
+
&& modelProvider(c.model, c.vendor) !== authorProvider
|
|
197
198
|
&& modelId(c.model) !== modelId(author.model)
|
|
198
199
|
&& !exclude.includes(channelKey(c))
|
|
199
200
|
&& (floor === undefined || TIER_RANK[c.tier] >= TIER_RANK[floor]))
|
|
200
201
|
.sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b));
|
|
201
|
-
const reviewer = [...ranked].sort((a, b) =>
|
|
202
|
+
const reviewer = [...ranked].sort((a, b) => Number(demoted.has(channelKey(a))) - Number(demoted.has(channelKey(b)))
|
|
203
|
+
|| history.lastIndexOf(channelKey(a)) - history.lastIndexOf(channelKey(b))
|
|
202
204
|
|| ranked.indexOf(a) - ranked.indexOf(b))[0] ?? null;
|
|
203
205
|
if (reviewer)
|
|
204
|
-
onSeat?.(ranked.indexOf(reviewer) + 1);
|
|
206
|
+
onSeat?.(ranked.indexOf(reviewer) + 1, ranked.length);
|
|
205
207
|
return reviewer;
|
|
206
208
|
}
|
|
207
209
|
/**
|
|
@@ -220,7 +222,7 @@ ${files.map((path) => `- ${path}`).join("\n")}`;
|
|
|
220
222
|
export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
|
|
221
223
|
// OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
|
|
222
224
|
// direct tests) skips persistence and changes nothing else.
|
|
223
|
-
artifactDir, reviewHistory) {
|
|
225
|
+
artifactDir, reviewHistory, demotedReviewers, carriedFindings = []) {
|
|
224
226
|
// R3 (OBS-186): participation is keyed on PATHS. The compiler's assignment comes from the DECLARED
|
|
225
227
|
// files[]; the operator's floor may RAISE it to full and can never lower it. `complexityThreshold` is
|
|
226
228
|
// retired — the branch that returned a green skip on a complexity comparison is gone, and with it the
|
|
@@ -241,8 +243,9 @@ artifactDir, reviewHistory) {
|
|
|
241
243
|
// so production rounds have journaled both siblings all along — only the fixtures were blind to it.
|
|
242
244
|
// Fixed in the ledger rather than in the oracles, because determinism run-to-run is a property of
|
|
243
245
|
// the journal, not of three test files that happen to assert it.
|
|
246
|
+
const priorMaterials = carriedFindings.filter((finding) => finding.class === "review:material");
|
|
244
247
|
const declaredPolicy = declaredReviewPolicy(task.files);
|
|
245
|
-
const policy = raiseReviewPolicy(declaredPolicy, cfg.review.policy);
|
|
248
|
+
const policy = priorMaterials.length ? "full" : raiseReviewPolicy(declaredPolicy, cfg.review.policy);
|
|
246
249
|
// PROMOTION: the declared assignment is a claim about paths, and the diff is the evidence. A
|
|
247
250
|
// judge-only task whose diff left the leaf class is reviewed in full — the claim never outranks
|
|
248
251
|
// what actually happened, and an empty diff promotes too (a skip earned by an absence is not earned).
|
|
@@ -293,15 +296,15 @@ artifactDir, reviewHistory) {
|
|
|
293
296
|
// historical seat for every task that never asked for review-tier coupling.
|
|
294
297
|
const reviewerFloor = task.routingHints?.floor;
|
|
295
298
|
let rotationSeat;
|
|
296
|
-
const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor, reviewHistory, reviewHistory ? (seat) => { rotationSeat = seat; } : undefined);
|
|
299
|
+
const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor, reviewHistory, reviewHistory ? (seat) => { rotationSeat = seat; } : undefined, demotedReviewers);
|
|
297
300
|
if (!reviewer) {
|
|
298
301
|
// meta.noEligibleReviewer lets run-gates' review-retry keep the ORIGINAL unparseable result when
|
|
299
302
|
// the retry finds no second seat — a truthful cause beats a synthetic no-reviewer failure.
|
|
300
303
|
const reason = reviewerFloor
|
|
301
304
|
? `no cross-vendor reviewer available at or above task-declared ${reviewerFloor} floor (diversity rule)`
|
|
302
305
|
: "no cross-vendor reviewer available (diversity rule)";
|
|
303
|
-
return cfg.review.required
|
|
304
|
-
? { gate: "review", pass: false, details: `unreadable — ${reason}; set review.required:false to waive`, meta: { noEligibleReviewer: true, unreadable: true, ...(reviewerFloor ? { reviewerFloor } : {}) } }
|
|
306
|
+
return cfg.review.required || priorMaterials.length > 0
|
|
307
|
+
? { gate: "review", pass: false, details: `unreadable — ${reason}; ${priorMaterials.length ? "carried materials require a review verdict" : "set review.required:false to waive"}`, meta: { noEligibleReviewer: true, unreadable: true, ...(reviewerFloor ? { reviewerFloor } : {}) } }
|
|
305
308
|
: { gate: "review", pass: true, details: `WARNING: ${reason} — review waived by config`, meta: { noEligibleReviewer: true, ...(reviewerFloor ? { reviewerFloor } : {}) } };
|
|
306
309
|
}
|
|
307
310
|
reviewHistory?.push(channelKey(reviewer));
|
|
@@ -327,7 +330,10 @@ ${task.acceptance.map((a) => `- ${renderAcceptanceItem(a)}`).join("\n")}
|
|
|
327
330
|
|
|
328
331
|
${renderDeclaredWriteScope(task.files)}
|
|
329
332
|
|
|
330
|
-
|
|
333
|
+
${priorMaterials.length ? `## Prior materials this attempt must close
|
|
334
|
+
${priorMaterials.map((finding) => `Fingerprint: ${finding.fingerprint}\n${finding.note}`).join("\n\n")}
|
|
335
|
+
|
|
336
|
+
` : ""}## Diff
|
|
331
337
|
\`\`\`diff
|
|
332
338
|
${diff}
|
|
333
339
|
\`\`\`
|
|
@@ -340,18 +346,36 @@ block approval. For a minor concern you have decided not to block on, set "defer
|
|
|
340
346
|
one-line "rationale" — it is recorded in the review, never dropped.
|
|
341
347
|
|
|
342
348
|
Respond with ONLY this JSON:
|
|
343
|
-
{"nonce": "${nonce}", "approve": true|false, "findings": [{"note": "...", "severity": "material"|"minor", "defer": false, "rationale": ""}], "comments": [{"path": "path/to/file", "line": 42, "body": "actionable feedback"}]}
|
|
344
|
-
|
|
349
|
+
{"nonce": "${nonce}", "approve": true|false, "resolved": [], "reraised": [], "findings": [{"note": "...", "severity": "material"|"minor", "defer": false, "rationale": ""}], "comments": [{"path": "path/to/file", "line": 42, "body": "actionable feedback"}]}
|
|
350
|
+
For every prior material, put its fingerprint in exactly one of resolved (verified fixed) or reraised
|
|
351
|
+
(still a blocking defect). Use only the listed fingerprints; never omit one or put it in both lists.
|
|
352
|
+
Approve iff no material finding remains and every prior material is resolved.
|
|
345
353
|
The top-level comments array is optional. Use it only for actionable line-anchored feedback.
|
|
346
354
|
`;
|
|
347
|
-
|
|
355
|
+
// Filenames are journaled (daemon.ts lifts meta.rawPath/briefPath onto the gate-result row), so they
|
|
356
|
+
// must be reproducible from the same inputs — the verdict nonce is cryptographically random and would
|
|
357
|
+
// make two otherwise-identical runs diverge in their journal bytes. The reviewer channel already
|
|
358
|
+
// disambiguates every call that matters: a retry always excludes the flaked channel (run-gates.ts),
|
|
359
|
+
// so it can never collide with the attempt it replaces.
|
|
360
|
+
const artifactId = `${task.id}-${channelKey(reviewer).replace(/[^a-zA-Z0-9_.-]/g, "-")}`;
|
|
361
|
+
const briefPath = artifactDir ? join(artifactDir, `review-brief-${artifactId}.md`) : undefined;
|
|
362
|
+
// Persistence is evidence, not a gate input: a full disk or a removed run dir never fails the gate.
|
|
363
|
+
let savedBrief;
|
|
364
|
+
if (briefPath) {
|
|
365
|
+
try {
|
|
366
|
+
writeFileSync(briefPath, redactSecrets(prompt));
|
|
367
|
+
savedBrief = briefPath;
|
|
368
|
+
}
|
|
369
|
+
catch {
|
|
370
|
+
savedBrief = undefined;
|
|
371
|
+
}
|
|
372
|
+
}
|
|
348
373
|
const llm = await runLlmDetailed(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? {
|
|
349
374
|
driver: via.driver,
|
|
350
375
|
keep: via.keep,
|
|
351
376
|
onSlot: via.onSlot,
|
|
352
377
|
name: via.nameFor("review", reviewer.adapter),
|
|
353
378
|
label: via.labelFor("review"),
|
|
354
|
-
onInactivity: () => { concludedOnInactivity = true; },
|
|
355
379
|
} : undefined,
|
|
356
380
|
// frontier reviewers routinely need >5min on a configured-cap-sized diff, and `claude -p` buffers all
|
|
357
381
|
// output until completion — runLlm's 300s default killed reviews mid-flight, returning empty
|
|
@@ -359,56 +383,85 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
359
383
|
// (run-20260709-104447 P87-09). The configured ceiling defaults to that measured 15 minutes.
|
|
360
384
|
cfg.review.timeoutMs);
|
|
361
385
|
const raw = llm.output;
|
|
386
|
+
let saved;
|
|
387
|
+
if (artifactDir) {
|
|
388
|
+
try {
|
|
389
|
+
saved = join(artifactDir, `review-raw-${artifactId}.txt`);
|
|
390
|
+
writeFileSync(saved, redactSecrets(raw));
|
|
391
|
+
}
|
|
392
|
+
catch {
|
|
393
|
+
saved = undefined; // persistence is evidence, not a gate input — never fail the gate on it
|
|
394
|
+
}
|
|
395
|
+
}
|
|
362
396
|
const provider = modelProvider(reviewer.model, reviewer.vendor);
|
|
363
397
|
const v = extractVerdictJson(raw, nonce);
|
|
364
398
|
const findings = v && Array.isArray(v.findings) ? v.findings : null;
|
|
399
|
+
const priorIds = new Set(priorMaterials.map((finding) => finding.fingerprint));
|
|
400
|
+
const closureLists = [v?.resolved, v?.reraised];
|
|
401
|
+
const closureInvalid = !!v && (priorIds.size > 0 || closureLists.some((list) => list !== undefined)) && (closureLists.some((list) => !Array.isArray(list) || list.some((id) => typeof id !== "string" || !priorIds.has(id)))
|
|
402
|
+
|| new Set([...(v?.resolved ?? []), ...(v?.reraised ?? [])]).size !== (v?.resolved?.length ?? 0) + (v?.reraised?.length ?? 0)
|
|
403
|
+
|| [...priorIds].some((id) => !v?.resolved?.includes(id) && !v?.reraised?.includes(id)));
|
|
365
404
|
// findings decides the verdict on its own; the legacy path still needs approve + issues to parse.
|
|
366
|
-
if (!v || (findings === null && (typeof v.approve !== "boolean" || !Array.isArray(v.issues)))) {
|
|
405
|
+
if (!v || closureInvalid || (findings === null && (typeof v.approve !== "boolean" || !Array.isArray(v.issues)))) {
|
|
367
406
|
// OBS-196: name the cause and persist the raw bytes — a ruled-on "unparseable" without its
|
|
368
407
|
// evidence cannot be audited, and a cutoff must never be indistinguishable from a parse defect.
|
|
369
|
-
const
|
|
370
|
-
const
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
}
|
|
377
|
-
catch {
|
|
378
|
-
saved = undefined; // persistence is evidence, not a gate input — never fail the gate on it
|
|
379
|
-
}
|
|
380
|
-
}
|
|
381
|
-
const failure = concludedOnInactivity
|
|
382
|
-
? "review dispatch concluded on the inactivity policy without a structurally valid nonce-bound response; output unparseable"
|
|
383
|
-
: cause === "malformed-verdict"
|
|
384
|
-
? "review output unparseable"
|
|
385
|
-
: "review dispatch failed — no structurally valid nonce-bound response; output unparseable";
|
|
408
|
+
const bytes = llm.seatAuthoredBytes ?? Buffer.byteLength(raw.trim(), "utf8");
|
|
409
|
+
const cause = closureInvalid ? "malformed-verdict" : llm.launchNeverStarted ? "launch-never-started"
|
|
410
|
+
: llm.timedOut ? (bytes > 0 ? "truncated" : "silent")
|
|
411
|
+
: classifyVerdictCause(raw, nonce, "approve", llm);
|
|
412
|
+
const failure = cause === "malformed-verdict"
|
|
413
|
+
? "review output unparseable"
|
|
414
|
+
: "review dispatch failed — no structurally valid nonce-bound response";
|
|
386
415
|
return {
|
|
387
416
|
gate: "review",
|
|
388
417
|
pass: false,
|
|
389
|
-
details: `${failure} (reviewer ${reviewer.adapter}:${reviewer.model}; vendor: ${reviewer.vendor}; provider: ${provider}; cause: ${cause}${
|
|
418
|
+
details: `${failure} (reviewer ${reviewer.adapter}:${reviewer.model}; vendor: ${reviewer.vendor}; provider: ${provider}; cause: ${cause}${llm.timedOut ? `; killed at configured review timeout ${cfg.review.timeoutMs}ms` : ""}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
|
|
390
419
|
meta: {
|
|
391
420
|
...policyMeta,
|
|
392
421
|
...rotationMeta,
|
|
393
422
|
reviewer: channelKey(reviewer),
|
|
394
423
|
vendor: reviewer.vendor,
|
|
395
424
|
provider,
|
|
396
|
-
unparseable: true,
|
|
425
|
+
...(cause === "malformed-verdict" ? { unparseable: true } : { noVerdict: true, classification: "infra", infra: true }),
|
|
397
426
|
cause,
|
|
398
|
-
|
|
399
|
-
...(
|
|
400
|
-
...(
|
|
427
|
+
bytes, seatAuthoredBytes: bytes,
|
|
428
|
+
...(saved ? { rawPath: saved } : {}),
|
|
429
|
+
...(savedBrief ? { briefPath: savedBrief } : {}),
|
|
430
|
+
...(llm.timedOut ? { timeoutMs: cfg.review.timeoutMs } : {}),
|
|
401
431
|
},
|
|
402
432
|
};
|
|
403
433
|
}
|
|
404
434
|
const decided = findings !== null
|
|
405
435
|
? classifyReviewFindings(findings)
|
|
406
436
|
: classifyReviewIssues(v.approve, v.issues);
|
|
437
|
+
const reraised = priorMaterials.filter((finding) => v.reraised?.includes(finding.fingerprint));
|
|
438
|
+
if (reraised.length) {
|
|
439
|
+
if (decided.pass)
|
|
440
|
+
decided.headline = "requested changes";
|
|
441
|
+
decided.pass = false;
|
|
442
|
+
// A reviewer may also restate a re-raised material in findings. Preserve the original
|
|
443
|
+
// prose once so an unchanged defect keeps the same failure brief across repair rounds.
|
|
444
|
+
for (const finding of reraised) {
|
|
445
|
+
const line = `- [material] ${finding.note}`;
|
|
446
|
+
if (!decided.lines.includes(line))
|
|
447
|
+
decided.lines.push(line);
|
|
448
|
+
}
|
|
449
|
+
}
|
|
407
450
|
const prose = `reviewer ${reviewer.adapter}:${reviewer.model} (vendor: ${reviewer.vendor}; provider: ${provider}): ${decided.headline}${decided.lines.length ? "\n" + decided.lines.join("\n") : ""}`;
|
|
451
|
+
const details = appendAnchoredReview(prose, v);
|
|
408
452
|
return {
|
|
409
453
|
gate: "review",
|
|
410
454
|
pass: decided.pass,
|
|
411
|
-
details
|
|
412
|
-
meta: {
|
|
455
|
+
details,
|
|
456
|
+
meta: {
|
|
457
|
+
...policyMeta, ...rotationMeta, reviewer: channelKey(reviewer), vendor: reviewer.vendor, provider,
|
|
458
|
+
...(priorMaterials.length ? { resolved: v.resolved, reraised: v.reraised } : {}),
|
|
459
|
+
...(reraised.length ? { findings: [
|
|
460
|
+
...structuredFindings("review", details).filter((finding) => !reraised.some((prior) => prior.note === finding.note)),
|
|
461
|
+
...reraised,
|
|
462
|
+
] } : {}),
|
|
463
|
+
...(saved ? { rawPath: saved } : {}),
|
|
464
|
+
...(savedBrief ? { briefPath: savedBrief } : {}),
|
|
465
|
+
},
|
|
413
466
|
};
|
|
414
467
|
}
|
|
@@ -4,6 +4,7 @@ import { type GateName, type Task } from "../graph/schema.js";
|
|
|
4
4
|
import { type Baseline } from "./baseline.js";
|
|
5
5
|
import { type GateVia } from "./llm.js";
|
|
6
6
|
import type { GateResult } from "./types.js";
|
|
7
|
+
import { type StructuredFinding } from "../run/journal.js";
|
|
7
8
|
export type LoadProvider = () => number;
|
|
8
9
|
/** Test seam — inject deterministic load samples; production always reads os.loadavg. */
|
|
9
10
|
export declare function setLoadProviderForTests(provider: LoadProvider): void;
|
|
@@ -52,7 +53,9 @@ export interface GateContext {
|
|
|
52
53
|
adapters: WorkerAdapter[];
|
|
53
54
|
cfg: TickmarkrConfig;
|
|
54
55
|
via?: GateVia;
|
|
56
|
+
carriedFindings?: readonly StructuredFinding[];
|
|
55
57
|
excludeReviewers?: string[];
|
|
58
|
+
demotedReviewers?: Set<string>;
|
|
56
59
|
reviewHistory?: string[];
|
|
57
60
|
artifactDir?: string;
|
|
58
61
|
pipeline?: "v185" | "legacy";
|
package/dist/gates/run-gates.js
CHANGED
|
@@ -590,13 +590,23 @@ export async function runGates(task, ctx) {
|
|
|
590
590
|
const dispatch = async (run) => {
|
|
591
591
|
const captured = await captureLlmDispatches(ctx.adapters, run);
|
|
592
592
|
invocations.push(...captured.invocations);
|
|
593
|
-
|
|
593
|
+
const rv = captured.value;
|
|
594
|
+
if (rv.meta?.noVerdict === true || rv.meta?.unparseable === true) {
|
|
595
|
+
await ctx.onGate?.({ phase: "note", gate: "review", name: "review-no-verdict", payload: { ...rv.meta }, result: rv });
|
|
596
|
+
if (rv.meta.seatAuthoredBytes === 0 && typeof rv.meta.reviewer === "string"
|
|
597
|
+
&& !ctx.demotedReviewers?.has(rv.meta.reviewer)) {
|
|
598
|
+
ctx.demotedReviewers?.add(rv.meta.reviewer);
|
|
599
|
+
await ctx.onGate?.({ phase: "note", gate: "review", name: "review-pool-demotion",
|
|
600
|
+
payload: { reviewer: rv.meta.reviewer, cause: rv.meta.cause, seatAuthoredBytes: 0 }, result: rv });
|
|
601
|
+
}
|
|
602
|
+
}
|
|
603
|
+
return rv;
|
|
594
604
|
};
|
|
595
|
-
let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir, ctx.reviewHistory));
|
|
605
|
+
let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings));
|
|
596
606
|
// OBS-193/574: an unparseable review verdict retries the REVIEW exactly once, preferring a
|
|
597
607
|
// different adapter. Only a single-adapter eligible pool may fall back to another channel on the
|
|
598
608
|
// flaked adapter. The flaked verdict never enters results; an exhausted pool preserves its cause.
|
|
599
|
-
if (rv.meta?.unparseable === true && typeof rv.meta.reviewer === "string") {
|
|
609
|
+
if ((rv.meta?.unparseable === true || rv.meta?.noVerdict === true) && typeof rv.meta.reviewer === "string") {
|
|
600
610
|
const flaked = rv.meta.reviewer;
|
|
601
611
|
const emptyOutput = rv.meta.cause === "empty-output";
|
|
602
612
|
if (emptyOutput) {
|
|
@@ -615,7 +625,7 @@ export async function runGates(task, ctx) {
|
|
|
615
625
|
const crossAdapter = pickReviewer(ctx.author, ctx.channels, [...priorExclusions, ...adapterExclusions], ctx.cfg.review.prefer ?? [], task.routingHints?.floor);
|
|
616
626
|
const exclusion = crossAdapter ? "adapter" : "channel";
|
|
617
627
|
const retryExclusions = [...priorExclusions, ...(crossAdapter ? adapterExclusions : [flaked])];
|
|
618
|
-
const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, retryExclusions, ctx.artifactDir, ctx.reviewHistory));
|
|
628
|
+
const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, retryExclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings));
|
|
619
629
|
if (second.meta?.noEligibleReviewer !== true) {
|
|
620
630
|
const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
|
|
621
631
|
const route = exclusion === "adapter"
|
package/dist/run/daemon.d.ts
CHANGED
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
import { type WorkerAdapter } from "../adapters/types.js";
|
|
2
2
|
import { type ModeResolution, type RoutingMode, type TickmarkrConfig } from "../config/config.js";
|
|
3
3
|
import { type DriverChoice } from "../drivers/index.js";
|
|
4
|
-
import { type ExecutorDriver } from "../drivers/types.js";
|
|
4
|
+
import { type ExecutorDriver, type Slot } from "../drivers/types.js";
|
|
5
5
|
import { type Baseline } from "../gates/baseline.js";
|
|
6
6
|
import type { GateResult } from "../gates/types.js";
|
|
7
7
|
import { Journal, type JournalEvent } from "./journal.js";
|
|
8
|
+
export declare function closeLiveSlot(liveSlots: Set<Slot>, driver: Pick<ExecutorDriver, "close">, slot: Slot): Promise<void>;
|
|
8
9
|
export { harvestCpuFlatWindowMs, resetHarvestCpuFlatMsForTests, setHarvestCpuFlatMsForTests, workerTreeCpuMs, } from "./stall.js";
|
|
9
10
|
export interface RunOptions {
|
|
10
11
|
runId?: string;
|
|
@@ -13,6 +14,8 @@ export interface RunOptions {
|
|
|
13
14
|
graphChanged?: boolean;
|
|
14
15
|
retryFailed?: boolean;
|
|
15
16
|
concurrency?: number;
|
|
17
|
+
/** Bounded wait at a drain caused solely by parked tasks. */
|
|
18
|
+
approvalWindowMs?: number;
|
|
16
19
|
driver?: ExecutorDriver;
|
|
17
20
|
driverOverride?: DriverChoice;
|
|
18
21
|
adapters?: WorkerAdapter[];
|
|
@@ -54,9 +57,8 @@ export interface RunSummary {
|
|
|
54
57
|
* T14, amended by v2.2 T3: approvals the run accepted and never acted on. `approved` above is still
|
|
55
58
|
* built ONCE at startup — replay determinism depends on it — but a live approval is no longer inert:
|
|
56
59
|
* the boundary sweep in the task loop releases what lands while the daemon runs, so an approval
|
|
57
|
-
* written mid-run is
|
|
58
|
-
* fold still
|
|
59
|
-
* run-end sample below — meets no further boundary, so nothing can release it before this run ends.
|
|
60
|
+
* written mid-run is enacted at a boundary, during the approval window, or by cancelling tip verify.
|
|
61
|
+
* This fold still exposes decisions that could not enact, including a failure before dispatch.
|
|
60
62
|
* Without this the run-end record stated only buckets and tipVerify, both accurate, over a milestone
|
|
61
63
|
* that was silently incomplete: run …230 ended tipVerify "passed" with two upheld approvals and zero
|
|
62
64
|
* subsequent dispatches. Scored per task on its NEWEST approval: a later approval is the live
|
|
@@ -107,6 +109,7 @@ export declare const SUITE_WAIT_CEILING_MS = 600000;
|
|
|
107
109
|
export declare const setSuiteWaitCeilingForTests: (ms: number) => void;
|
|
108
110
|
export declare const resetSuiteWaitCeilingForTests: () => void;
|
|
109
111
|
export declare const APPROVAL_POLL_MS = 250;
|
|
112
|
+
export declare const APPROVAL_WINDOW_MS = 1000;
|
|
110
113
|
export declare const EARLY_LAUNCH_LIVENESS_MS = 60000;
|
|
111
114
|
/** Test seam — lowers the empty-pane liveness window without sleeping 60s per case. */
|
|
112
115
|
export declare function setEarlyLaunchLivenessMsForTests(ms: number): void;
|
|
@@ -150,6 +153,7 @@ export declare function commandsHash(commands: Record<string, string>): string;
|
|
|
150
153
|
export declare function verifyIntegrationTipCached(intWt: string, commands: Record<string, string>, journal: Journal, opts?: {
|
|
151
154
|
lastMergedTask?: string;
|
|
152
155
|
baseline?: Baseline;
|
|
156
|
+
signal?: AbortSignal;
|
|
153
157
|
}): Promise<boolean>;
|
|
154
158
|
type SuitePidProbe = (pid: number) => number | undefined;
|
|
155
159
|
/** Count full-suite roots in one process-table snapshot. The probes are arguments so the ownership
|