tickmarkr 2.5.5 → 2.5.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -1
- package/dist/adapters/prompt.js +21 -1
- package/dist/cli/commands/approve.js +69 -8
- package/dist/cli/commands/doctor.d.ts +10 -0
- package/dist/cli/commands/fleet.js +4 -0
- package/dist/cli/commands/plan.js +20 -4
- package/dist/cli/commands/status.js +95 -34
- package/dist/cli/commands/verify.js +108 -85
- package/dist/config/config.d.ts +11 -0
- package/dist/config/config.js +21 -12
- package/dist/config/fleet-overlay.js +55 -17
- package/dist/drivers/index.js +2 -1
- package/dist/drivers/orca.d.ts +21 -1
- package/dist/drivers/orca.js +209 -27
- package/dist/gates/baseline.d.ts +21 -5
- package/dist/gates/baseline.js +67 -17
- package/dist/gates/cache.d.ts +14 -13
- package/dist/gates/cache.js +17 -5
- package/dist/gates/llm.d.ts +3 -0
- package/dist/gates/llm.js +11 -0
- package/dist/gates/review.d.ts +28 -3
- package/dist/gates/review.js +118 -15
- package/dist/gates/run-gates.d.ts +10 -2
- package/dist/gates/run-gates.js +362 -108
- package/dist/gates/test-manifest.d.ts +33 -1
- package/dist/gates/test-manifest.js +132 -40
- package/dist/gates/test-reporter.js +20 -7
- package/dist/graph/graph.d.ts +4 -0
- package/dist/graph/graph.js +50 -1
- package/dist/run/activity.d.ts +28 -0
- package/dist/run/activity.js +194 -0
- package/dist/run/consult.js +5 -4
- package/dist/run/daemon.d.ts +11 -0
- package/dist/run/daemon.js +663 -128
- package/dist/run/execution-budget.d.ts +25 -0
- package/dist/run/execution-budget.js +142 -0
- package/dist/run/git.d.ts +46 -1
- package/dist/run/git.js +149 -12
- package/dist/run/journal.d.ts +23 -5
- package/dist/run/journal.js +98 -25
- package/dist/run/lease.d.ts +44 -0
- package/dist/run/lease.js +226 -3
- package/dist/run/operator-page-summary.d.ts +56 -0
- package/dist/run/operator-page-summary.js +68 -0
- package/dist/run/operator-summary.d.ts +69 -0
- package/dist/run/operator-summary.js +77 -0
- package/dist/run/protocol.d.ts +71 -0
- package/dist/run/protocol.js +32 -0
- package/dist/run/recovery.d.ts +8 -0
- package/dist/run/recovery.js +25 -0
- package/dist/run/repair-selection.d.ts +12 -0
- package/dist/run/repair-selection.js +56 -0
- package/dist/run/stall.d.ts +6 -1
- package/dist/run/stall.js +60 -3
- package/dist/tui/cockpit/board.d.ts +9 -0
- package/dist/tui/cockpit/board.js +10 -0
- package/dist/tui/cockpit/derive.d.ts +35 -0
- package/dist/tui/cockpit/derive.js +152 -10
- package/dist/tui/cockpit/evidence-view.d.ts +2 -0
- package/dist/tui/cockpit/evidence-view.js +42 -12
- package/dist/tui/cockpit/run-cockpit.d.ts +8 -1
- package/dist/tui/cockpit/run-cockpit.js +95 -1
- package/dist/tui/cockpit/run-view.d.ts +37 -0
- package/dist/tui/cockpit/run-view.js +189 -2
- package/dist/tui/ink/fleet-app.d.ts +10 -2
- package/dist/tui/ink/fleet-app.js +33 -15
- package/package.json +2 -2
- package/skills/tickmarkr-overseer/SKILL.md +55 -3
- package/skills/tickmarkr-overseer/scripts/grade-ci.sh +29 -2
package/dist/gates/cache.d.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import type { Baseline } from "./baseline.js";
|
|
2
2
|
import type { GateResult } from "./types.js";
|
|
3
|
-
import { type RunCapacity } from "../run/git.js";
|
|
3
|
+
import { type RunCapacity, type VerificationProtocol } from "../run/git.js";
|
|
4
4
|
export declare const DEFAULT_VERDICT_CACHE_BOUND = 128;
|
|
5
5
|
export declare function setVerdictCacheBoundForTests(bound: number | undefined): void;
|
|
6
6
|
export declare function resetVerdictCacheBoundForTests(): void;
|
|
@@ -16,15 +16,19 @@ export interface GateEnvironmentInput {
|
|
|
16
16
|
capacity?: RunCapacity;
|
|
17
17
|
selectedSet?: readonly string[];
|
|
18
18
|
scope?: VerificationScope;
|
|
19
|
+
/** R41: the verification protocol + runner lifecycle policy; defaults to this process's. */
|
|
20
|
+
verification?: VerificationProtocol;
|
|
21
|
+
}
|
|
22
|
+
export interface EnvironmentParts {
|
|
23
|
+
nodeRuntime: string;
|
|
24
|
+
lockfile: string;
|
|
25
|
+
capacity: RunCapacity;
|
|
26
|
+
selectedSet?: readonly string[];
|
|
27
|
+
verification: VerificationProtocol;
|
|
19
28
|
}
|
|
20
29
|
export declare function environmentFingerprint(env: GateEnvironmentInput): {
|
|
21
30
|
fingerprint: string;
|
|
22
|
-
parts:
|
|
23
|
-
nodeRuntime: string;
|
|
24
|
-
lockfile: string;
|
|
25
|
-
capacity: RunCapacity;
|
|
26
|
-
selectedSet?: readonly string[];
|
|
27
|
-
};
|
|
31
|
+
parts: EnvironmentParts;
|
|
28
32
|
};
|
|
29
33
|
export type VerificationScope = "battery" | "tip" | "standalone";
|
|
30
34
|
export interface VerificationIdentity {
|
|
@@ -38,12 +42,7 @@ export interface VerificationIdentity {
|
|
|
38
42
|
command: string;
|
|
39
43
|
baseline: string;
|
|
40
44
|
environment: string;
|
|
41
|
-
envParts?:
|
|
42
|
-
nodeRuntime: string;
|
|
43
|
-
lockfile: string;
|
|
44
|
-
capacity: RunCapacity;
|
|
45
|
-
selectedSet?: readonly string[];
|
|
46
|
-
};
|
|
45
|
+
envParts?: EnvironmentParts;
|
|
47
46
|
}
|
|
48
47
|
export declare function computeVerificationIdentity(params: {
|
|
49
48
|
worktree: string;
|
|
@@ -56,6 +55,7 @@ export declare function computeVerificationIdentity(params: {
|
|
|
56
55
|
tree?: string;
|
|
57
56
|
lockfile?: string;
|
|
58
57
|
nodeRuntime?: string;
|
|
58
|
+
verification?: VerificationProtocol;
|
|
59
59
|
}): Promise<VerificationIdentity | undefined>;
|
|
60
60
|
export declare function verificationIdentityKey(id: VerificationIdentity): string;
|
|
61
61
|
export declare function formatReusedDetails(originalDetails: string, id: VerificationIdentity): string;
|
|
@@ -89,6 +89,7 @@ export declare class VerdictStore {
|
|
|
89
89
|
readonly dir: string;
|
|
90
90
|
constructor(dir: string);
|
|
91
91
|
private initSequenceFromDisk;
|
|
92
|
+
private static unknownPolicy;
|
|
92
93
|
get(id?: VerificationIdentity): CachedVerdict | undefined;
|
|
93
94
|
set(id: VerificationIdentity | undefined, verdict: GateResult | CachedVerdict): boolean;
|
|
94
95
|
size(): number;
|
package/dist/gates/cache.js
CHANGED
|
@@ -3,7 +3,7 @@ import { execSync } from "node:child_process";
|
|
|
3
3
|
import { existsSync, mkdirSync, mkdtempSync, readFileSync, readdirSync, realpathSync, rmSync, unlinkSync, writeFileSync } from "node:fs";
|
|
4
4
|
import { dirname, join, resolve } from "node:path";
|
|
5
5
|
import { tmpdir } from "node:os";
|
|
6
|
-
import { describeCapacity, resolvedCapacity, shGit } from "../run/git.js";
|
|
6
|
+
import { describeCapacity, resolvedCapacity, shGit, verificationProtocol } from "../run/git.js";
|
|
7
7
|
import { shq } from "../adapters/types.js";
|
|
8
8
|
export const DEFAULT_VERDICT_CACHE_BOUND = 128;
|
|
9
9
|
let testCacheBound;
|
|
@@ -83,17 +83,23 @@ export function environmentFingerprint(env) {
|
|
|
83
83
|
const cap = env.capacity ?? resolvedCapacity();
|
|
84
84
|
const capacity = { forkCap: cap.forkCap, cores: cap.cores };
|
|
85
85
|
const selectedSet = env.selectedSet ? [...env.selectedSet].sort() : undefined;
|
|
86
|
+
const verification = env.verification ?? verificationProtocol(process.env, env.worktree ?? process.cwd());
|
|
87
|
+
// R41: the protocol and the EFFECTIVE lifecycle are IN the hashed payload, so every entry written
|
|
88
|
+
// before this stamp — green or red — keys differently and is never answered; no store surgery is
|
|
89
|
+
// needed. `source` is provenance (kept in parts, printed on the row) and never enters the key: an
|
|
90
|
+
// explicit `false` and an npmrc `false` are the same policy for the child that ran.
|
|
86
91
|
const payload = canonicalJson({
|
|
87
92
|
nodeRuntime,
|
|
88
93
|
lockfile,
|
|
89
94
|
capacity,
|
|
90
95
|
selectedSet: selectedSet ?? null,
|
|
91
96
|
scope: env.scope ?? "battery",
|
|
97
|
+
verification: { protocol: verification.protocol, lifecycle: verification.lifecycle },
|
|
92
98
|
});
|
|
93
99
|
const fingerprint = createHash("sha256").update(payload).digest("hex").slice(0, 16);
|
|
94
100
|
return {
|
|
95
101
|
fingerprint,
|
|
96
|
-
parts: { nodeRuntime, lockfile, capacity, selectedSet },
|
|
102
|
+
parts: { nodeRuntime, lockfile, capacity, selectedSet, verification },
|
|
97
103
|
};
|
|
98
104
|
}
|
|
99
105
|
export async function computeVerificationIdentity(params) {
|
|
@@ -108,6 +114,7 @@ export async function computeVerificationIdentity(params) {
|
|
|
108
114
|
lockfile: params.lockfile,
|
|
109
115
|
nodeRuntime: params.nodeRuntime,
|
|
110
116
|
scope: params.scope,
|
|
117
|
+
verification: params.verification,
|
|
111
118
|
});
|
|
112
119
|
return {
|
|
113
120
|
gate: params.gate,
|
|
@@ -130,7 +137,7 @@ export function verificationIdentityKey(id) {
|
|
|
130
137
|
export function formatReusedDetails(originalDetails, id) {
|
|
131
138
|
const unadorned = originalDetails.replace(/^reused verdict \(identity: [^)]+\):\s*/, "");
|
|
132
139
|
const envDesc = id.envParts
|
|
133
|
-
? ` [node=${id.envParts.nodeRuntime}, lockfile=${id.envParts.lockfile}, capacity=${describeCapacity(id.envParts.capacity)}${id.envParts.selectedSet ? `, selected=${id.envParts.selectedSet.join(",")}` : ""}]`
|
|
140
|
+
? ` [node=${id.envParts.nodeRuntime}, lockfile=${id.envParts.lockfile}, capacity=${describeCapacity(id.envParts.capacity)}${id.envParts.selectedSet ? `, selected=${id.envParts.selectedSet.join(",")}` : ""}, protocol=${id.envParts.verification.protocol}, lifecycle=${id.envParts.verification.lifecycle} (${id.envParts.verification.source})]`
|
|
134
141
|
: "";
|
|
135
142
|
const gate = id.gate ?? "gate";
|
|
136
143
|
const prefix = `reused ${id.scope === "tip" ? "tip " : ""}verdict (identity: gate=${gate} tree=${id.tree}${id.worktree ? ` worktree=${id.worktree}` : ""} command=${id.command} baseline=${id.baseline} env=${id.environment}${envDesc})`;
|
|
@@ -258,8 +265,13 @@ export class VerdictStore {
|
|
|
258
265
|
// ignore
|
|
259
266
|
}
|
|
260
267
|
}
|
|
268
|
+
// R41: an identity whose lifecycle policy could not be measured is never answered and never
|
|
269
|
+
// stored — an unknown policy is not comparable to anything, so the battery runs the command.
|
|
270
|
+
static unknownPolicy(id) {
|
|
271
|
+
return id.envParts?.verification?.lifecycle === "unknown";
|
|
272
|
+
}
|
|
261
273
|
get(id) {
|
|
262
|
-
if (!id || !id.tree)
|
|
274
|
+
if (!id || !id.tree || VerdictStore.unknownPolicy(id))
|
|
263
275
|
return undefined;
|
|
264
276
|
const key = verificationIdentityKey(id);
|
|
265
277
|
const p = join(this.dir, `verdict-${key}.json`);
|
|
@@ -276,7 +288,7 @@ export class VerdictStore {
|
|
|
276
288
|
return undefined;
|
|
277
289
|
}
|
|
278
290
|
set(id, verdict) {
|
|
279
|
-
if (!id || !id.tree)
|
|
291
|
+
if (!id || !id.tree || VerdictStore.unknownPolicy(id))
|
|
280
292
|
return false;
|
|
281
293
|
if (isInfraResult(verdict))
|
|
282
294
|
return false;
|
package/dist/gates/llm.d.ts
CHANGED
|
@@ -62,11 +62,14 @@ export interface LlmRunResult {
|
|
|
62
62
|
exitCode?: number;
|
|
63
63
|
timedOut: boolean;
|
|
64
64
|
launchNeverStarted?: boolean;
|
|
65
|
+
/** OBS-1039: seat-authored bytes stayed under REVIEW_SILENT_BYTE_FLOOR at the first liveness beat. */
|
|
66
|
+
silentAtBeat?: boolean;
|
|
65
67
|
seatAuthoredBytes?: number;
|
|
66
68
|
}
|
|
67
69
|
export declare const PROMPT_GLYPHS: readonly ["➜", "❯", "$", "%", ">>", ">"];
|
|
68
70
|
export declare function reviewSeatOutput(raw: string, nonce: string, adapterBannerRows?: readonly string[]): string;
|
|
69
71
|
export declare const REVIEW_FIRST_LIVENESS_MS = 30000;
|
|
72
|
+
export declare const REVIEW_SILENT_BYTE_FLOOR = 64;
|
|
70
73
|
export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number): Promise<string>;
|
|
71
74
|
export declare function runViaDriver(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via: LlmVia, timeoutMs?: number): Promise<string>;
|
|
72
75
|
export declare function dewrapPaneVerdict(out: string, nonce: string): string;
|
package/dist/gates/llm.js
CHANGED
|
@@ -300,6 +300,10 @@ export function reviewSeatOutput(raw, nonce, adapterBannerRows = []) {
|
|
|
300
300
|
return trailer ? seat.slice(0, trailer.index) : seat;
|
|
301
301
|
}
|
|
302
302
|
export const REVIEW_FIRST_LIVENESS_MS = 30_000;
|
|
303
|
+
// OBS-1039: a seat that wrote ten bytes and went quiet escaped the zero-byte beat and sat to the
|
|
304
|
+
// ceiling. Below this many seat-authored bytes at the first beat the seat is `silent` — demoted and
|
|
305
|
+
// re-routed then, not at the ceiling. Pane path only; a headless runner buffers and keeps its ceiling.
|
|
306
|
+
export const REVIEW_SILENT_BYTE_FLOOR = 64;
|
|
303
307
|
async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 300000) {
|
|
304
308
|
const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
|
|
305
309
|
try {
|
|
@@ -356,6 +360,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
|
|
|
356
360
|
let out;
|
|
357
361
|
let timedOut = false;
|
|
358
362
|
let launchNeverStarted = false;
|
|
363
|
+
let silentAtBeat = false;
|
|
359
364
|
let seatAuthoredBytes = 0;
|
|
360
365
|
const gatePrompt = prompt.startsWith("TICKMARKR-JUDGE") || prompt.startsWith("TICKMARKR-REVIEW");
|
|
361
366
|
if (!gatePrompt) {
|
|
@@ -412,6 +417,11 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
|
|
|
412
417
|
forceClose = true;
|
|
413
418
|
break;
|
|
414
419
|
}
|
|
420
|
+
if (seatAuthoredBytes < REVIEW_SILENT_BYTE_FLOOR) {
|
|
421
|
+
silentAtBeat = true;
|
|
422
|
+
forceClose = true;
|
|
423
|
+
break;
|
|
424
|
+
}
|
|
415
425
|
}
|
|
416
426
|
// Producing reviews own their full ceiling; inactivity is not a review verdict.
|
|
417
427
|
if (reviewing)
|
|
@@ -458,6 +468,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
|
|
|
458
468
|
...(Number.isFinite(exitCode) ? { exitCode } : {}),
|
|
459
469
|
timedOut,
|
|
460
470
|
launchNeverStarted,
|
|
471
|
+
silentAtBeat,
|
|
461
472
|
seatAuthoredBytes,
|
|
462
473
|
};
|
|
463
474
|
}
|
package/dist/gates/review.d.ts
CHANGED
|
@@ -64,6 +64,13 @@ export declare function matchClosureId(candidate: unknown, fingerprints: Iterabl
|
|
|
64
64
|
* all route through matchClosureId.
|
|
65
65
|
*/
|
|
66
66
|
export declare function isReviewClosureInvalid(v: Pick<ReviewVerdict, "resolved" | "reraised"> | null | undefined, priorIds: ReadonlySet<string> | readonly string[]): boolean;
|
|
67
|
+
/**
|
|
68
|
+
* OBS-1013 add.3: the reviewer ECHOED closure ids and at least one matches no carried fingerprint —
|
|
69
|
+
* it answered about the materials and missed the id (a retyped, truncated or paraphrased fingerprint).
|
|
70
|
+
* That is a no-verdict about the carried work (re-route), not a parse defect. A verdict that omits a
|
|
71
|
+
* list, carries a non-string or duplicates an id stays malformed: its shape, not its ids, is wrong.
|
|
72
|
+
*/
|
|
73
|
+
export declare function isReviewClosureMismatch(v: Pick<ReviewVerdict, "resolved" | "reraised"> | null | undefined, priorIds: ReadonlySet<string> | readonly string[]): boolean;
|
|
67
74
|
export type ReviewerFloorCause = "author-tier" | "task-floor" | "config" | "prior-reviewer";
|
|
68
75
|
/**
|
|
69
76
|
* RF-1 (OBS-922 add.2/3): the tier a reviewer must meet is the maximum of the author's tier, the
|
|
@@ -101,12 +108,30 @@ export declare function pickReviewer(author: Assignment, channels: BillingChanne
|
|
|
101
108
|
prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
|
|
102
109
|
floor?: Tier, // task/config/prior floor from the caller; the author's own tier is ALWAYS applied here (RF-1)
|
|
103
110
|
history?: string[], // run-scoped picks, oldest to newest; empty preserves the established ranking
|
|
104
|
-
onSeat?: (seat: number, count: number) => void, demoted?: ReadonlySet<string>): BillingChannel | null;
|
|
105
|
-
export type ReviewUnparseableCause = VerdictUnparseableCause | "launch-never-started" | "truncated" | "silent";
|
|
111
|
+
onSeat?: (seat: number, count: number) => void, demoted?: ReadonlySet<string>, excludeVendors?: ReadonlySet<string>): BillingChannel | null;
|
|
112
|
+
export type ReviewUnparseableCause = VerdictUnparseableCause | "launch-never-started" | "truncated" | "silent" | "closure-mismatch";
|
|
106
113
|
/**
|
|
107
114
|
* This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
|
|
108
115
|
* remains a separate stated input, so whether its touched paths fit the declaration stays a reviewer
|
|
109
116
|
* judgement rather than a guarantee made by this renderer.
|
|
110
117
|
*/
|
|
111
118
|
export declare function renderDeclaredWriteScope(files: ReadonlyArray<string>): string;
|
|
112
|
-
|
|
119
|
+
/**
|
|
120
|
+
* OBS-1033: the vendors of the seats that authored commits inside the accumulated diff. A seat is
|
|
121
|
+
* never handed its own earlier work to approve. A prior author not resolvable in the pool excludes
|
|
122
|
+
* its adapter's vendors instead (fail closed: the seat is known, its vendor is whatever it bills as).
|
|
123
|
+
*/
|
|
124
|
+
export declare function carriedAuthorVendors(channels: BillingChannel[], carriedAuthors?: readonly string[]): Set<string>;
|
|
125
|
+
/**
|
|
126
|
+
* OBS-1020: the compiled goal is the contract. After `resume --graph-changed` the worktree's copy of
|
|
127
|
+
* the spec is the pre-change text on the integration branch, so a reviewer that reads it grades a
|
|
128
|
+
* superseded contract. The daemon's repository root is where specs and planning records are current.
|
|
129
|
+
*/
|
|
130
|
+
export declare function renderGoalSection(goal: string, repoRoot?: string): string;
|
|
131
|
+
/**
|
|
132
|
+
* OBS-1013 add.3: each carried id is printed ONCE, verbatim, inside a fenced block the reviewer can
|
|
133
|
+
* copy; the notes follow in the same order. A reviewer that retyped a 600-byte id from prose lost
|
|
134
|
+
* closure on a typo and that read as malformed — the block is what a closure list is copied from.
|
|
135
|
+
*/
|
|
136
|
+
export declare function renderPriorMaterials(priorMaterials: readonly StructuredFinding[]): string;
|
|
137
|
+
export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string, reviewHistory?: string[], demotedReviewers?: ReadonlySet<string>, carriedFindings?: readonly StructuredFinding[], priorReviewers?: readonly PriorReviewer[], carriedAuthors?: readonly string[]): Promise<GateResult>;
|
package/dist/gates/review.js
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { existsSync, writeFileSync } from "node:fs";
|
|
2
|
-
import { join } from "node:path";
|
|
2
|
+
import { dirname, join } from "node:path";
|
|
3
3
|
import { channelKey, shq } from "../adapters/types.js";
|
|
4
4
|
import { criticalPathHits, DEFAULT_DIFF_CAP, DEFAULT_REVIEW_CRITICAL_PATHS, declaredReviewPolicy, isReviewLeafPath, raiseReviewPolicy, REVIEW_VERSION_MIRRORS, TIER_RANK, } from "../config/config.js";
|
|
5
5
|
import { filesGlob } from "../graph/files-glob.js";
|
|
@@ -10,6 +10,7 @@ import { structuredFindings } from "../run/journal.js";
|
|
|
10
10
|
import { redactSecrets } from "../run/redact.js";
|
|
11
11
|
import { marginalCostRank } from "../route/router.js";
|
|
12
12
|
import { modelProvider } from "../route/preference.js";
|
|
13
|
+
import { resolveStateDir } from "./cache.js";
|
|
13
14
|
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlmDetailed, verdictNonceLine } from "./llm.js";
|
|
14
15
|
import { classifyVerdictCause } from "./verdict-cause.js";
|
|
15
16
|
import { captureDiffCapFor, measureArtifactDiff, reviewableLogicDiff, } from "./artifact-manifest.js";
|
|
@@ -130,8 +131,14 @@ export function checkDiffCap(gate, measured, cap, prefix = "") {
|
|
|
130
131
|
gate,
|
|
131
132
|
pass: false,
|
|
132
133
|
details: prefix + `diff exceeds verifiable cap (${measured} > ${cap}) — ${DIFF_CAP_REMEDY}`,
|
|
133
|
-
|
|
134
|
-
|
|
134
|
+
meta: {
|
|
135
|
+
park: "diff-cap",
|
|
136
|
+
parkKind: "diff-cap",
|
|
137
|
+
measuredBytes: measured,
|
|
138
|
+
permittedBytes: cap,
|
|
139
|
+
measured,
|
|
140
|
+
permitted: cap,
|
|
141
|
+
},
|
|
135
142
|
};
|
|
136
143
|
}
|
|
137
144
|
/** Apply the strict reviewable-logic cap and the finite, larger capture cap independently. */
|
|
@@ -147,12 +154,19 @@ export function checkTaskDiffCaps(gate, measured, logicCap, prefix = "") {
|
|
|
147
154
|
pass: false,
|
|
148
155
|
details: prefix
|
|
149
156
|
+ `captured artifact diff exceeds verifiable capture cap (${measured.captureBytes} > ${captureCap}) — ${DIFF_CAP_REMEDY}`,
|
|
150
|
-
meta: {
|
|
157
|
+
meta: {
|
|
158
|
+
park: "diff-cap",
|
|
159
|
+
parkKind: "diff-cap",
|
|
160
|
+
measuredBytes: measured.captureBytes,
|
|
161
|
+
permittedBytes: captureCap,
|
|
162
|
+
measured: measured.captureBytes,
|
|
163
|
+
permitted: captureCap,
|
|
164
|
+
},
|
|
151
165
|
};
|
|
152
166
|
}
|
|
153
167
|
export function isDiffCapPark(result) {
|
|
154
168
|
return result.pass === false
|
|
155
|
-
&& result.meta?.
|
|
169
|
+
&& result.meta?.parkKind === "diff-cap"
|
|
156
170
|
&& /diff exceeds verifiable (?:capture )?cap/i.test(result.details);
|
|
157
171
|
}
|
|
158
172
|
// ponytail: single policy hook for callers after runGates — skips the escalation ladder on diff-cap trips.
|
|
@@ -171,7 +185,10 @@ export { modelProvider };
|
|
|
171
185
|
export function matchClosureId(candidate, target) {
|
|
172
186
|
if (typeof candidate !== "string")
|
|
173
187
|
return typeof target === "string" ? false : undefined;
|
|
174
|
-
|
|
188
|
+
// OBS-1068: the brief's copy block printed each id as `Fingerprint: <id>` and told the seat to copy
|
|
189
|
+
// it EXACTLY — three faithful approvals in one run were discarded as closure-mismatch. A leading
|
|
190
|
+
// `Fingerprint:` label is the brief's own wording, never the reviewer's id; strip it before comparing.
|
|
191
|
+
const normCandidate = candidate.replace(/^\s*fingerprint\s*:\s*/i, "").replace(/\s+/g, "");
|
|
175
192
|
if (typeof target === "string") {
|
|
176
193
|
return normCandidate === target.replace(/\s+/g, "");
|
|
177
194
|
}
|
|
@@ -193,6 +210,21 @@ export function isReviewClosureInvalid(v, priorIds) {
|
|
|
193
210
|
|| new Set(allCandidateIds.map((id) => matchClosureId(id, priors) ?? id)).size !== allCandidateIds.length
|
|
194
211
|
|| [...priors].some((id) => !allCandidateIds.some((candidate) => matchClosureId(candidate, id))));
|
|
195
212
|
}
|
|
213
|
+
/**
|
|
214
|
+
* OBS-1013 add.3: the reviewer ECHOED closure ids and at least one matches no carried fingerprint —
|
|
215
|
+
* it answered about the materials and missed the id (a retyped, truncated or paraphrased fingerprint).
|
|
216
|
+
* That is a no-verdict about the carried work (re-route), not a parse defect. A verdict that omits a
|
|
217
|
+
* list, carries a non-string or duplicates an id stays malformed: its shape, not its ids, is wrong.
|
|
218
|
+
*/
|
|
219
|
+
export function isReviewClosureMismatch(v, priorIds) {
|
|
220
|
+
if (!v || !Array.isArray(v.resolved) || !Array.isArray(v.reraised))
|
|
221
|
+
return false;
|
|
222
|
+
const ids = [...v.resolved, ...v.reraised];
|
|
223
|
+
if (!ids.every((id) => typeof id === "string"))
|
|
224
|
+
return false;
|
|
225
|
+
const priors = priorIds instanceof Set ? priorIds : new Set(priorIds);
|
|
226
|
+
return ids.some((id) => matchClosureId(id, priors) === undefined);
|
|
227
|
+
}
|
|
196
228
|
// v1.53 T2: same entry grammar as routing.map.prefer (router.ts preferIndex — router is out of this
|
|
197
229
|
// module's dependency direction for a private fn, so the 3 lines live here too): `adapter` matches
|
|
198
230
|
// every channel of that adapter, `adapter:model` exactly one; unmatched channels sort after all entries.
|
|
@@ -247,7 +279,9 @@ export function pickReviewer(author, channels, exclude = [], // v1.1 failover: r
|
|
|
247
279
|
prefer = [], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
|
|
248
280
|
floor, // task/config/prior floor from the caller; the author's own tier is ALWAYS applied here (RF-1)
|
|
249
281
|
history = [], // run-scoped picks, oldest to newest; empty preserves the established ranking
|
|
250
|
-
onSeat, demoted = new Set()
|
|
282
|
+
onSeat, demoted = new Set(),
|
|
283
|
+
// OBS-1033: vendors that authored a carried commit inside the accumulated diff — excluded for the round.
|
|
284
|
+
excludeVendors = new Set()) {
|
|
251
285
|
// FLEET-05 success criterion 2: an author not resolvable in the channel list yields NO reviewer.
|
|
252
286
|
// The old `?? author.adapter` fallback compared an adapter id to vendor names, matched nothing, and
|
|
253
287
|
// admitted every reviewer — including the author's own channel (fail-OPEN). null lands on reviewGate's
|
|
@@ -267,6 +301,7 @@ onSeat, demoted = new Set()) {
|
|
|
267
301
|
&& modelProvider(c.model, c.vendor) !== authorProvider
|
|
268
302
|
&& modelId(c.model) !== modelId(author.model)
|
|
269
303
|
&& !exclude.includes(channelKey(c))
|
|
304
|
+
&& !excludeVendors.has(c.vendor)
|
|
270
305
|
&& TIER_RANK[c.tier] >= TIER_RANK[effectiveFloor])
|
|
271
306
|
.sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b));
|
|
272
307
|
const reviewer = [...ranked].sort((a, b) => Number(demoted.has(channelKey(a))) - Number(demoted.has(channelKey(b)))
|
|
@@ -289,13 +324,66 @@ export function renderDeclaredWriteScope(files) {
|
|
|
289
324
|
The task DECLARED these write-scope patterns:
|
|
290
325
|
${files.map((path) => `- ${path}`).join("\n")}`;
|
|
291
326
|
}
|
|
327
|
+
/**
|
|
328
|
+
* OBS-1033: the vendors of the seats that authored commits inside the accumulated diff. A seat is
|
|
329
|
+
* never handed its own earlier work to approve. A prior author not resolvable in the pool excludes
|
|
330
|
+
* its adapter's vendors instead (fail closed: the seat is known, its vendor is whatever it bills as).
|
|
331
|
+
*/
|
|
332
|
+
export function carriedAuthorVendors(channels, carriedAuthors = []) {
|
|
333
|
+
const vendors = new Set();
|
|
334
|
+
for (const key of carriedAuthors) {
|
|
335
|
+
const adapter = key.split(":")[0];
|
|
336
|
+
const exact = channels.filter((c) => channelKey(c) === key);
|
|
337
|
+
for (const c of exact.length ? exact : channels.filter((c) => c.adapter === adapter))
|
|
338
|
+
vendors.add(c.vendor);
|
|
339
|
+
}
|
|
340
|
+
return vendors;
|
|
341
|
+
}
|
|
342
|
+
/**
|
|
343
|
+
* OBS-1020: the compiled goal is the contract. After `resume --graph-changed` the worktree's copy of
|
|
344
|
+
* the spec is the pre-change text on the integration branch, so a reviewer that reads it grades a
|
|
345
|
+
* superseded contract. The daemon's repository root is where specs and planning records are current.
|
|
346
|
+
*/
|
|
347
|
+
export function renderGoalSection(goal, repoRoot) {
|
|
348
|
+
return `## Goal (authoritative — compiled from the sealed graph; the worktree's spec file may be stale after resume --graph-changed)
|
|
349
|
+
${goal}
|
|
350
|
+
${repoRoot ? `Specs and planning records are read in the daemon's repository root ${repoRoot} (its specs/ and .planning/), never this worktree's copies.` : ""}`;
|
|
351
|
+
}
|
|
352
|
+
/**
|
|
353
|
+
* The daemon's repository root: the parent of the state dir the run's artifacts live under. Named
|
|
354
|
+
* only when that state dir exists — a guessed one would send the reviewer to a path that holds nothing.
|
|
355
|
+
*/
|
|
356
|
+
function daemonRepoRoot(worktree, artifactDir) {
|
|
357
|
+
try {
|
|
358
|
+
const stateDir = resolveStateDir(worktree, artifactDir);
|
|
359
|
+
return stateDir.endsWith("/.tickmarkr") && existsSync(stateDir) ? dirname(stateDir) : undefined;
|
|
360
|
+
}
|
|
361
|
+
catch {
|
|
362
|
+
return undefined;
|
|
363
|
+
}
|
|
364
|
+
}
|
|
365
|
+
/**
|
|
366
|
+
* OBS-1013 add.3: each carried id is printed ONCE, verbatim, inside a fenced block the reviewer can
|
|
367
|
+
* copy; the notes follow in the same order. A reviewer that retyped a 600-byte id from prose lost
|
|
368
|
+
* closure on a typo and that read as malformed — the block is what a closure list is copied from.
|
|
369
|
+
*/
|
|
370
|
+
export function renderPriorMaterials(priorMaterials) {
|
|
371
|
+
return `## Prior materials this attempt must close
|
|
372
|
+
Copy each fingerprint below EXACTLY (they appear once, in this block) into resolved or reraised:
|
|
373
|
+
\`\`\`text
|
|
374
|
+
${priorMaterials.map((finding) => `Fingerprint: ${finding.fingerprint}`).join("\n")}
|
|
375
|
+
\`\`\`
|
|
376
|
+
${priorMaterials.map((finding, i) => `${i + 1}. ${finding.note}`).join("\n\n")}`;
|
|
377
|
+
}
|
|
292
378
|
export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
|
|
293
379
|
// OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
|
|
294
380
|
// direct tests) skips persistence and changes nothing else.
|
|
295
381
|
artifactDir, reviewHistory, demotedReviewers, carriedFindings = [],
|
|
296
382
|
// RF-1: channel keys of THIS task's prior reviewers (earlier rounds, a flaked seat) — task-scoped,
|
|
297
383
|
// never the run-wide rotation history nor excludeReviewers; the seat holds the highest of their tiers.
|
|
298
|
-
priorReviewers = []
|
|
384
|
+
priorReviewers = [],
|
|
385
|
+
// OBS-1033: channel keys of the seats that authored the carried commits (the daemon's tried list).
|
|
386
|
+
carriedAuthors = []) {
|
|
299
387
|
// R3 (OBS-186): participation is keyed on PATHS. The compiler's assignment comes from the DECLARED
|
|
300
388
|
// files[]; the operator's floor may RAISE it to full and can never lower it. `complexityThreshold` is
|
|
301
389
|
// retired — the branch that returned a green skip on a complexity comparison is gone, and with it the
|
|
@@ -370,7 +458,7 @@ priorReviewers = []) {
|
|
|
370
458
|
const { floor: reviewerFloor, cause: reviewerFloorCause } = gateReviewerFloor(task, cfg, author, channels, priorReviewers);
|
|
371
459
|
const floorMeta = { reviewerFloor, reviewerFloorCause };
|
|
372
460
|
let rotationSeat;
|
|
373
|
-
const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor, reviewHistory, reviewHistory ? (seat) => { rotationSeat = seat; } : undefined, demotedReviewers);
|
|
461
|
+
const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor, reviewHistory, reviewHistory ? (seat) => { rotationSeat = seat; } : undefined, demotedReviewers, carriedAuthorVendors(channels, carriedAuthors));
|
|
374
462
|
if (!reviewer) {
|
|
375
463
|
// meta.noEligibleReviewer lets run-gates' review-retry keep the ORIGINAL unparseable result when
|
|
376
464
|
// the retry finds no second seat — a truthful cause beats a synthetic no-reviewer failure.
|
|
@@ -390,6 +478,12 @@ priorReviewers = []) {
|
|
|
390
478
|
if (capFail)
|
|
391
479
|
return capFail;
|
|
392
480
|
const nonce = generateVerdictNonce();
|
|
481
|
+
const repoRoot = daemonRepoRoot(worktree, artifactDir);
|
|
482
|
+
// OBS-880 add.1: guidance only. Do not expand scope globs into permission to run suites.
|
|
483
|
+
const ownTestFiles = [...new Set(task.files.filter((file) => !/[*?[\]{}()!]/.test(file) && /(?:^|\/)[^/]*\.(?:test|spec)\.[cm]?[jt]sx?$/.test(file)))];
|
|
484
|
+
const suiteBudget = ownTestFiles.length
|
|
485
|
+
? `You may run at most the task's own test files explicitly named in files[]; these are the only suites you may run: ${ownTestFiles.map((file) => `\`${file}\``).join(", ")}.`
|
|
486
|
+
: "No suite may be run: files[] names no explicit test file owned by this task.";
|
|
393
487
|
const prompt = `TICKMARKR-REVIEW
|
|
394
488
|
You are a skeptical cross-vendor code reviewer. Another agent (vendor: ${author.adapter}) authored this diff.
|
|
395
489
|
Look for correctness bugs, security issues, and acceptance-criteria gaps. Approve only if you would merge it.
|
|
@@ -397,13 +491,16 @@ Look for correctness bugs, security issues, and acceptance-criteria gaps. Approv
|
|
|
397
491
|
${COMPLETION_FAKING_CHECKLIST}
|
|
398
492
|
|
|
399
493
|
## Task ${task.id}: ${task.title} (complexity ${task.complexity})
|
|
494
|
+
${renderGoalSection(task.goal, repoRoot)}
|
|
400
495
|
## Acceptance criteria
|
|
401
496
|
${task.acceptance.map((a) => `- ${renderAcceptanceItem(a)}`).join("\n")}
|
|
402
497
|
|
|
403
498
|
${renderDeclaredWriteScope(task.files)}
|
|
404
499
|
|
|
405
|
-
|
|
406
|
-
${
|
|
500
|
+
## Reviewer suite budget
|
|
501
|
+
${suiteBudget} Never run the whole suite (including an unfiltered npm test or vitest run). The gate suite owns the runner lease; a parallel full suite starves the gate.
|
|
502
|
+
|
|
503
|
+
${priorMaterials.length ? `${renderPriorMaterials(priorMaterials)}
|
|
407
504
|
|
|
408
505
|
` : ""}## Diff
|
|
409
506
|
\`\`\`diff
|
|
@@ -483,17 +580,22 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
483
580
|
const findings = v && Array.isArray(v.findings) ? v.findings : null;
|
|
484
581
|
const priorIds = new Set(priorMaterials.map((finding) => finding.fingerprint));
|
|
485
582
|
const closureInvalid = isReviewClosureInvalid(v, priorIds);
|
|
583
|
+
const closureMismatch = closureInvalid && isReviewClosureMismatch(v, priorIds);
|
|
486
584
|
// findings decides the verdict on its own; the legacy path still needs approve + issues to parse.
|
|
487
585
|
if (!v || closureInvalid || (findings === null && (typeof v.approve !== "boolean" || !Array.isArray(v.issues)))) {
|
|
488
586
|
// OBS-196: name the cause and persist the raw bytes — a ruled-on "unparseable" without its
|
|
489
587
|
// evidence cannot be audited, and a cutoff must never be indistinguishable from a parse defect.
|
|
490
588
|
const bytes = llm.seatAuthoredBytes ?? Buffer.byteLength(raw.trim(), "utf8");
|
|
491
|
-
const cause =
|
|
492
|
-
: llm.
|
|
493
|
-
:
|
|
589
|
+
const cause = closureMismatch ? "closure-mismatch" : closureInvalid ? "malformed-verdict"
|
|
590
|
+
: llm.launchNeverStarted ? "launch-never-started"
|
|
591
|
+
: llm.silentAtBeat ? "silent"
|
|
592
|
+
: llm.timedOut ? (bytes > 0 ? "truncated" : "silent")
|
|
593
|
+
: classifyVerdictCause(raw, nonce, "approve", llm);
|
|
494
594
|
const failure = cause === "malformed-verdict"
|
|
495
595
|
? "review output unparseable"
|
|
496
|
-
:
|
|
596
|
+
: cause === "closure-mismatch"
|
|
597
|
+
? "review verdict closes no carried fingerprint — closure ids match none of the carried materials"
|
|
598
|
+
: "review dispatch failed — no structurally valid nonce-bound response";
|
|
497
599
|
return {
|
|
498
600
|
gate: "review",
|
|
499
601
|
pass: false,
|
|
@@ -508,6 +610,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
508
610
|
provider,
|
|
509
611
|
...(cause === "malformed-verdict" ? { unparseable: true } : { noVerdict: true, classification: "infra", infra: true }),
|
|
510
612
|
cause,
|
|
613
|
+
...(closureMismatch ? { resolved: v?.resolved, reraised: v?.reraised, carriedFingerprints: [...priorIds] } : {}),
|
|
511
614
|
bytes, seatAuthoredBytes: bytes,
|
|
512
615
|
...(saved ? { rawPath: saved } : {}),
|
|
513
616
|
...(savedBrief ? { briefPath: savedBrief } : {}),
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import type { CommandReceiptAttribution } from "../run/protocol.js";
|
|
1
2
|
import { type Assignment, type BillingChannel, type WorkerAdapter, type WorkerResult } from "../adapters/types.js";
|
|
2
3
|
import { type TickmarkrConfig } from "../config/config.js";
|
|
3
4
|
import { type GateName, type Task } from "../graph/schema.js";
|
|
@@ -5,6 +6,7 @@ import { type Baseline } from "./baseline.js";
|
|
|
5
6
|
import { type GateVia } from "./llm.js";
|
|
6
7
|
import { type PriorReviewer } from "./review.js";
|
|
7
8
|
import type { GateResult } from "./types.js";
|
|
9
|
+
import { type VerificationRetryCause } from "../run/recovery.js";
|
|
8
10
|
import { type StructuredFinding } from "../run/journal.js";
|
|
9
11
|
import { type VerificationScope } from "./cache.js";
|
|
10
12
|
export type LoadProvider = () => number;
|
|
@@ -41,9 +43,11 @@ export type GateEvent = {
|
|
|
41
43
|
gate: GateName;
|
|
42
44
|
name: string;
|
|
43
45
|
payload: Record<string, unknown>;
|
|
44
|
-
result
|
|
46
|
+
result?: GateResult;
|
|
45
47
|
};
|
|
46
48
|
export interface GateContext {
|
|
49
|
+
buildReceiptIdentity?: Omit<CommandReceiptAttribution, "invocation">;
|
|
50
|
+
authorizeInfraRetry?: (subject: string, cause?: VerificationRetryCause) => boolean;
|
|
47
51
|
verificationScope?: VerificationScope;
|
|
48
52
|
worktree: string;
|
|
49
53
|
baseRef: string;
|
|
@@ -59,11 +63,15 @@ export interface GateContext {
|
|
|
59
63
|
carriedFindings?: readonly StructuredFinding[];
|
|
60
64
|
excludeReviewers?: string[];
|
|
61
65
|
demotedReviewers?: Set<string>;
|
|
66
|
+
reviewNoVerdicts?: Map<string, string[]>;
|
|
67
|
+
recheck?: boolean;
|
|
68
|
+
carriedAuthors?: readonly string[];
|
|
62
69
|
reviewHistory?: string[];
|
|
63
70
|
priorReviewers?: PriorReviewer[];
|
|
64
71
|
artifactDir?: string;
|
|
65
|
-
pipeline?: "v185" | "legacy";
|
|
66
72
|
selectTests?: boolean;
|
|
73
|
+
requiredRepairTests?: readonly string[];
|
|
74
|
+
selectionReason?: string;
|
|
67
75
|
collateral?: ReadonlyArray<string>;
|
|
68
76
|
onGate?: (e: GateEvent) => void | Promise<void>;
|
|
69
77
|
stateDir?: string;
|