tickmarkr 2.3.0 → 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/catalog-remote.d.ts +1 -4
- package/dist/adapters/catalog-remote.js +52 -42
- package/dist/adapters/catalog.js +5 -3
- package/dist/adapters/claude-code.d.ts +1 -1
- package/dist/adapters/claude-code.js +8 -5
- package/dist/adapters/model-lints.d.ts +9 -5
- package/dist/adapters/model-lints.js +56 -15
- package/dist/adapters/model-windows.js +11 -0
- package/dist/adapters/prompt.js +1 -0
- package/dist/adapters/qwen.d.ts +5 -0
- package/dist/adapters/qwen.js +153 -0
- package/dist/adapters/types.d.ts +1 -0
- package/dist/adapters/types.js +1 -0
- package/dist/cli/commands/compile.js +32 -6
- package/dist/cli/commands/doctor.d.ts +4 -3
- package/dist/cli/commands/doctor.js +20 -6
- package/dist/cli/commands/fleet.d.ts +4 -0
- package/dist/cli/commands/fleet.js +53 -14
- package/dist/cli/commands/plan.js +29 -5
- package/dist/cli/commands/status.d.ts +1 -0
- package/dist/cli/commands/status.js +45 -1
- package/dist/cli/commands/verify.d.ts +1 -0
- package/dist/cli/commands/verify.js +5 -0
- package/dist/cli/index.d.ts +1 -1
- package/dist/cli/index.js +2 -2
- package/dist/compile/collateral.d.ts +2 -9
- package/dist/compile/collateral.js +2 -9
- package/dist/compile/index.d.ts +4 -1
- package/dist/compile/index.js +41 -7
- package/dist/compile/native.d.ts +4 -2
- package/dist/compile/native.js +55 -6
- package/dist/compile/ownership.js +34 -9
- package/dist/config/config.d.ts +1 -0
- package/dist/config/config.js +51 -5
- package/dist/drivers/herdr.d.ts +1 -0
- package/dist/drivers/herdr.js +11 -1
- package/dist/drivers/orca.d.ts +14 -1
- package/dist/drivers/orca.js +114 -15
- package/dist/drivers/types.d.ts +10 -0
- package/dist/gates/baseline.js +26 -7
- package/dist/gates/review.d.ts +4 -2
- package/dist/gates/review.js +24 -13
- package/dist/gates/run-gates.d.ts +5 -2
- package/dist/gates/run-gates.js +34 -17
- package/dist/route/preference.d.ts +4 -0
- package/dist/route/preference.js +40 -0
- package/dist/route/router.js +15 -2
- package/dist/run/consult.d.ts +1 -0
- package/dist/run/consult.js +34 -7
- package/dist/run/daemon.d.ts +4 -0
- package/dist/run/daemon.js +128 -50
- package/dist/run/git.d.ts +2 -0
- package/dist/run/git.js +36 -5
- package/dist/run/journal.js +4 -1
- package/dist/tui/ink/fleet-app.d.ts +4 -0
- package/dist/tui/ink/fleet-app.js +45 -16
- package/package.json +59 -1
package/dist/drivers/types.d.ts
CHANGED
|
@@ -66,9 +66,17 @@ export declare function panesToClose(agents: FleetAgent[], desired: Set<string>,
|
|
|
66
66
|
tabId?: string;
|
|
67
67
|
}[];
|
|
68
68
|
export declare function canonicalizeLegacyName(name: string, runId: string): OwnedName;
|
|
69
|
+
export interface SlotPlacement {
|
|
70
|
+
surface?: string;
|
|
71
|
+
hostPlatform?: string;
|
|
72
|
+
}
|
|
69
73
|
export interface ExecutorDriver {
|
|
70
74
|
id: string;
|
|
71
75
|
interactive: boolean;
|
|
76
|
+
/** The exact terminal read surface used for liveness evidence. */
|
|
77
|
+
readSource?: string;
|
|
78
|
+
/** Placement facts returned by drivers whose terminal host exposes them. */
|
|
79
|
+
describe?(slot: Slot): SlotPlacement | Promise<SlotPlacement>;
|
|
72
80
|
slot(cwd: string, name: string, opts?: SlotOpts): Promise<Slot>;
|
|
73
81
|
run(slot: Slot, cmd: string): Promise<void>;
|
|
74
82
|
waitOutput(slot: Slot, pattern: string, timeoutMs: number, opts?: {
|
|
@@ -84,5 +92,7 @@ export interface ExecutorDriver {
|
|
|
84
92
|
narrateWith?(narrate: (event: JournalEvent) => void): void;
|
|
85
93
|
worktree(repo: string, branch: string, baseRef: string): Promise<string>;
|
|
86
94
|
narrator?: (cwd: string, command: string, runId?: string) => Promise<Slot>;
|
|
95
|
+
/** Best-effort projection of a task's lifecycle onto the execution host. */
|
|
96
|
+
project?: (taskId: string, state: "in-progress" | "in-review" | "completed") => Promise<void>;
|
|
87
97
|
reconcile?: (desired: Set<string>, runId: string, opts?: PanesToCloseOpts) => Promise<void>;
|
|
88
98
|
}
|
package/dist/gates/baseline.js
CHANGED
|
@@ -5,10 +5,19 @@ import { DEFAULT_SHELL_TIMEOUT_MS, describeCapacity, sameCapacity, sh } from "..
|
|
|
5
5
|
// codes that varied between baseline and worktree runs, was reported as a "new failure". Strip ANSI first;
|
|
6
6
|
// a pass-marker line is never a failure. [\d;#] covers raw ANSI and digit-normalized ANSI ("\x1b[#m") from
|
|
7
7
|
// baselines stored by pre-hardening code.
|
|
8
|
-
|
|
8
|
+
// OBS-891 (run 3372): vitest toggles the cursor (`\x1b[?25l` / `\x1b[?25h`) around its progress
|
|
9
|
+
// output, and the private-mode parameter byte `?` never matched [\d;#], so an echo-block HEADER glued
|
|
10
|
+
// to a cursor-show sequence stayed invisible to withoutVitestEchoBlocks and its whole block leaked as
|
|
11
|
+
// runner evidence — seven prose-only "infra" parks in one night. Full CSI grammar: parameter bytes
|
|
12
|
+
// 0x30–0x3F (plus `#` for digit-normalized stored baselines), intermediates 0x20–0x2F, final 0x40–0x7E.
|
|
13
|
+
const ANSI_RE = /\x1b\[[0-?#]*[ -/]*[@-~]/g;
|
|
9
14
|
// ponytail: only leading ✓/✔ after optional "label:" prefixes (turbo/vitest), or tickmarkr's own run
|
|
10
15
|
// summary, counts as a pass line — other runners' pass markers (PASS, ok) stay fingerprintable
|
|
11
16
|
const PASS_LINE_RE = /^\s*(?:(?:[\w@./-]+:\s*)*[✓✔]|(?:\[tickmarkr\]\s+)?(?:tickmarkr\s+[\w.-]+:\s+)?(?:\d+|#)\s+done,\s+(?:\d+|#)\s+failed(?:,\s+(?:\d+|#)\s+awaiting human)?\b)/;
|
|
17
|
+
// OBS-888: tickmarkr's own operator lines (`tickmarkr: baseline capture for "test" … spawn EAGAIN`)
|
|
18
|
+
// are printed by this product, never by a runner about the work. When this repository's tests exercise
|
|
19
|
+
// the capture path they print them too, carrying errno tokens INFRA_RE would read as host evidence.
|
|
20
|
+
const OPERATOR_LINE_RE = /^\s*tickmarkr: /;
|
|
12
21
|
// HYG-08 (D-01, incident run-20260711-154920): a failing test went unnamed for 3 attempts because details
|
|
13
22
|
// headlined benign fingerprint-diff noise. These anchors harvest the runner's OWN failure naming from fresh
|
|
14
23
|
// output to headline it. \s is fine in a TS regex — the BSD [[:space:]] rule binds shell grep only.
|
|
@@ -149,7 +158,10 @@ const isFingerprintShaped = (l) => isFailureShaped(l) || INFRA_RE.test(l);
|
|
|
149
158
|
* unreadable-runner case the existing fail-closed path already owns.
|
|
150
159
|
*/
|
|
151
160
|
export function classifyFailureOutput(output) {
|
|
152
|
-
|
|
161
|
+
// OBS-891: the gate reads the WHOLE output here when the fresh-fingerprint diff is empty, so an errno
|
|
162
|
+
// token inside a test's echoed stdout/stderr block must be as invisible to this classifier as it is
|
|
163
|
+
// to fingerprint(): test-owned output is never runner evidence about the work.
|
|
164
|
+
const lines = withoutVitestEchoBlocks(output).map((l) => l.replace(ANSI_RE, "")).filter((l) => !PASS_LINE_RE.test(l) && !OPERATOR_LINE_RE.test(l));
|
|
153
165
|
if (lines.some(namesRegression))
|
|
154
166
|
return "regression";
|
|
155
167
|
return lines.some(isInfraLine) ? "infra" : undefined;
|
|
@@ -161,7 +173,11 @@ export function classifyFailureOutput(output) {
|
|
|
161
173
|
* no failure in that incomplete environment can safely become pre-existing forgiveness. Keep this
|
|
162
174
|
* separate from `classifyFailureOutput` so changing capture policy cannot move gate verdicts.
|
|
163
175
|
*/
|
|
164
|
-
|
|
176
|
+
// OBS-888 row 1: vitest 3.2.7 heads an echo block `std{out,err} | <file> > <test>` when the log is
|
|
177
|
+
// attributed to a test, `stderr | unknown test` when it is not, `stderr | <file>` for file-level output
|
|
178
|
+
// and `stderr | <task id>` (digits and underscores) when the reporter no longer knows the task
|
|
179
|
+
// (dist/chunks/index.*.js, `headerText`). The stripper knew only the first form.
|
|
180
|
+
const VITEST_ECHO_BLOCK_RE = /^\s*std(?:out|err)\s+\|\s+(?:unknown test\s*$|\d+_[\d_]+\s*$|\S+\.(?:test|spec)\.[cm]?[jt]sx?(?:\s*$|\s+>\s+\S))/;
|
|
165
181
|
/** Keep runner output while omitting Vitest's echoed test-owned stdout/stderr diagnostic blocks. */
|
|
166
182
|
const withoutVitestEchoBlocks = (output) => {
|
|
167
183
|
const outside = [];
|
|
@@ -186,7 +202,7 @@ const captureInvalidatingLines = (output) => {
|
|
|
186
202
|
const invalidating = [];
|
|
187
203
|
for (const line of withoutVitestEchoBlocks(output)) {
|
|
188
204
|
const clean = line.replace(ANSI_RE, "");
|
|
189
|
-
if (!PASS_LINE_RE.test(clean) && CAPTURE_EXHAUSTION_RE.test(clean))
|
|
205
|
+
if (!PASS_LINE_RE.test(clean) && !OPERATOR_LINE_RE.test(clean) && CAPTURE_EXHAUSTION_RE.test(clean))
|
|
190
206
|
invalidating.push(line);
|
|
191
207
|
}
|
|
192
208
|
return invalidating;
|
|
@@ -230,16 +246,18 @@ export const UNRECOGNIZED_FAILURE = "<unrecognized failure output>";
|
|
|
230
246
|
export function fingerprint(output) {
|
|
231
247
|
const lines = withoutVitestEchoBlocks(output)
|
|
232
248
|
.map((l) => l.replace(ANSI_RE, ""))
|
|
233
|
-
.filter((l) => !PASS_LINE_RE.test(l));
|
|
249
|
+
.filter((l) => !PASS_LINE_RE.test(l) && !OPERATOR_LINE_RE.test(l));
|
|
234
250
|
// GATE-FIX-4 DEFECT 4: every line is read twice — as printed, and with a turbo `<pkg>:<task>:`
|
|
235
251
|
// prefix removed. A recognized stripped line fingerprints as its STRIPPED text, so the same
|
|
236
252
|
// failure fingerprints identically whether turbo prefixed it or a bare runner printed it; the
|
|
237
253
|
// prefixed form keeps fingerprinting too (baseline-recorded package-level reds stay forgivable).
|
|
238
254
|
const shaped = [];
|
|
239
255
|
for (const l of lines) {
|
|
256
|
+
const stripped = stripTurboPrefix(l);
|
|
257
|
+
if (stripped !== undefined && OPERATOR_LINE_RE.test(stripped))
|
|
258
|
+
continue; // OBS-888: operator prose under a turbo prefix
|
|
240
259
|
if (isFingerprintShaped(l))
|
|
241
260
|
shaped.push(l);
|
|
242
|
-
const stripped = stripTurboPrefix(l);
|
|
243
261
|
if (stripped !== undefined && !PASS_LINE_RE.test(stripped) && isFingerprintShaped(stripped))
|
|
244
262
|
shaped.push(stripped);
|
|
245
263
|
}
|
|
@@ -576,7 +594,8 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
|
576
594
|
// result rather than re-derived after the fact. The skip row above ran no command and therefore
|
|
577
595
|
// states no capacity — a row that never divided the machine must not claim that it did.
|
|
578
596
|
const record = (g) => {
|
|
579
|
-
|
|
597
|
+
const withReap = r.reapedGroup ? { ...g, meta: { ...g.meta, reapedGroup: true } } : g;
|
|
598
|
+
results.push(r.capacity ? { ...withReap, capacity: r.capacity } : withReap);
|
|
580
599
|
};
|
|
581
600
|
// …and whether the entry that would forgive this command was measured in the same world. A
|
|
582
601
|
// baseline captured under a different fork cap forgives nothing: its fingerprints describe a
|
package/dist/gates/review.d.ts
CHANGED
|
@@ -52,7 +52,9 @@ export declare function modelId(model: string): string;
|
|
|
52
52
|
export declare function modelProvider(model: string, fallback?: string): string;
|
|
53
53
|
export declare function pickReviewer(author: Assignment, channels: BillingChannel[], exclude?: string[], // v1.1 failover: reviewer channels that already produced garbage for this task
|
|
54
54
|
prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
|
|
55
|
-
floor?: Tier
|
|
55
|
+
floor?: Tier, // task-declared only; config floors govern workers and must not silently move review seats
|
|
56
|
+
history?: string[], // run-scoped picks, oldest to newest; empty preserves the established ranking
|
|
57
|
+
onSeat?: (seat: number) => void): BillingChannel | null;
|
|
56
58
|
export type ReviewUnparseableCause = VerdictUnparseableCause;
|
|
57
59
|
/**
|
|
58
60
|
* This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
|
|
@@ -60,4 +62,4 @@ export type ReviewUnparseableCause = VerdictUnparseableCause;
|
|
|
60
62
|
* judgement rather than a guarantee made by this renderer.
|
|
61
63
|
*/
|
|
62
64
|
export declare function renderDeclaredWriteScope(files: ReadonlyArray<string>): string;
|
|
63
|
-
export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string): Promise<GateResult>;
|
|
65
|
+
export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string, reviewHistory?: string[]): Promise<GateResult>;
|
package/dist/gates/review.js
CHANGED
|
@@ -192,7 +192,9 @@ function reviewPreferIndex(c, prefer) {
|
|
|
192
192
|
}
|
|
193
193
|
export function pickReviewer(author, channels, exclude = [], // v1.1 failover: reviewer channels that already produced garbage for this task
|
|
194
194
|
prefer = [], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
|
|
195
|
-
floor
|
|
195
|
+
floor, // task-declared only; config floors govern workers and must not silently move review seats
|
|
196
|
+
history = [], // run-scoped picks, oldest to newest; empty preserves the established ranking
|
|
197
|
+
onSeat) {
|
|
196
198
|
// FLEET-05 success criterion 2: an author not resolvable in the channel list yields NO reviewer.
|
|
197
199
|
// The old `?? author.adapter` fallback compared an adapter id to vendor names, matched nothing, and
|
|
198
200
|
// admitted every reviewer — including the author's own channel (fail-OPEN). null lands on reviewGate's
|
|
@@ -200,18 +202,23 @@ floor) {
|
|
|
200
202
|
const authorChannel = channels.find((c) => c.adapter === author.adapter && c.model === author.model);
|
|
201
203
|
if (!authorChannel)
|
|
202
204
|
return null;
|
|
203
|
-
// Failover also guards true provider; the initial pick keeps the established stamped-vendor contract.
|
|
204
205
|
const authorProvider = modelProvider(author.model, authorChannel.vendor);
|
|
205
|
-
|
|
206
|
+
const ranked = channels
|
|
206
207
|
// two independent axes: different vendor AND different base-model identity (ADDED TO the vendor
|
|
207
|
-
// rule, never replacing it — a future edit can't silently drop either).
|
|
208
|
-
//
|
|
208
|
+
// rule, never replacing it — a future edit can't silently drop either). Failover additionally guards
|
|
209
|
+
// true provider identity; the initial pick keeps the established stamped-vendor contract. The diversity
|
|
210
|
+
// filter runs BEFORE preference ranking, so prefer cannot resurrect an excluded channel.
|
|
209
211
|
.filter((c) => c.vendor !== authorChannel.vendor
|
|
210
212
|
&& (exclude.length === 0 || modelProvider(c.model, c.vendor) !== authorProvider)
|
|
211
213
|
&& modelId(c.model) !== modelId(author.model)
|
|
212
214
|
&& !exclude.includes(channelKey(c))
|
|
213
215
|
&& (floor === undefined || TIER_RANK[c.tier] >= TIER_RANK[floor]))
|
|
214
|
-
.sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b))
|
|
216
|
+
.sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b));
|
|
217
|
+
const reviewer = [...ranked].sort((a, b) => history.lastIndexOf(channelKey(a)) - history.lastIndexOf(channelKey(b))
|
|
218
|
+
|| ranked.indexOf(a) - ranked.indexOf(b))[0] ?? null;
|
|
219
|
+
if (reviewer)
|
|
220
|
+
onSeat?.(ranked.indexOf(reviewer) + 1);
|
|
221
|
+
return reviewer;
|
|
215
222
|
}
|
|
216
223
|
/**
|
|
217
224
|
* This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
|
|
@@ -229,7 +236,7 @@ ${files.map((path) => `- ${path}`).join("\n")}`;
|
|
|
229
236
|
export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
|
|
230
237
|
// OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
|
|
231
238
|
// direct tests) skips persistence and changes nothing else.
|
|
232
|
-
artifactDir) {
|
|
239
|
+
artifactDir, reviewHistory) {
|
|
233
240
|
// R3 (OBS-186): participation is keyed on PATHS. The compiler's assignment comes from the DECLARED
|
|
234
241
|
// files[]; the operator's floor may RAISE it to full and can never lower it. `complexityThreshold` is
|
|
235
242
|
// retired — the branch that returned a green skip on a complexity comparison is gone, and with it the
|
|
@@ -301,7 +308,8 @@ artifactDir) {
|
|
|
301
308
|
// A reviewer floor is opt-in at the task. Applying cfg.routing.floors here would change the
|
|
302
309
|
// historical seat for every task that never asked for review-tier coupling.
|
|
303
310
|
const reviewerFloor = task.routingHints?.floor;
|
|
304
|
-
|
|
311
|
+
let rotationSeat;
|
|
312
|
+
const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor, reviewHistory, reviewHistory ? (seat) => { rotationSeat = seat; } : undefined);
|
|
305
313
|
if (!reviewer) {
|
|
306
314
|
// meta.noEligibleReviewer lets run-gates' review-retry keep the ORIGINAL unparseable result when
|
|
307
315
|
// the retry finds no second seat — a truthful cause beats a synthetic no-reviewer failure.
|
|
@@ -312,6 +320,8 @@ artifactDir) {
|
|
|
312
320
|
? { gate: "review", pass: false, details: `unreadable — ${reason}; set review.required:false to waive`, meta: { noEligibleReviewer: true, unreadable: true, ...(reviewerFloor ? { reviewerFloor } : {}) } }
|
|
313
321
|
: { gate: "review", pass: true, details: `WARNING: ${reason} — review waived by config`, meta: { noEligibleReviewer: true, ...(reviewerFloor ? { reviewerFloor } : {}) } };
|
|
314
322
|
}
|
|
323
|
+
reviewHistory?.push(channelKey(reviewer));
|
|
324
|
+
const rotationMeta = rotationSeat === undefined ? {} : { rotationSeat };
|
|
315
325
|
const measuredDiff = await fetchTaskDiff(worktree, baseRef, task.files);
|
|
316
326
|
// Keep the reader payload identical to the text charged to the strict cap:
|
|
317
327
|
// whole-file source deletions are represented by their citable operation fact.
|
|
@@ -362,9 +372,8 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
362
372
|
// frontier reviewers routinely need >5min on a configured-cap-sized diff, and `claude -p` buffers all
|
|
363
373
|
// output until completion — runLlm's 300s default killed reviews mid-flight, returning empty
|
|
364
374
|
// stdout that read as "unparseable" and escalated to re-implementation of green code
|
|
365
|
-
// (run-20260709-104447 P87-09).
|
|
366
|
-
|
|
367
|
-
900_000);
|
|
375
|
+
// (run-20260709-104447 P87-09). The configured ceiling defaults to that measured 15 minutes.
|
|
376
|
+
cfg.review.timeoutMs);
|
|
368
377
|
const raw = llm.output;
|
|
369
378
|
const provider = modelProvider(reviewer.model, reviewer.vendor);
|
|
370
379
|
const v = extractVerdictJson(raw, nonce);
|
|
@@ -393,15 +402,17 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
393
402
|
return {
|
|
394
403
|
gate: "review",
|
|
395
404
|
pass: false,
|
|
396
|
-
details: `${failure} (reviewer ${reviewer.adapter}:${reviewer.model}; vendor: ${reviewer.vendor}; provider: ${provider}; cause: ${cause}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
|
|
405
|
+
details: `${failure} (reviewer ${reviewer.adapter}:${reviewer.model}; vendor: ${reviewer.vendor}; provider: ${provider}; cause: ${cause}${cause === "timeout" ? `; killed at configured review timeout ${cfg.review.timeoutMs}ms` : ""}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
|
|
397
406
|
meta: {
|
|
398
407
|
...policyMeta,
|
|
408
|
+
...rotationMeta,
|
|
399
409
|
reviewer: channelKey(reviewer),
|
|
400
410
|
vendor: reviewer.vendor,
|
|
401
411
|
provider,
|
|
402
412
|
unparseable: true,
|
|
403
413
|
cause,
|
|
404
414
|
...(cause === "empty-output" ? { bytes } : {}),
|
|
415
|
+
...(cause === "timeout" ? { timeoutMs: cfg.review.timeoutMs } : {}),
|
|
405
416
|
...(concludedOnInactivity ? { classification: "infra", infra: true } : {}),
|
|
406
417
|
},
|
|
407
418
|
};
|
|
@@ -414,6 +425,6 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
414
425
|
gate: "review",
|
|
415
426
|
pass: decided.pass,
|
|
416
427
|
details: appendAnchoredReview(prose, v),
|
|
417
|
-
meta: { ...policyMeta, reviewer: channelKey(reviewer), vendor: reviewer.vendor, provider },
|
|
428
|
+
meta: { ...policyMeta, ...rotationMeta, reviewer: channelKey(reviewer), vendor: reviewer.vendor, provider },
|
|
418
429
|
};
|
|
419
430
|
}
|
|
@@ -13,13 +13,15 @@ export declare function resetLoadProviderForTests(): void;
|
|
|
13
13
|
* intervals and nothing between them, so the composite `test` gate (a selected screen, then other
|
|
14
14
|
* gates, then the full suite) reports the two suites' cost rather than the span containing them —
|
|
15
15
|
* and no consumer has to re-derive a duration by subtracting journal timestamps, which measures the
|
|
16
|
-
* queue as well as the work.
|
|
17
|
-
*
|
|
16
|
+
* queue as well as the work. Load is sampled at each interval's endpoints and every second within it;
|
|
17
|
+
* start preserves the scheduling input while max and mean retain sustained interior saturation.
|
|
18
18
|
*/
|
|
19
19
|
export interface GateTelemetry {
|
|
20
20
|
durationMs: number;
|
|
21
21
|
load1Start: number;
|
|
22
22
|
load1End: number;
|
|
23
|
+
load1Max: number;
|
|
24
|
+
load1Mean: number;
|
|
23
25
|
}
|
|
24
26
|
export type GateEvent = {
|
|
25
27
|
phase: "start";
|
|
@@ -51,6 +53,7 @@ export interface GateContext {
|
|
|
51
53
|
cfg: TickmarkrConfig;
|
|
52
54
|
via?: GateVia;
|
|
53
55
|
excludeReviewers?: string[];
|
|
56
|
+
reviewHistory?: string[];
|
|
54
57
|
artifactDir?: string;
|
|
55
58
|
pipeline?: "v185" | "legacy";
|
|
56
59
|
selectTests?: boolean;
|
package/dist/gates/run-gates.js
CHANGED
|
@@ -208,24 +208,42 @@ export async function runGates(task, ctx) {
|
|
|
208
208
|
// executing is added HERE, at the call site that runs it, so a gate that runs twice (the test
|
|
209
209
|
// gate's screen and its full suite) sums to its own cost and never to the span between them.
|
|
210
210
|
const spans = new Map();
|
|
211
|
+
const loadSamples = new Map();
|
|
211
212
|
// The test gate's two halves, kept apart as well as summed: `durationMs` alone cannot say whether
|
|
212
213
|
// a slow round was a slow subset or a slow full suite, and the parked scheduler's threshold is
|
|
213
214
|
// defined over the full-suite cost.
|
|
214
215
|
let selectedDurationMs;
|
|
215
216
|
let fullDurationMs;
|
|
216
|
-
const
|
|
217
|
+
const startMeasurement = () => {
|
|
217
218
|
const at = Date.now();
|
|
218
|
-
const
|
|
219
|
+
const samples = [loadProvider()];
|
|
220
|
+
const timer = setInterval(() => samples.push(loadProvider()), 1_000);
|
|
221
|
+
timer.unref();
|
|
222
|
+
return () => {
|
|
223
|
+
clearInterval(timer);
|
|
224
|
+
samples.push(loadProvider());
|
|
225
|
+
return { durationMs: Date.now() - at, samples };
|
|
226
|
+
};
|
|
227
|
+
};
|
|
228
|
+
const addMeasurement = (gate, measured) => {
|
|
229
|
+
const prior = spans.get(gate);
|
|
230
|
+
const samples = [...(loadSamples.get(gate) ?? []), ...measured.samples];
|
|
231
|
+
loadSamples.set(gate, samples);
|
|
232
|
+
spans.set(gate, {
|
|
233
|
+
durationMs: (prior?.durationMs ?? 0) + measured.durationMs,
|
|
234
|
+
load1Start: samples[0],
|
|
235
|
+
load1End: samples[samples.length - 1],
|
|
236
|
+
load1Max: Math.max(...samples),
|
|
237
|
+
load1Mean: samples.reduce((sum, value) => sum + value, 0) / samples.length,
|
|
238
|
+
});
|
|
239
|
+
};
|
|
240
|
+
const measure = async (gate, run) => {
|
|
241
|
+
const finish = startMeasurement();
|
|
219
242
|
try {
|
|
220
243
|
return await run();
|
|
221
244
|
}
|
|
222
245
|
finally {
|
|
223
|
-
|
|
224
|
-
spans.set(gate, {
|
|
225
|
-
durationMs: (prior?.durationMs ?? 0) + (Date.now() - at),
|
|
226
|
-
load1Start: prior?.load1Start ?? load1Start,
|
|
227
|
-
load1End: loadProvider(),
|
|
228
|
-
});
|
|
246
|
+
addMeasurement(gate, finish());
|
|
229
247
|
}
|
|
230
248
|
};
|
|
231
249
|
// The measurement is attached at the ONE seam every result leaves this function through, so a
|
|
@@ -347,12 +365,11 @@ export async function runGates(task, ctx) {
|
|
|
347
365
|
// suppresses them anyway; split compareToBaseline only if a tool gate ever gets slow.
|
|
348
366
|
// ponytail: legacy runs adjacent tools in ONE compareToBaseline call, so there is one interval
|
|
349
367
|
// to measure and each of its gates carries it. Split it only if this branch ever stops batching.
|
|
350
|
-
const
|
|
351
|
-
const batchLoadStart = loadProvider();
|
|
368
|
+
const finish = startMeasurement();
|
|
352
369
|
const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [...gates]);
|
|
353
|
-
const batch =
|
|
370
|
+
const batch = finish();
|
|
354
371
|
for (const g of gates)
|
|
355
|
-
|
|
372
|
+
addMeasurement(g, batch);
|
|
356
373
|
// The same refusal AFTER the commands, because a green command can dirty the tree the check
|
|
357
374
|
// above just proved clean. Batched, legacy cannot say WHICH command did it, so the refusal
|
|
358
375
|
// lands on the last gate that had one — the round dies there either way. A red battery is
|
|
@@ -575,7 +592,7 @@ export async function runGates(task, ctx) {
|
|
|
575
592
|
invocations.push(...captured.invocations);
|
|
576
593
|
return captured.value;
|
|
577
594
|
};
|
|
578
|
-
let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir));
|
|
595
|
+
let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir, ctx.reviewHistory));
|
|
579
596
|
// OBS-193/574: an unparseable review verdict retries the REVIEW exactly once, preferring a
|
|
580
597
|
// different adapter. Only a single-adapter eligible pool may fall back to another channel on the
|
|
581
598
|
// flaked adapter. The flaked verdict never enters results; an exhausted pool preserves its cause.
|
|
@@ -598,7 +615,7 @@ export async function runGates(task, ctx) {
|
|
|
598
615
|
const crossAdapter = pickReviewer(ctx.author, ctx.channels, [...priorExclusions, ...adapterExclusions], ctx.cfg.review.prefer ?? [], task.routingHints?.floor);
|
|
599
616
|
const exclusion = crossAdapter ? "adapter" : "channel";
|
|
600
617
|
const retryExclusions = [...priorExclusions, ...(crossAdapter ? adapterExclusions : [flaked])];
|
|
601
|
-
const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, retryExclusions, ctx.artifactDir));
|
|
618
|
+
const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, retryExclusions, ctx.artifactDir, ctx.reviewHistory));
|
|
602
619
|
if (second.meta?.noEligibleReviewer !== true) {
|
|
603
620
|
const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
|
|
604
621
|
const route = exclusion === "adapter"
|
|
@@ -630,11 +647,11 @@ export async function runGates(task, ctx) {
|
|
|
630
647
|
// The check runs BEFORE any gate, so on a clean tree it belongs to no gate: charging every round's
|
|
631
648
|
// first gate for it would inflate the one measurement the parked recalibrations key on. It becomes
|
|
632
649
|
// that gate's interval only on the path where it IS what the gate did — the refusal below.
|
|
633
|
-
const
|
|
634
|
-
const entryLoad = loadProvider();
|
|
650
|
+
const finishEntry = startMeasurement();
|
|
635
651
|
const entryDirt = sequence.length ? await dirtyWorktree() : undefined;
|
|
652
|
+
const entryMeasurement = finishEntry();
|
|
636
653
|
if (entryDirt) {
|
|
637
|
-
|
|
654
|
+
addMeasurement(sequence[0], entryMeasurement);
|
|
638
655
|
await emitStart(sequence[0]);
|
|
639
656
|
await record(dirtyRefusal(sequence[0], entryDirt));
|
|
640
657
|
return done();
|
|
@@ -5,6 +5,10 @@ export interface Disallowed {
|
|
|
5
5
|
entry: string;
|
|
6
6
|
}
|
|
7
7
|
export type PreferenceRole = "worker" | "judge" | "review" | "consult";
|
|
8
|
+
export declare function routingModelProvider(model: string, fallback?: string): string;
|
|
9
|
+
export declare const modelRouteIdentity: (model: string, fallback?: string) => string;
|
|
10
|
+
export declare const channelRouteIdentity: (key: string, fallback?: string) => string;
|
|
11
|
+
export declare function routingEntrySeatLines(cfg: TickmarkrConfig): string[];
|
|
8
12
|
export declare function excludedChannels(cfg: TickmarkrConfig, adapters: {
|
|
9
13
|
id: string;
|
|
10
14
|
}[] | string[], health: Record<string, AuthHealth>): {
|
package/dist/route/preference.js
CHANGED
|
@@ -1,6 +1,46 @@
|
|
|
1
1
|
import { channelKey, channelsFromConfig } from "../adapters/types.js";
|
|
2
2
|
import { validateGraph } from "../graph/schema.js";
|
|
3
3
|
import { route, RoutingError } from "./router.js";
|
|
4
|
+
const PREFERENCE_ROLES = ["worker", "judge", "review", "consult"];
|
|
5
|
+
// Keep routing retries on the same identity review diversity uses: the served provider plus the
|
|
6
|
+
// unprefixed model id. Gate review owns the original modelProvider policy; this dependency-leaf copy
|
|
7
|
+
// avoids importing gates back into route/run and must move with that helper when the scope permits.
|
|
8
|
+
export function routingModelProvider(model, fallback = "unknown") {
|
|
9
|
+
const id = model.toLowerCase();
|
|
10
|
+
const prefix = id.includes("/") ? id.slice(0, id.indexOf("/")) : "";
|
|
11
|
+
if (prefix === "openai" || prefix === "openai-codex" || /^(?:gpt|o\d)/.test(id))
|
|
12
|
+
return "openai";
|
|
13
|
+
if (prefix === "anthropic" || /^(?:claude|opus|sonnet|haiku|fable)(?:-|$)/.test(id))
|
|
14
|
+
return "anthropic";
|
|
15
|
+
if (prefix === "google" || /^gemini(?:-|$)/.test(id))
|
|
16
|
+
return "google";
|
|
17
|
+
if (prefix === "xai" || /^grok(?:-|$)/.test(id))
|
|
18
|
+
return "xai";
|
|
19
|
+
if (["zai", "zhipu", "zai-coding-plan"].includes(prefix) || /^glm(?:-|$)/.test(id))
|
|
20
|
+
return "zhipu";
|
|
21
|
+
if (["kimi-code", "moonshot"].includes(prefix) || /^kimi(?:-|$)/.test(id))
|
|
22
|
+
return "moonshot";
|
|
23
|
+
return fallback;
|
|
24
|
+
}
|
|
25
|
+
export const modelRouteIdentity = (model, fallback = "unknown") => `${routingModelProvider(model, fallback)}/${model.slice(model.lastIndexOf("/") + 1).toLowerCase()}`;
|
|
26
|
+
export const channelRouteIdentity = (key, fallback = "unknown") => {
|
|
27
|
+
const i = key.indexOf(":");
|
|
28
|
+
return i < 0 ? key : modelRouteIdentity(key.slice(i + 1), fallback);
|
|
29
|
+
};
|
|
30
|
+
export function routingEntrySeatLines(cfg) {
|
|
31
|
+
const lines = [];
|
|
32
|
+
const add = (path, entries, roles) => {
|
|
33
|
+
for (const entry of entries ?? [])
|
|
34
|
+
lines.push(`${path} '${entry}' reaches seats: ${roles.join(", ")}`);
|
|
35
|
+
};
|
|
36
|
+
add("routing.allow.adapters", cfg.routing.allow?.adapters, PREFERENCE_ROLES);
|
|
37
|
+
add("routing.allow.models", cfg.routing.allow?.models, PREFERENCE_ROLES);
|
|
38
|
+
add("routing.deny.adapters", cfg.routing.deny?.adapters, PREFERENCE_ROLES);
|
|
39
|
+
add("routing.deny.models", cfg.routing.deny?.models, PREFERENCE_ROLES);
|
|
40
|
+
add("routing.deny.workers.adapters", cfg.routing.deny?.workers?.adapters, ["worker"]);
|
|
41
|
+
add("routing.deny.workers.models", cfg.routing.deny?.workers?.models, ["worker"]);
|
|
42
|
+
return lines;
|
|
43
|
+
}
|
|
4
44
|
const adapterIds = (adapters) => typeof adapters[0] === "string" ? adapters : adapters.map((a) => a.id);
|
|
5
45
|
export function excludedChannels(cfg, adapters, health) {
|
|
6
46
|
const { allow, deny } = cfg.routing;
|
package/dist/route/router.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { channelKey, channelsFromConfig } from "../adapters/types.js";
|
|
2
2
|
import { TIER_RANK } from "../config/config.js";
|
|
3
|
-
import { disallowedBy } from "./preference.js";
|
|
3
|
+
import { channelRouteIdentity, disallowedBy, modelRouteIdentity, routingModelProvider } from "./preference.js";
|
|
4
4
|
import { cellOf, EXPLORE_CAP, explorationBonus, learnedScore, MIN_SAMPLES } from "./profile.js";
|
|
5
5
|
export const NO_EXPLORE_ENV = "TICKMARKR_NO_EXPLORE";
|
|
6
6
|
// OBS-89 (v1.60): the TICKMARKR_QUALITY variable is RETIRED — nothing in src reads it anymore and
|
|
@@ -337,7 +337,20 @@ export function nextChannel(current, task, cfg, channels, tried, profile, exclud
|
|
|
337
337
|
// profile-dependent filter. NO exploration bonus here (route():110 has one; a probe on the
|
|
338
338
|
// failure path would spend a real retry). Absent profile ⇒ every score is 0 ⇒ third key
|
|
339
339
|
// all-ties ⇒ the stable sort preserves the exact v1.7 candidate ORDER.
|
|
340
|
-
const
|
|
340
|
+
const triedKeys = new Set(tried);
|
|
341
|
+
const triedIdentities = new Set(tried.map((key) => {
|
|
342
|
+
const channel = channels.find((c) => channelKey(c) === key);
|
|
343
|
+
return channel ? modelRouteIdentity(channel.model, channel.vendor) : channelRouteIdentity(key);
|
|
344
|
+
}));
|
|
345
|
+
// excludeAdapter is expanded by the daemon into every channel key of the failed adapter. Once that
|
|
346
|
+
// complete set is present, the outage follows the current served provider across gateway aliases.
|
|
347
|
+
const currentAdapterExcluded = channels.some((c) => c.adapter === current.adapter)
|
|
348
|
+
&& channels.filter((c) => c.adapter === current.adapter).every((c) => triedKeys.has(channelKey(c)));
|
|
349
|
+
const currentChannel = channels.find((c) => c.adapter === current.adapter && c.model === current.model);
|
|
350
|
+
const excludedProvider = currentAdapterExcluded ? routingModelProvider(current.model, currentChannel?.vendor) : undefined;
|
|
351
|
+
const pool = channels.filter((c) => !triedIdentities.has(modelRouteIdentity(c.model, c.vendor))
|
|
352
|
+
&& (!excludedProvider || routingModelProvider(c.model, c.vendor) !== excludedProvider)
|
|
353
|
+
&& TIER_RANK[c.tier] >= TIER_RANK[current.tier]);
|
|
341
354
|
const scores = new Map(pool.map((c) => [channelKey(c), profile ? learnedScore(profile, task.shape, channelKey(c), c.channel, { availWeight: cfg.routing.learnedTuning?.availWeight }) : 0]));
|
|
342
355
|
const scoreOf = (c) => scores.get(channelKey(c));
|
|
343
356
|
const candidates = pool.sort((a, b) => TIER_RANK[a.tier] - TIER_RANK[b.tier] || marginalCostRank(a) - marginalCostRank(b) || scoreOf(b) - scoreOf(a));
|
package/dist/run/consult.d.ts
CHANGED
package/dist/run/consult.js
CHANGED
|
@@ -4,7 +4,7 @@ import { getAdapter } from "../adapters/registry.js";
|
|
|
4
4
|
import { bannerShell, paneDispatchCommand } from "../brand.js";
|
|
5
5
|
import { dewrapPaneVerdict, extractVerdictJson, gateExitTrailer, gatePaneName, generateVerdictNonce, verdictNonceLine } from "../gates/llm.js";
|
|
6
6
|
import { classifyVerdictCause } from "../gates/verdict-cause.js";
|
|
7
|
-
import { disallowedBy } from "../route/preference.js";
|
|
7
|
+
import { disallowedBy, routingModelProvider } from "../route/preference.js";
|
|
8
8
|
import { sh } from "./git.js";
|
|
9
9
|
import { redactSecrets } from "./redact.js";
|
|
10
10
|
import { filterLlmTranscript } from "./stall.js";
|
|
@@ -59,6 +59,22 @@ export function augmentRetryBrief(feedback, opts) {
|
|
|
59
59
|
return parts.join("\n\n");
|
|
60
60
|
}
|
|
61
61
|
const ACTIONS = ["retry", "reroute", "decompose", "human"];
|
|
62
|
+
function excludedProviderFromDossier(d, adapter) {
|
|
63
|
+
try {
|
|
64
|
+
const events = JSON.parse(d.journalTail);
|
|
65
|
+
const assignment = [...events].reverse().find((event) => event.event === "task-dispatch"
|
|
66
|
+
&& event.data?.assignment?.adapter === adapter
|
|
67
|
+
&& typeof event.data.assignment.model === "string")?.data?.assignment;
|
|
68
|
+
if (typeof assignment?.model !== "string")
|
|
69
|
+
return undefined;
|
|
70
|
+
const provider = routingModelProvider(assignment.model);
|
|
71
|
+
return provider === "unknown" ? undefined : provider;
|
|
72
|
+
}
|
|
73
|
+
catch {
|
|
74
|
+
// Legacy/non-JSON dossier tails retain the adapter exclusion without inventing a provider.
|
|
75
|
+
return undefined;
|
|
76
|
+
}
|
|
77
|
+
}
|
|
62
78
|
export function parseConsultVerdict(out, nonce) {
|
|
63
79
|
const v = extractVerdictJson(out, nonce);
|
|
64
80
|
if (!v)
|
|
@@ -110,10 +126,10 @@ ${d.journalTail}
|
|
|
110
126
|
Verdict meanings: retry = same assignment with your notes as feedback; reroute = different CLI/model;
|
|
111
127
|
decompose = task too big, needs human re-planning; human = a person must look at this.
|
|
112
128
|
|
|
113
|
-
On reroute only, optional excludeAdapter is
|
|
114
|
-
|
|
115
|
-
trust dialog, broken install) — not when a
|
|
116
|
-
reroutes so other models
|
|
129
|
+
On reroute only, optional excludeAdapter is the failed adapter id (e.g. "cursor-agent"). Tickmarkr
|
|
130
|
+
resolves the adapter's current model to its provider and bans that provider for this task. Use it for
|
|
131
|
+
environmental/provider failures ("the CLI is blocked", trust dialog, broken install) — not when a
|
|
132
|
+
single model produced bad code. Omit for model-level reroutes so other models remain eligible.
|
|
117
133
|
|
|
118
134
|
${verdictNonceLine(nonce)}
|
|
119
135
|
|
|
@@ -230,8 +246,19 @@ opts = {}) {
|
|
|
230
246
|
for (const [i, seat] of allowedSeats.entries()) {
|
|
231
247
|
try {
|
|
232
248
|
const parsed = await invokeSeat(seat.adapter, seat.model, i);
|
|
233
|
-
if (parsed.verdict)
|
|
234
|
-
|
|
249
|
+
if (parsed.verdict) {
|
|
250
|
+
const excludeProvider = parsed.verdict.excludeAdapter
|
|
251
|
+
? excludedProviderFromDossier(d, parsed.verdict.excludeAdapter)
|
|
252
|
+
: undefined;
|
|
253
|
+
return {
|
|
254
|
+
...parsed.verdict,
|
|
255
|
+
...(excludeProvider ? {
|
|
256
|
+
excludeProvider,
|
|
257
|
+
notes: `${parsed.verdict.notes} — excluded provider ${excludeProvider}`,
|
|
258
|
+
} : {}),
|
|
259
|
+
...seatIdentity(seat),
|
|
260
|
+
};
|
|
261
|
+
}
|
|
235
262
|
}
|
|
236
263
|
catch {
|
|
237
264
|
// failed seat (unknown adapter, dead driver/pane, shell error) — fall to the next entry
|
package/dist/run/daemon.d.ts
CHANGED
|
@@ -68,6 +68,7 @@ export declare function formatSummary(s: RunSummary): string;
|
|
|
68
68
|
* run id (cli/commands/status.ts positionalRunId), so naming it here is what stops the board from
|
|
69
69
|
* following the newest journal in a repo that already carries a second, newer run — a board showing
|
|
70
70
|
* the wrong run is a recorded incident (skills/tickmarkr-overseer/SKILL.md). */
|
|
71
|
+
export declare const daemonEntrypoint: string;
|
|
71
72
|
export declare const watchCommand: (runId: string) => string;
|
|
72
73
|
/**
|
|
73
74
|
* R3 (OBS-186): a gate that DECLINED to run is not a gate that failed. The review gate's skip branch
|
|
@@ -103,6 +104,9 @@ export declare const gateSatisfied: (g: GateResult) => boolean;
|
|
|
103
104
|
*/
|
|
104
105
|
export declare function decisiveReviewRounds(events: JournalEvent[]): JournalEvent[];
|
|
105
106
|
export declare const SUITE_POLL_MS = 250;
|
|
107
|
+
export declare const SUITE_WAIT_CEILING_MS = 600000;
|
|
108
|
+
export declare const setSuiteWaitCeilingForTests: (ms: number) => void;
|
|
109
|
+
export declare const resetSuiteWaitCeilingForTests: () => void;
|
|
106
110
|
export declare const APPROVAL_POLL_MS = 250;
|
|
107
111
|
export declare const EARLY_LAUNCH_LIVENESS_MS = 60000;
|
|
108
112
|
/** Test seam — lowers the empty-pane liveness window without sleeping 60s per case. */
|