tickmarkr 2.3.0 → 2.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/catalog-remote.d.ts +12 -4
- package/dist/adapters/catalog-remote.js +97 -45
- package/dist/adapters/catalog.js +5 -3
- package/dist/adapters/claude-code.d.ts +1 -1
- package/dist/adapters/claude-code.js +8 -5
- package/dist/adapters/codex.js +6 -7
- package/dist/adapters/model-lints.d.ts +9 -5
- package/dist/adapters/model-lints.js +56 -15
- package/dist/adapters/model-windows.js +11 -0
- package/dist/adapters/prompt.js +1 -0
- package/dist/adapters/qwen.d.ts +5 -0
- package/dist/adapters/qwen.js +153 -0
- package/dist/adapters/registry.js +13 -1
- package/dist/adapters/types.d.ts +1 -0
- package/dist/adapters/types.js +1 -0
- package/dist/cli/commands/compile.d.ts +3 -0
- package/dist/cli/commands/compile.js +91 -34
- package/dist/cli/commands/doctor.d.ts +4 -3
- package/dist/cli/commands/doctor.js +28 -9
- package/dist/cli/commands/fleet.d.ts +4 -0
- package/dist/cli/commands/fleet.js +60 -15
- package/dist/cli/commands/init.js +12 -13
- package/dist/cli/commands/plan.js +75 -11
- package/dist/cli/commands/run.js +20 -1
- package/dist/cli/commands/status.d.ts +1 -0
- package/dist/cli/commands/status.js +59 -16
- package/dist/cli/commands/verify.d.ts +1 -0
- package/dist/cli/commands/verify.js +5 -0
- package/dist/cli/commands/version.js +2 -2
- package/dist/cli/index.d.ts +1 -1
- package/dist/cli/index.js +2 -2
- package/dist/compile/collateral.d.ts +14 -12
- package/dist/compile/collateral.js +32 -33
- package/dist/compile/index.d.ts +4 -1
- package/dist/compile/index.js +53 -8
- package/dist/compile/native.d.ts +4 -2
- package/dist/compile/native.js +63 -6
- package/dist/compile/ownership.js +41 -10
- package/dist/config/config.d.ts +1 -0
- package/dist/config/config.js +51 -5
- package/dist/drivers/herdr.d.ts +2 -0
- package/dist/drivers/herdr.js +43 -4
- package/dist/drivers/orca.d.ts +18 -1
- package/dist/drivers/orca.js +163 -15
- package/dist/drivers/types.d.ts +10 -0
- package/dist/gates/baseline.d.ts +26 -2
- package/dist/gates/baseline.js +115 -13
- package/dist/gates/review.d.ts +6 -4
- package/dist/gates/review.js +26 -31
- package/dist/gates/run-gates.d.ts +5 -2
- package/dist/gates/run-gates.js +34 -17
- package/dist/graph/graph.d.ts +20 -0
- package/dist/graph/graph.js +66 -1
- package/dist/route/preference.d.ts +6 -0
- package/dist/route/preference.js +40 -0
- package/dist/route/router.js +15 -2
- package/dist/run/consult.d.ts +1 -0
- package/dist/run/consult.js +35 -7
- package/dist/run/daemon.d.ts +9 -0
- package/dist/run/daemon.js +266 -59
- package/dist/run/git.d.ts +4 -0
- package/dist/run/git.js +51 -6
- package/dist/run/journal.d.ts +1 -1
- package/dist/run/journal.js +5 -2
- package/dist/run/lock.d.ts +6 -0
- package/dist/run/lock.js +41 -1
- package/dist/tui/ink/fleet-app.d.ts +4 -0
- package/dist/tui/ink/fleet-app.js +45 -16
- package/package.json +59 -1
- package/skills/tickmarkr-overseer/SKILL.md +39 -4
- package/skills/tickmarkr-overseer/scripts/grade-ci.sh +79 -0
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +90 -4
package/dist/gates/review.js
CHANGED
|
@@ -8,6 +8,7 @@ import { getAdapter } from "../adapters/registry.js";
|
|
|
8
8
|
import { shOk } from "../run/git.js";
|
|
9
9
|
import { redactSecrets } from "../run/redact.js";
|
|
10
10
|
import { marginalCostRank } from "../route/router.js";
|
|
11
|
+
import { modelProvider } from "../route/preference.js";
|
|
11
12
|
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlmDetailed, verdictNonceLine } from "./llm.js";
|
|
12
13
|
import { classifyVerdictCause } from "./verdict-cause.js";
|
|
13
14
|
import { captureDiffCapFor, measureArtifactDiff, reviewableLogicDiff, } from "./artifact-manifest.js";
|
|
@@ -165,24 +166,7 @@ export function diffCapParkReason(results) {
|
|
|
165
166
|
export function modelId(model) {
|
|
166
167
|
return model.slice(model.lastIndexOf("/") + 1);
|
|
167
168
|
}
|
|
168
|
-
|
|
169
|
-
export function modelProvider(model, fallback = "unknown") {
|
|
170
|
-
const id = model.toLowerCase();
|
|
171
|
-
const prefix = id.includes("/") ? id.slice(0, id.indexOf("/")) : "";
|
|
172
|
-
if (prefix === "openai" || prefix === "openai-codex" || /^(?:gpt|o\d)/.test(id))
|
|
173
|
-
return "openai";
|
|
174
|
-
if (prefix === "anthropic" || /^(?:claude|opus|sonnet|haiku|fable)(?:-|$)/.test(id))
|
|
175
|
-
return "anthropic";
|
|
176
|
-
if (prefix === "google" || /^gemini(?:-|$)/.test(id))
|
|
177
|
-
return "google";
|
|
178
|
-
if (prefix === "xai" || /^grok(?:-|$)/.test(id))
|
|
179
|
-
return "xai";
|
|
180
|
-
if (["zai", "zhipu", "zai-coding-plan"].includes(prefix) || /^glm(?:-|$)/.test(id))
|
|
181
|
-
return "zhipu";
|
|
182
|
-
if (["kimi-code", "moonshot"].includes(prefix) || /^kimi(?:-|$)/.test(id))
|
|
183
|
-
return "moonshot";
|
|
184
|
-
return fallback;
|
|
185
|
-
}
|
|
169
|
+
export { modelProvider };
|
|
186
170
|
// v1.53 T2: same entry grammar as routing.map.prefer (router.ts preferIndex — router is out of this
|
|
187
171
|
// module's dependency direction for a private fn, so the 3 lines live here too): `adapter` matches
|
|
188
172
|
// every channel of that adapter, `adapter:model` exactly one; unmatched channels sort after all entries.
|
|
@@ -192,7 +176,9 @@ function reviewPreferIndex(c, prefer) {
|
|
|
192
176
|
}
|
|
193
177
|
export function pickReviewer(author, channels, exclude = [], // v1.1 failover: reviewer channels that already produced garbage for this task
|
|
194
178
|
prefer = [], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
|
|
195
|
-
floor
|
|
179
|
+
floor, // task-declared only; config floors govern workers and must not silently move review seats
|
|
180
|
+
history = [], // run-scoped picks, oldest to newest; empty preserves the established ranking
|
|
181
|
+
onSeat) {
|
|
196
182
|
// FLEET-05 success criterion 2: an author not resolvable in the channel list yields NO reviewer.
|
|
197
183
|
// The old `?? author.adapter` fallback compared an adapter id to vendor names, matched nothing, and
|
|
198
184
|
// admitted every reviewer — including the author's own channel (fail-OPEN). null lands on reviewGate's
|
|
@@ -200,18 +186,23 @@ floor) {
|
|
|
200
186
|
const authorChannel = channels.find((c) => c.adapter === author.adapter && c.model === author.model);
|
|
201
187
|
if (!authorChannel)
|
|
202
188
|
return null;
|
|
203
|
-
// Failover also guards true provider; the initial pick keeps the established stamped-vendor contract.
|
|
204
189
|
const authorProvider = modelProvider(author.model, authorChannel.vendor);
|
|
205
|
-
|
|
190
|
+
const ranked = channels
|
|
206
191
|
// two independent axes: different vendor AND different base-model identity (ADDED TO the vendor
|
|
207
|
-
// rule, never replacing it — a future edit can't silently drop either).
|
|
208
|
-
//
|
|
192
|
+
// rule, never replacing it — a future edit can't silently drop either). Failover additionally guards
|
|
193
|
+
// true provider identity; the initial pick keeps the established stamped-vendor contract. The diversity
|
|
194
|
+
// filter runs BEFORE preference ranking, so prefer cannot resurrect an excluded channel.
|
|
209
195
|
.filter((c) => c.vendor !== authorChannel.vendor
|
|
210
196
|
&& (exclude.length === 0 || modelProvider(c.model, c.vendor) !== authorProvider)
|
|
211
197
|
&& modelId(c.model) !== modelId(author.model)
|
|
212
198
|
&& !exclude.includes(channelKey(c))
|
|
213
199
|
&& (floor === undefined || TIER_RANK[c.tier] >= TIER_RANK[floor]))
|
|
214
|
-
.sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b))
|
|
200
|
+
.sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b));
|
|
201
|
+
const reviewer = [...ranked].sort((a, b) => history.lastIndexOf(channelKey(a)) - history.lastIndexOf(channelKey(b))
|
|
202
|
+
|| ranked.indexOf(a) - ranked.indexOf(b))[0] ?? null;
|
|
203
|
+
if (reviewer)
|
|
204
|
+
onSeat?.(ranked.indexOf(reviewer) + 1);
|
|
205
|
+
return reviewer;
|
|
215
206
|
}
|
|
216
207
|
/**
|
|
217
208
|
* This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
|
|
@@ -229,7 +220,7 @@ ${files.map((path) => `- ${path}`).join("\n")}`;
|
|
|
229
220
|
export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
|
|
230
221
|
// OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
|
|
231
222
|
// direct tests) skips persistence and changes nothing else.
|
|
232
|
-
artifactDir) {
|
|
223
|
+
artifactDir, reviewHistory) {
|
|
233
224
|
// R3 (OBS-186): participation is keyed on PATHS. The compiler's assignment comes from the DECLARED
|
|
234
225
|
// files[]; the operator's floor may RAISE it to full and can never lower it. `complexityThreshold` is
|
|
235
226
|
// retired — the branch that returned a green skip on a complexity comparison is gone, and with it the
|
|
@@ -301,7 +292,8 @@ artifactDir) {
|
|
|
301
292
|
// A reviewer floor is opt-in at the task. Applying cfg.routing.floors here would change the
|
|
302
293
|
// historical seat for every task that never asked for review-tier coupling.
|
|
303
294
|
const reviewerFloor = task.routingHints?.floor;
|
|
304
|
-
|
|
295
|
+
let rotationSeat;
|
|
296
|
+
const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor, reviewHistory, reviewHistory ? (seat) => { rotationSeat = seat; } : undefined);
|
|
305
297
|
if (!reviewer) {
|
|
306
298
|
// meta.noEligibleReviewer lets run-gates' review-retry keep the ORIGINAL unparseable result when
|
|
307
299
|
// the retry finds no second seat — a truthful cause beats a synthetic no-reviewer failure.
|
|
@@ -312,6 +304,8 @@ artifactDir) {
|
|
|
312
304
|
? { gate: "review", pass: false, details: `unreadable — ${reason}; set review.required:false to waive`, meta: { noEligibleReviewer: true, unreadable: true, ...(reviewerFloor ? { reviewerFloor } : {}) } }
|
|
313
305
|
: { gate: "review", pass: true, details: `WARNING: ${reason} — review waived by config`, meta: { noEligibleReviewer: true, ...(reviewerFloor ? { reviewerFloor } : {}) } };
|
|
314
306
|
}
|
|
307
|
+
reviewHistory?.push(channelKey(reviewer));
|
|
308
|
+
const rotationMeta = rotationSeat === undefined ? {} : { rotationSeat };
|
|
315
309
|
const measuredDiff = await fetchTaskDiff(worktree, baseRef, task.files);
|
|
316
310
|
// Keep the reader payload identical to the text charged to the strict cap:
|
|
317
311
|
// whole-file source deletions are represented by their citable operation fact.
|
|
@@ -362,9 +356,8 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
362
356
|
// frontier reviewers routinely need >5min on a configured-cap-sized diff, and `claude -p` buffers all
|
|
363
357
|
// output until completion — runLlm's 300s default killed reviews mid-flight, returning empty
|
|
364
358
|
// stdout that read as "unparseable" and escalated to re-implementation of green code
|
|
365
|
-
// (run-20260709-104447 P87-09).
|
|
366
|
-
|
|
367
|
-
900_000);
|
|
359
|
+
// (run-20260709-104447 P87-09). The configured ceiling defaults to that measured 15 minutes.
|
|
360
|
+
cfg.review.timeoutMs);
|
|
368
361
|
const raw = llm.output;
|
|
369
362
|
const provider = modelProvider(reviewer.model, reviewer.vendor);
|
|
370
363
|
const v = extractVerdictJson(raw, nonce);
|
|
@@ -393,15 +386,17 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
393
386
|
return {
|
|
394
387
|
gate: "review",
|
|
395
388
|
pass: false,
|
|
396
|
-
details: `${failure} (reviewer ${reviewer.adapter}:${reviewer.model}; vendor: ${reviewer.vendor}; provider: ${provider}; cause: ${cause}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
|
|
389
|
+
details: `${failure} (reviewer ${reviewer.adapter}:${reviewer.model}; vendor: ${reviewer.vendor}; provider: ${provider}; cause: ${cause}${cause === "timeout" ? `; killed at configured review timeout ${cfg.review.timeoutMs}ms` : ""}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
|
|
397
390
|
meta: {
|
|
398
391
|
...policyMeta,
|
|
392
|
+
...rotationMeta,
|
|
399
393
|
reviewer: channelKey(reviewer),
|
|
400
394
|
vendor: reviewer.vendor,
|
|
401
395
|
provider,
|
|
402
396
|
unparseable: true,
|
|
403
397
|
cause,
|
|
404
398
|
...(cause === "empty-output" ? { bytes } : {}),
|
|
399
|
+
...(cause === "timeout" ? { timeoutMs: cfg.review.timeoutMs } : {}),
|
|
405
400
|
...(concludedOnInactivity ? { classification: "infra", infra: true } : {}),
|
|
406
401
|
},
|
|
407
402
|
};
|
|
@@ -414,6 +409,6 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
414
409
|
gate: "review",
|
|
415
410
|
pass: decided.pass,
|
|
416
411
|
details: appendAnchoredReview(prose, v),
|
|
417
|
-
meta: { ...policyMeta, reviewer: channelKey(reviewer), vendor: reviewer.vendor, provider },
|
|
412
|
+
meta: { ...policyMeta, ...rotationMeta, reviewer: channelKey(reviewer), vendor: reviewer.vendor, provider },
|
|
418
413
|
};
|
|
419
414
|
}
|
|
@@ -13,13 +13,15 @@ export declare function resetLoadProviderForTests(): void;
|
|
|
13
13
|
* intervals and nothing between them, so the composite `test` gate (a selected screen, then other
|
|
14
14
|
* gates, then the full suite) reports the two suites' cost rather than the span containing them —
|
|
15
15
|
* and no consumer has to re-derive a duration by subtracting journal timestamps, which measures the
|
|
16
|
-
* queue as well as the work.
|
|
17
|
-
*
|
|
16
|
+
* queue as well as the work. Load is sampled at each interval's endpoints and every second within it;
|
|
17
|
+
* start preserves the scheduling input while max and mean retain sustained interior saturation.
|
|
18
18
|
*/
|
|
19
19
|
export interface GateTelemetry {
|
|
20
20
|
durationMs: number;
|
|
21
21
|
load1Start: number;
|
|
22
22
|
load1End: number;
|
|
23
|
+
load1Max: number;
|
|
24
|
+
load1Mean: number;
|
|
23
25
|
}
|
|
24
26
|
export type GateEvent = {
|
|
25
27
|
phase: "start";
|
|
@@ -51,6 +53,7 @@ export interface GateContext {
|
|
|
51
53
|
cfg: TickmarkrConfig;
|
|
52
54
|
via?: GateVia;
|
|
53
55
|
excludeReviewers?: string[];
|
|
56
|
+
reviewHistory?: string[];
|
|
54
57
|
artifactDir?: string;
|
|
55
58
|
pipeline?: "v185" | "legacy";
|
|
56
59
|
selectTests?: boolean;
|
package/dist/gates/run-gates.js
CHANGED
|
@@ -208,24 +208,42 @@ export async function runGates(task, ctx) {
|
|
|
208
208
|
// executing is added HERE, at the call site that runs it, so a gate that runs twice (the test
|
|
209
209
|
// gate's screen and its full suite) sums to its own cost and never to the span between them.
|
|
210
210
|
const spans = new Map();
|
|
211
|
+
const loadSamples = new Map();
|
|
211
212
|
// The test gate's two halves, kept apart as well as summed: `durationMs` alone cannot say whether
|
|
212
213
|
// a slow round was a slow subset or a slow full suite, and the parked scheduler's threshold is
|
|
213
214
|
// defined over the full-suite cost.
|
|
214
215
|
let selectedDurationMs;
|
|
215
216
|
let fullDurationMs;
|
|
216
|
-
const
|
|
217
|
+
const startMeasurement = () => {
|
|
217
218
|
const at = Date.now();
|
|
218
|
-
const
|
|
219
|
+
const samples = [loadProvider()];
|
|
220
|
+
const timer = setInterval(() => samples.push(loadProvider()), 1_000);
|
|
221
|
+
timer.unref();
|
|
222
|
+
return () => {
|
|
223
|
+
clearInterval(timer);
|
|
224
|
+
samples.push(loadProvider());
|
|
225
|
+
return { durationMs: Date.now() - at, samples };
|
|
226
|
+
};
|
|
227
|
+
};
|
|
228
|
+
const addMeasurement = (gate, measured) => {
|
|
229
|
+
const prior = spans.get(gate);
|
|
230
|
+
const samples = [...(loadSamples.get(gate) ?? []), ...measured.samples];
|
|
231
|
+
loadSamples.set(gate, samples);
|
|
232
|
+
spans.set(gate, {
|
|
233
|
+
durationMs: (prior?.durationMs ?? 0) + measured.durationMs,
|
|
234
|
+
load1Start: samples[0],
|
|
235
|
+
load1End: samples[samples.length - 1],
|
|
236
|
+
load1Max: Math.max(...samples),
|
|
237
|
+
load1Mean: samples.reduce((sum, value) => sum + value, 0) / samples.length,
|
|
238
|
+
});
|
|
239
|
+
};
|
|
240
|
+
const measure = async (gate, run) => {
|
|
241
|
+
const finish = startMeasurement();
|
|
219
242
|
try {
|
|
220
243
|
return await run();
|
|
221
244
|
}
|
|
222
245
|
finally {
|
|
223
|
-
|
|
224
|
-
spans.set(gate, {
|
|
225
|
-
durationMs: (prior?.durationMs ?? 0) + (Date.now() - at),
|
|
226
|
-
load1Start: prior?.load1Start ?? load1Start,
|
|
227
|
-
load1End: loadProvider(),
|
|
228
|
-
});
|
|
246
|
+
addMeasurement(gate, finish());
|
|
229
247
|
}
|
|
230
248
|
};
|
|
231
249
|
// The measurement is attached at the ONE seam every result leaves this function through, so a
|
|
@@ -347,12 +365,11 @@ export async function runGates(task, ctx) {
|
|
|
347
365
|
// suppresses them anyway; split compareToBaseline only if a tool gate ever gets slow.
|
|
348
366
|
// ponytail: legacy runs adjacent tools in ONE compareToBaseline call, so there is one interval
|
|
349
367
|
// to measure and each of its gates carries it. Split it only if this branch ever stops batching.
|
|
350
|
-
const
|
|
351
|
-
const batchLoadStart = loadProvider();
|
|
368
|
+
const finish = startMeasurement();
|
|
352
369
|
const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [...gates]);
|
|
353
|
-
const batch =
|
|
370
|
+
const batch = finish();
|
|
354
371
|
for (const g of gates)
|
|
355
|
-
|
|
372
|
+
addMeasurement(g, batch);
|
|
356
373
|
// The same refusal AFTER the commands, because a green command can dirty the tree the check
|
|
357
374
|
// above just proved clean. Batched, legacy cannot say WHICH command did it, so the refusal
|
|
358
375
|
// lands on the last gate that had one — the round dies there either way. A red battery is
|
|
@@ -575,7 +592,7 @@ export async function runGates(task, ctx) {
|
|
|
575
592
|
invocations.push(...captured.invocations);
|
|
576
593
|
return captured.value;
|
|
577
594
|
};
|
|
578
|
-
let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir));
|
|
595
|
+
let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir, ctx.reviewHistory));
|
|
579
596
|
// OBS-193/574: an unparseable review verdict retries the REVIEW exactly once, preferring a
|
|
580
597
|
// different adapter. Only a single-adapter eligible pool may fall back to another channel on the
|
|
581
598
|
// flaked adapter. The flaked verdict never enters results; an exhausted pool preserves its cause.
|
|
@@ -598,7 +615,7 @@ export async function runGates(task, ctx) {
|
|
|
598
615
|
const crossAdapter = pickReviewer(ctx.author, ctx.channels, [...priorExclusions, ...adapterExclusions], ctx.cfg.review.prefer ?? [], task.routingHints?.floor);
|
|
599
616
|
const exclusion = crossAdapter ? "adapter" : "channel";
|
|
600
617
|
const retryExclusions = [...priorExclusions, ...(crossAdapter ? adapterExclusions : [flaked])];
|
|
601
|
-
const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, retryExclusions, ctx.artifactDir));
|
|
618
|
+
const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, retryExclusions, ctx.artifactDir, ctx.reviewHistory));
|
|
602
619
|
if (second.meta?.noEligibleReviewer !== true) {
|
|
603
620
|
const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
|
|
604
621
|
const route = exclusion === "adapter"
|
|
@@ -630,11 +647,11 @@ export async function runGates(task, ctx) {
|
|
|
630
647
|
// The check runs BEFORE any gate, so on a clean tree it belongs to no gate: charging every round's
|
|
631
648
|
// first gate for it would inflate the one measurement the parked recalibrations key on. It becomes
|
|
632
649
|
// that gate's interval only on the path where it IS what the gate did — the refusal below.
|
|
633
|
-
const
|
|
634
|
-
const entryLoad = loadProvider();
|
|
650
|
+
const finishEntry = startMeasurement();
|
|
635
651
|
const entryDirt = sequence.length ? await dirtyWorktree() : undefined;
|
|
652
|
+
const entryMeasurement = finishEntry();
|
|
636
653
|
if (entryDirt) {
|
|
637
|
-
|
|
654
|
+
addMeasurement(sequence[0], entryMeasurement);
|
|
638
655
|
await emitStart(sequence[0]);
|
|
639
656
|
await record(dirtyRefusal(sequence[0], entryDirt));
|
|
640
657
|
return done();
|
package/dist/graph/graph.d.ts
CHANGED
|
@@ -1,6 +1,26 @@
|
|
|
1
1
|
import { type RunGraph, type Task, type TaskStatus } from "./schema.js";
|
|
2
2
|
export declare function stateDirName(_repoRoot: string): string;
|
|
3
3
|
export declare function graphPath(repoRoot: string): string;
|
|
4
|
+
export interface CompileRefusalRecord {
|
|
5
|
+
refusedAt: string;
|
|
6
|
+
source: string;
|
|
7
|
+
error: string;
|
|
8
|
+
}
|
|
9
|
+
export declare function compileRefusalPath(repoRoot: string): string;
|
|
10
|
+
export declare function readCompileRefusal(repoRoot: string): CompileRefusalRecord | undefined;
|
|
11
|
+
export declare function saveCompileRefusal(repoRoot: string, record: CompileRefusalRecord): void;
|
|
12
|
+
export declare function clearCompileRefusal(repoRoot: string): void;
|
|
13
|
+
export interface OnDiskSpecHash {
|
|
14
|
+
path: string;
|
|
15
|
+
hash: string;
|
|
16
|
+
}
|
|
17
|
+
/**
|
|
18
|
+
* Re-hash the exact single source file recorded by CLI compiles. Native and PRD record the source
|
|
19
|
+
* markdown; Spec Kit records its tasks.md. GSD combines several plan bodies and is deliberately not
|
|
20
|
+
* reconstructed here. A missing file is the one fail-open case: the refusal record is the fail-closed
|
|
21
|
+
* evidence for failed recompiles, while moved/deleted source files need not strand a compiled graph.
|
|
22
|
+
*/
|
|
23
|
+
export declare function onDiskSpecHash(_repoRoot: string, graph: RunGraph): OnDiskSpecHash | undefined;
|
|
4
24
|
export declare function graphDefinitionHash(g: RunGraph): string;
|
|
5
25
|
export declare function taskContentDigest(task: Pick<Task, "goal" | "files" | "acceptance">): string;
|
|
6
26
|
export declare function tickmarkrDir(repoRoot: string): string;
|
package/dist/graph/graph.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { createHash } from "node:crypto";
|
|
2
2
|
import { existsSync, mkdirSync, readFileSync, renameSync, rmSync, writeFileSync } from "node:fs";
|
|
3
|
-
import { join } from "node:path";
|
|
3
|
+
import { isAbsolute, join } from "node:path";
|
|
4
4
|
import { validateGraph } from "./schema.js";
|
|
5
5
|
export function stateDirName(_repoRoot) {
|
|
6
6
|
return ".tickmarkr";
|
|
@@ -8,6 +8,71 @@ export function stateDirName(_repoRoot) {
|
|
|
8
8
|
export function graphPath(repoRoot) {
|
|
9
9
|
return join(repoRoot, stateDirName(repoRoot), "graph.json");
|
|
10
10
|
}
|
|
11
|
+
export function compileRefusalPath(repoRoot) {
|
|
12
|
+
return join(repoRoot, stateDirName(repoRoot), "compile-refusal.json");
|
|
13
|
+
}
|
|
14
|
+
export function readCompileRefusal(repoRoot) {
|
|
15
|
+
const path = compileRefusalPath(repoRoot);
|
|
16
|
+
if (!existsSync(path))
|
|
17
|
+
return undefined;
|
|
18
|
+
let value;
|
|
19
|
+
try {
|
|
20
|
+
value = JSON.parse(readFileSync(path, "utf8"));
|
|
21
|
+
}
|
|
22
|
+
catch (error) {
|
|
23
|
+
throw new Error(`compile refusal record at ${path} is unreadable: ${error instanceof Error ? error.message : String(error)}`);
|
|
24
|
+
}
|
|
25
|
+
if (typeof value !== "object" || value === null
|
|
26
|
+
|| typeof value.refusedAt !== "string"
|
|
27
|
+
|| typeof value.source !== "string"
|
|
28
|
+
|| typeof value.error !== "string") {
|
|
29
|
+
throw new Error(`compile refusal record at ${path} is malformed`);
|
|
30
|
+
}
|
|
31
|
+
return value;
|
|
32
|
+
}
|
|
33
|
+
export function saveCompileRefusal(repoRoot, record) {
|
|
34
|
+
tickmarkrDir(repoRoot);
|
|
35
|
+
const path = compileRefusalPath(repoRoot);
|
|
36
|
+
const tmp = `${path}.${process.pid}.tmp`;
|
|
37
|
+
try {
|
|
38
|
+
writeFileSync(tmp, JSON.stringify(record, null, 2) + "\n");
|
|
39
|
+
renameSync(tmp, path);
|
|
40
|
+
}
|
|
41
|
+
catch (error) {
|
|
42
|
+
rmSync(tmp, { force: true });
|
|
43
|
+
throw error;
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
export function clearCompileRefusal(repoRoot) {
|
|
47
|
+
rmSync(compileRefusalPath(repoRoot), { force: true });
|
|
48
|
+
}
|
|
49
|
+
/**
|
|
50
|
+
* Re-hash the exact single source file recorded by CLI compiles. Native and PRD record the source
|
|
51
|
+
* markdown; Spec Kit records its tasks.md. GSD combines several plan bodies and is deliberately not
|
|
52
|
+
* reconstructed here. A missing file is the one fail-open case: the refusal record is the fail-closed
|
|
53
|
+
* evidence for failed recompiles, while moved/deleted source files need not strand a compiled graph.
|
|
54
|
+
*/
|
|
55
|
+
export function onDiskSpecHash(_repoRoot, graph) {
|
|
56
|
+
if (graph.spec.source === "gsd")
|
|
57
|
+
return undefined;
|
|
58
|
+
if (graph.spec.paths.length !== 1)
|
|
59
|
+
return undefined;
|
|
60
|
+
const recorded = graph.spec.paths[0];
|
|
61
|
+
// CLI compilation records an absolute source. Relative paths belong to legacy/programmatic
|
|
62
|
+
// graphs whose resolution context is unknowable here, so preserve their prior fail-open behavior.
|
|
63
|
+
if (!isAbsolute(recorded))
|
|
64
|
+
return undefined;
|
|
65
|
+
const path = recorded;
|
|
66
|
+
try {
|
|
67
|
+
const content = readFileSync(path);
|
|
68
|
+
return { path, hash: createHash("sha256").update(content).digest("hex") };
|
|
69
|
+
}
|
|
70
|
+
catch (error) {
|
|
71
|
+
if (error.code === "ENOENT")
|
|
72
|
+
return undefined;
|
|
73
|
+
throw new Error(`cannot verify compiled spec hash from ${path}: ${error instanceof Error ? error.message : String(error)}`);
|
|
74
|
+
}
|
|
75
|
+
}
|
|
11
76
|
// T3 (Sol #2 / Fable F2): ONE canonical engagement identity over COMPILED TASK DEFINITIONS only.
|
|
12
77
|
// status/evidence are runtime-mutated (the daemon flips status, accumulates evidence every attempt) so
|
|
13
78
|
// they are excluded — the identity survives a status flip or evidence growth but changes the instant a
|
|
@@ -5,6 +5,12 @@ export interface Disallowed {
|
|
|
5
5
|
entry: string;
|
|
6
6
|
}
|
|
7
7
|
export type PreferenceRole = "worker" | "judge" | "review" | "consult";
|
|
8
|
+
/** Provider identity comes from the served model, not a gateway adapter's stamped vendor. */
|
|
9
|
+
export declare function modelProvider(model: string, fallback?: string): string;
|
|
10
|
+
export declare const routingModelProvider: typeof modelProvider;
|
|
11
|
+
export declare const modelRouteIdentity: (model: string, fallback?: string) => string;
|
|
12
|
+
export declare const channelRouteIdentity: (key: string, fallback?: string) => string;
|
|
13
|
+
export declare function routingEntrySeatLines(cfg: TickmarkrConfig): string[];
|
|
8
14
|
export declare function excludedChannels(cfg: TickmarkrConfig, adapters: {
|
|
9
15
|
id: string;
|
|
10
16
|
}[] | string[], health: Record<string, AuthHealth>): {
|
package/dist/route/preference.js
CHANGED
|
@@ -1,6 +1,46 @@
|
|
|
1
1
|
import { channelKey, channelsFromConfig } from "../adapters/types.js";
|
|
2
2
|
import { validateGraph } from "../graph/schema.js";
|
|
3
3
|
import { route, RoutingError } from "./router.js";
|
|
4
|
+
const PREFERENCE_ROLES = ["worker", "judge", "review", "consult"];
|
|
5
|
+
/** Provider identity comes from the served model, not a gateway adapter's stamped vendor. */
|
|
6
|
+
export function modelProvider(model, fallback = "unknown") {
|
|
7
|
+
const id = model.toLowerCase();
|
|
8
|
+
const prefix = id.includes("/") ? id.slice(0, id.indexOf("/")) : "";
|
|
9
|
+
if (prefix === "openai" || prefix === "openai-codex" || /^(?:gpt|o\d)/.test(id))
|
|
10
|
+
return "openai";
|
|
11
|
+
if (prefix === "anthropic" || /^(?:claude|opus|sonnet|haiku|fable)(?:-|$)/.test(id))
|
|
12
|
+
return "anthropic";
|
|
13
|
+
if (prefix === "google" || /^gemini(?:-|$)/.test(id))
|
|
14
|
+
return "google";
|
|
15
|
+
if (prefix === "xai" || /^grok(?:-|$)/.test(id))
|
|
16
|
+
return "xai";
|
|
17
|
+
if (["zai", "zhipu", "zai-coding-plan"].includes(prefix) || /^glm(?:-|$)/.test(id))
|
|
18
|
+
return "zhipu";
|
|
19
|
+
if (["kimi-code", "moonshot"].includes(prefix) || /^kimi(?:-|$)/.test(id))
|
|
20
|
+
return "moonshot";
|
|
21
|
+
return fallback;
|
|
22
|
+
}
|
|
23
|
+
// Router naming remains explicit while sharing the review gate's exact function object.
|
|
24
|
+
export const routingModelProvider = modelProvider;
|
|
25
|
+
export const modelRouteIdentity = (model, fallback = "unknown") => `${routingModelProvider(model, fallback)}/${model.slice(model.lastIndexOf("/") + 1).toLowerCase()}`;
|
|
26
|
+
export const channelRouteIdentity = (key, fallback = "unknown") => {
|
|
27
|
+
const i = key.indexOf(":");
|
|
28
|
+
return i < 0 ? key : modelRouteIdentity(key.slice(i + 1), fallback);
|
|
29
|
+
};
|
|
30
|
+
export function routingEntrySeatLines(cfg) {
|
|
31
|
+
const lines = [];
|
|
32
|
+
const add = (path, entries, roles) => {
|
|
33
|
+
for (const entry of entries ?? [])
|
|
34
|
+
lines.push(`${path} '${entry}' reaches seats: ${roles.join(", ")}`);
|
|
35
|
+
};
|
|
36
|
+
add("routing.allow.adapters", cfg.routing.allow?.adapters, PREFERENCE_ROLES);
|
|
37
|
+
add("routing.allow.models", cfg.routing.allow?.models, PREFERENCE_ROLES);
|
|
38
|
+
add("routing.deny.adapters", cfg.routing.deny?.adapters, PREFERENCE_ROLES);
|
|
39
|
+
add("routing.deny.models", cfg.routing.deny?.models, PREFERENCE_ROLES);
|
|
40
|
+
add("routing.deny.workers.adapters", cfg.routing.deny?.workers?.adapters, ["worker"]);
|
|
41
|
+
add("routing.deny.workers.models", cfg.routing.deny?.workers?.models, ["worker"]);
|
|
42
|
+
return lines;
|
|
43
|
+
}
|
|
4
44
|
const adapterIds = (adapters) => typeof adapters[0] === "string" ? adapters : adapters.map((a) => a.id);
|
|
5
45
|
export function excludedChannels(cfg, adapters, health) {
|
|
6
46
|
const { allow, deny } = cfg.routing;
|
package/dist/route/router.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { channelKey, channelsFromConfig } from "../adapters/types.js";
|
|
2
2
|
import { TIER_RANK } from "../config/config.js";
|
|
3
|
-
import { disallowedBy } from "./preference.js";
|
|
3
|
+
import { channelRouteIdentity, disallowedBy, modelRouteIdentity, routingModelProvider } from "./preference.js";
|
|
4
4
|
import { cellOf, EXPLORE_CAP, explorationBonus, learnedScore, MIN_SAMPLES } from "./profile.js";
|
|
5
5
|
export const NO_EXPLORE_ENV = "TICKMARKR_NO_EXPLORE";
|
|
6
6
|
// OBS-89 (v1.60): the TICKMARKR_QUALITY variable is RETIRED — nothing in src reads it anymore and
|
|
@@ -337,7 +337,20 @@ export function nextChannel(current, task, cfg, channels, tried, profile, exclud
|
|
|
337
337
|
// profile-dependent filter. NO exploration bonus here (route():110 has one; a probe on the
|
|
338
338
|
// failure path would spend a real retry). Absent profile ⇒ every score is 0 ⇒ third key
|
|
339
339
|
// all-ties ⇒ the stable sort preserves the exact v1.7 candidate ORDER.
|
|
340
|
-
const
|
|
340
|
+
const triedKeys = new Set(tried);
|
|
341
|
+
const triedIdentities = new Set(tried.map((key) => {
|
|
342
|
+
const channel = channels.find((c) => channelKey(c) === key);
|
|
343
|
+
return channel ? modelRouteIdentity(channel.model, channel.vendor) : channelRouteIdentity(key);
|
|
344
|
+
}));
|
|
345
|
+
// excludeAdapter is expanded by the daemon into every channel key of the failed adapter. Once that
|
|
346
|
+
// complete set is present, the outage follows the current served provider across gateway aliases.
|
|
347
|
+
const currentAdapterExcluded = channels.some((c) => c.adapter === current.adapter)
|
|
348
|
+
&& channels.filter((c) => c.adapter === current.adapter).every((c) => triedKeys.has(channelKey(c)));
|
|
349
|
+
const currentChannel = channels.find((c) => c.adapter === current.adapter && c.model === current.model);
|
|
350
|
+
const excludedProvider = currentAdapterExcluded ? routingModelProvider(current.model, currentChannel?.vendor) : undefined;
|
|
351
|
+
const pool = channels.filter((c) => !triedIdentities.has(modelRouteIdentity(c.model, c.vendor))
|
|
352
|
+
&& (!excludedProvider || routingModelProvider(c.model, c.vendor) !== excludedProvider)
|
|
353
|
+
&& TIER_RANK[c.tier] >= TIER_RANK[current.tier]);
|
|
341
354
|
const scores = new Map(pool.map((c) => [channelKey(c), profile ? learnedScore(profile, task.shape, channelKey(c), c.channel, { availWeight: cfg.routing.learnedTuning?.availWeight }) : 0]));
|
|
342
355
|
const scoreOf = (c) => scores.get(channelKey(c));
|
|
343
356
|
const candidates = pool.sort((a, b) => TIER_RANK[a.tier] - TIER_RANK[b.tier] || marginalCostRank(a) - marginalCostRank(b) || scoreOf(b) - scoreOf(a));
|
package/dist/run/consult.d.ts
CHANGED
package/dist/run/consult.js
CHANGED
|
@@ -4,7 +4,7 @@ import { getAdapter } from "../adapters/registry.js";
|
|
|
4
4
|
import { bannerShell, paneDispatchCommand } from "../brand.js";
|
|
5
5
|
import { dewrapPaneVerdict, extractVerdictJson, gateExitTrailer, gatePaneName, generateVerdictNonce, verdictNonceLine } from "../gates/llm.js";
|
|
6
6
|
import { classifyVerdictCause } from "../gates/verdict-cause.js";
|
|
7
|
-
import { disallowedBy } from "../route/preference.js";
|
|
7
|
+
import { disallowedBy, routingModelProvider } from "../route/preference.js";
|
|
8
8
|
import { sh } from "./git.js";
|
|
9
9
|
import { redactSecrets } from "./redact.js";
|
|
10
10
|
import { filterLlmTranscript } from "./stall.js";
|
|
@@ -59,6 +59,23 @@ export function augmentRetryBrief(feedback, opts) {
|
|
|
59
59
|
return parts.join("\n\n");
|
|
60
60
|
}
|
|
61
61
|
const ACTIONS = ["retry", "reroute", "decompose", "human"];
|
|
62
|
+
function excludedProviderFromDossier(d, adapter) {
|
|
63
|
+
try {
|
|
64
|
+
const events = JSON.parse(d.journalTail);
|
|
65
|
+
const assignment = [...events].reverse().find((event) => event.event === "task-dispatch"
|
|
66
|
+
&& event.taskId === d.taskId
|
|
67
|
+
&& event.data?.assignment?.adapter === adapter
|
|
68
|
+
&& typeof event.data.assignment.model === "string")?.data?.assignment;
|
|
69
|
+
if (typeof assignment?.model !== "string")
|
|
70
|
+
return undefined;
|
|
71
|
+
const provider = routingModelProvider(assignment.model);
|
|
72
|
+
return provider === "unknown" ? undefined : provider;
|
|
73
|
+
}
|
|
74
|
+
catch {
|
|
75
|
+
// Legacy/non-JSON dossier tails retain the adapter exclusion without inventing a provider.
|
|
76
|
+
return undefined;
|
|
77
|
+
}
|
|
78
|
+
}
|
|
62
79
|
export function parseConsultVerdict(out, nonce) {
|
|
63
80
|
const v = extractVerdictJson(out, nonce);
|
|
64
81
|
if (!v)
|
|
@@ -110,10 +127,10 @@ ${d.journalTail}
|
|
|
110
127
|
Verdict meanings: retry = same assignment with your notes as feedback; reroute = different CLI/model;
|
|
111
128
|
decompose = task too big, needs human re-planning; human = a person must look at this.
|
|
112
129
|
|
|
113
|
-
On reroute only, optional excludeAdapter is
|
|
114
|
-
|
|
115
|
-
trust dialog, broken install) — not when a
|
|
116
|
-
reroutes so other models
|
|
130
|
+
On reroute only, optional excludeAdapter is the failed adapter id (e.g. "cursor-agent"). Tickmarkr
|
|
131
|
+
resolves the adapter's current model to its provider and bans that provider for this task. Use it for
|
|
132
|
+
environmental/provider failures ("the CLI is blocked", trust dialog, broken install) — not when a
|
|
133
|
+
single model produced bad code. Omit for model-level reroutes so other models remain eligible.
|
|
117
134
|
|
|
118
135
|
${verdictNonceLine(nonce)}
|
|
119
136
|
|
|
@@ -230,8 +247,19 @@ opts = {}) {
|
|
|
230
247
|
for (const [i, seat] of allowedSeats.entries()) {
|
|
231
248
|
try {
|
|
232
249
|
const parsed = await invokeSeat(seat.adapter, seat.model, i);
|
|
233
|
-
if (parsed.verdict)
|
|
234
|
-
|
|
250
|
+
if (parsed.verdict) {
|
|
251
|
+
const excludeProvider = parsed.verdict.excludeAdapter
|
|
252
|
+
? excludedProviderFromDossier(d, parsed.verdict.excludeAdapter)
|
|
253
|
+
: undefined;
|
|
254
|
+
return {
|
|
255
|
+
...parsed.verdict,
|
|
256
|
+
...(excludeProvider ? {
|
|
257
|
+
excludeProvider,
|
|
258
|
+
notes: `${parsed.verdict.notes} — excluded provider ${excludeProvider}`,
|
|
259
|
+
} : {}),
|
|
260
|
+
...seatIdentity(seat),
|
|
261
|
+
};
|
|
262
|
+
}
|
|
235
263
|
}
|
|
236
264
|
catch {
|
|
237
265
|
// failed seat (unknown adapter, dead driver/pane, shell error) — fall to the next entry
|
package/dist/run/daemon.d.ts
CHANGED
|
@@ -68,6 +68,7 @@ export declare function formatSummary(s: RunSummary): string;
|
|
|
68
68
|
* run id (cli/commands/status.ts positionalRunId), so naming it here is what stops the board from
|
|
69
69
|
* following the newest journal in a repo that already carries a second, newer run — a board showing
|
|
70
70
|
* the wrong run is a recorded incident (skills/tickmarkr-overseer/SKILL.md). */
|
|
71
|
+
export declare const daemonEntrypoint: string;
|
|
71
72
|
export declare const watchCommand: (runId: string) => string;
|
|
72
73
|
/**
|
|
73
74
|
* R3 (OBS-186): a gate that DECLINED to run is not a gate that failed. The review gate's skip branch
|
|
@@ -103,6 +104,9 @@ export declare const gateSatisfied: (g: GateResult) => boolean;
|
|
|
103
104
|
*/
|
|
104
105
|
export declare function decisiveReviewRounds(events: JournalEvent[]): JournalEvent[];
|
|
105
106
|
export declare const SUITE_POLL_MS = 250;
|
|
107
|
+
export declare const SUITE_WAIT_CEILING_MS = 600000;
|
|
108
|
+
export declare const setSuiteWaitCeilingForTests: (ms: number) => void;
|
|
109
|
+
export declare const resetSuiteWaitCeilingForTests: () => void;
|
|
106
110
|
export declare const APPROVAL_POLL_MS = 250;
|
|
107
111
|
export declare const EARLY_LAUNCH_LIVENESS_MS = 60000;
|
|
108
112
|
/** Test seam — lowers the empty-pane liveness window without sleeping 60s per case. */
|
|
@@ -113,6 +117,11 @@ export declare const WORKER_NUDGE_MESSAGE = "tickmarkr liveness check: if the ta
|
|
|
113
117
|
/** Test seam — shrink the nudge gate and grace without minute-long sleeps. */
|
|
114
118
|
export declare function setNudgeTimingForTests(silentMs: number, graceMs: number): void;
|
|
115
119
|
export declare function resetNudgeTimingForTests(): void;
|
|
120
|
+
export declare const WORKER_STARTUP_WINDOW_MS = 60000;
|
|
121
|
+
export declare const WORKER_STARTUP_WINDOW_BYTES: number;
|
|
122
|
+
/** Test seam — exercises both sides of the startup boundary without a production-length fixture. */
|
|
123
|
+
export declare function setWorkerStartupWindowMsForTests(ms: number): void;
|
|
124
|
+
export declare function resetWorkerStartupWindowMsForTests(): void;
|
|
116
125
|
/** Test seam — shrink the quota-banner silence gate without minute-long sleeps. */
|
|
117
126
|
export declare function setQuotaBannerSilentMsForTests(ms: number): void;
|
|
118
127
|
export declare function resetQuotaBannerSilentMsForTests(): void;
|