tickmarkr 2.2.1 → 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -9
- package/dist/adapters/catalog-remote.d.ts +1 -4
- package/dist/adapters/catalog-remote.js +52 -42
- package/dist/adapters/catalog.js +5 -3
- package/dist/adapters/claude-code.d.ts +1 -1
- package/dist/adapters/claude-code.js +8 -5
- package/dist/adapters/model-lints.d.ts +9 -5
- package/dist/adapters/model-lints.js +56 -15
- package/dist/adapters/model-windows.js +11 -0
- package/dist/adapters/prompt.js +1 -0
- package/dist/adapters/qwen.d.ts +5 -0
- package/dist/adapters/qwen.js +153 -0
- package/dist/adapters/types.d.ts +21 -1
- package/dist/adapters/types.js +43 -2
- package/dist/cli/commands/approve.js +5 -4
- package/dist/cli/commands/beat.js +7 -4
- package/dist/cli/commands/compile.js +32 -6
- package/dist/cli/commands/doctor.d.ts +9 -4
- package/dist/cli/commands/doctor.js +87 -13
- package/dist/cli/commands/fleet.d.ts +4 -0
- package/dist/cli/commands/fleet.js +53 -14
- package/dist/cli/commands/init.js +36 -21
- package/dist/cli/commands/plan.js +45 -7
- package/dist/cli/commands/report.js +37 -1
- package/dist/cli/commands/status.d.ts +1 -0
- package/dist/cli/commands/status.js +45 -1
- package/dist/cli/commands/verify.d.ts +6 -0
- package/dist/cli/commands/verify.js +145 -25
- package/dist/cli/index.d.ts +1 -1
- package/dist/cli/index.js +2 -2
- package/dist/compile/collateral.d.ts +2 -9
- package/dist/compile/collateral.js +17 -18
- package/dist/compile/index.d.ts +4 -1
- package/dist/compile/index.js +41 -7
- package/dist/compile/native.d.ts +4 -2
- package/dist/compile/native.js +58 -9
- package/dist/compile/ownership.js +34 -9
- package/dist/config/config.d.ts +1 -0
- package/dist/config/config.js +52 -6
- package/dist/drivers/herdr.d.ts +1 -0
- package/dist/drivers/herdr.js +11 -1
- package/dist/drivers/index.d.ts +6 -0
- package/dist/drivers/index.js +19 -4
- package/dist/drivers/orca.d.ts +35 -1
- package/dist/drivers/orca.js +260 -20
- package/dist/drivers/subprocess.d.ts +3 -3
- package/dist/drivers/subprocess.js +16 -9
- package/dist/drivers/types.d.ts +12 -0
- package/dist/gates/baseline.d.ts +2 -0
- package/dist/gates/baseline.js +47 -11
- package/dist/gates/llm.d.ts +6 -0
- package/dist/gates/llm.js +25 -9
- package/dist/gates/review.d.ts +7 -3
- package/dist/gates/review.js +61 -22
- package/dist/gates/run-gates.d.ts +5 -2
- package/dist/gates/run-gates.js +50 -26
- package/dist/gates/verdict-cause.d.ts +6 -2
- package/dist/gates/verdict-cause.js +8 -4
- package/dist/route/preference.d.ts +4 -0
- package/dist/route/preference.js +40 -0
- package/dist/route/router.js +15 -2
- package/dist/run/consult.d.ts +1 -0
- package/dist/run/consult.js +39 -8
- package/dist/run/daemon.d.ts +16 -0
- package/dist/run/daemon.js +345 -74
- package/dist/run/git.d.ts +3 -0
- package/dist/run/git.js +40 -5
- package/dist/run/journal.d.ts +15 -2
- package/dist/run/journal.js +73 -12
- package/dist/run/supervision.d.ts +6 -0
- package/dist/run/supervision.js +29 -1
- package/dist/tui/ink/fleet-app.d.ts +4 -0
- package/dist/tui/ink/fleet-app.js +45 -16
- package/dist/tui/ink/init-app.js +4 -4
- package/package.json +59 -1
- package/skills/tickmarkr-overseer/SKILL.md +77 -18
- package/skills/tickmarkr-overseer/scripts/seat-send.sh +88 -18
- package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +36 -2
- package/skills/tickmarkr-overseer/scripts/watch-contamination.sh +36 -15
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +33 -7
- package/skills/tickmarkr-overseer/scripts/watch-pending-input.sh +32 -8
package/dist/gates/run-gates.js
CHANGED
|
@@ -11,7 +11,7 @@ import { evidenceGate } from "./evidence.js";
|
|
|
11
11
|
import { captureLlmOutput } from "./llm.js";
|
|
12
12
|
import { disallowedBy } from "../route/preference.js";
|
|
13
13
|
import { marginalCostRank } from "../route/router.js";
|
|
14
|
-
import { reviewGate } from "./review.js";
|
|
14
|
+
import { pickReviewer, reviewGate } from "./review.js";
|
|
15
15
|
import { scopeGate } from "./scope.js";
|
|
16
16
|
import { shGit } from "../run/git.js";
|
|
17
17
|
import { withJudgeInvocationEvidence } from "../run/journal.js";
|
|
@@ -208,24 +208,42 @@ export async function runGates(task, ctx) {
|
|
|
208
208
|
// executing is added HERE, at the call site that runs it, so a gate that runs twice (the test
|
|
209
209
|
// gate's screen and its full suite) sums to its own cost and never to the span between them.
|
|
210
210
|
const spans = new Map();
|
|
211
|
+
const loadSamples = new Map();
|
|
211
212
|
// The test gate's two halves, kept apart as well as summed: `durationMs` alone cannot say whether
|
|
212
213
|
// a slow round was a slow subset or a slow full suite, and the parked scheduler's threshold is
|
|
213
214
|
// defined over the full-suite cost.
|
|
214
215
|
let selectedDurationMs;
|
|
215
216
|
let fullDurationMs;
|
|
216
|
-
const
|
|
217
|
+
const startMeasurement = () => {
|
|
217
218
|
const at = Date.now();
|
|
218
|
-
const
|
|
219
|
+
const samples = [loadProvider()];
|
|
220
|
+
const timer = setInterval(() => samples.push(loadProvider()), 1_000);
|
|
221
|
+
timer.unref();
|
|
222
|
+
return () => {
|
|
223
|
+
clearInterval(timer);
|
|
224
|
+
samples.push(loadProvider());
|
|
225
|
+
return { durationMs: Date.now() - at, samples };
|
|
226
|
+
};
|
|
227
|
+
};
|
|
228
|
+
const addMeasurement = (gate, measured) => {
|
|
229
|
+
const prior = spans.get(gate);
|
|
230
|
+
const samples = [...(loadSamples.get(gate) ?? []), ...measured.samples];
|
|
231
|
+
loadSamples.set(gate, samples);
|
|
232
|
+
spans.set(gate, {
|
|
233
|
+
durationMs: (prior?.durationMs ?? 0) + measured.durationMs,
|
|
234
|
+
load1Start: samples[0],
|
|
235
|
+
load1End: samples[samples.length - 1],
|
|
236
|
+
load1Max: Math.max(...samples),
|
|
237
|
+
load1Mean: samples.reduce((sum, value) => sum + value, 0) / samples.length,
|
|
238
|
+
});
|
|
239
|
+
};
|
|
240
|
+
const measure = async (gate, run) => {
|
|
241
|
+
const finish = startMeasurement();
|
|
219
242
|
try {
|
|
220
243
|
return await run();
|
|
221
244
|
}
|
|
222
245
|
finally {
|
|
223
|
-
|
|
224
|
-
spans.set(gate, {
|
|
225
|
-
durationMs: (prior?.durationMs ?? 0) + (Date.now() - at),
|
|
226
|
-
load1Start: prior?.load1Start ?? load1Start,
|
|
227
|
-
load1End: loadProvider(),
|
|
228
|
-
});
|
|
246
|
+
addMeasurement(gate, finish());
|
|
229
247
|
}
|
|
230
248
|
};
|
|
231
249
|
// The measurement is attached at the ONE seam every result leaves this function through, so a
|
|
@@ -347,12 +365,11 @@ export async function runGates(task, ctx) {
|
|
|
347
365
|
// suppresses them anyway; split compareToBaseline only if a tool gate ever gets slow.
|
|
348
366
|
// ponytail: legacy runs adjacent tools in ONE compareToBaseline call, so there is one interval
|
|
349
367
|
// to measure and each of its gates carries it. Split it only if this branch ever stops batching.
|
|
350
|
-
const
|
|
351
|
-
const batchLoadStart = loadProvider();
|
|
368
|
+
const finish = startMeasurement();
|
|
352
369
|
const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [...gates]);
|
|
353
|
-
const batch =
|
|
370
|
+
const batch = finish();
|
|
354
371
|
for (const g of gates)
|
|
355
|
-
|
|
372
|
+
addMeasurement(g, batch);
|
|
356
373
|
// The same refusal AFTER the commands, because a green command can dirty the tree the check
|
|
357
374
|
// above just proved clean. Batched, legacy cannot say WHICH command did it, so the refusal
|
|
358
375
|
// lands on the last gate that had one — the round dies there either way. A red battery is
|
|
@@ -540,7 +557,7 @@ export async function runGates(task, ctx) {
|
|
|
540
557
|
.sort((x, y) => TIER_RANK[y.tier] - TIER_RANK[x.tier] || marginalCostRank(x) - marginalCostRank(y))[0];
|
|
541
558
|
// v1.87 T2: the failover seat obeys the same policy the primary judge just passed — a denied
|
|
542
559
|
// channel is refused here too, never reached by falling through the exclusion arms below.
|
|
543
|
-
const judgePool = (ctx.judgeChannels ??
|
|
560
|
+
const judgePool = (ctx.judgeChannels ?? []).filter((c) => disallowedBy(c, ctx.cfg.routing, "judge") === null);
|
|
544
561
|
const crossAdapter = pick(judgePool.filter((c) => c.adapter !== flakedAdapter));
|
|
545
562
|
const sameAdapter = pick(judgePool.filter((c) => c.adapter === flakedAdapter && channelKey(c) !== flakedKey));
|
|
546
563
|
// Prefer a different adapter; if the fleet only has one adapter, retry on a different channel of
|
|
@@ -575,12 +592,10 @@ export async function runGates(task, ctx) {
|
|
|
575
592
|
invocations.push(...captured.invocations);
|
|
576
593
|
return captured.value;
|
|
577
594
|
};
|
|
578
|
-
let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir));
|
|
579
|
-
// OBS-193: an unparseable review verdict retries the REVIEW exactly once
|
|
580
|
-
//
|
|
581
|
-
//
|
|
582
|
-
// parameter, so pickReviewer's diversity rules still govern the retry seat; a fleet with no second
|
|
583
|
-
// eligible seat keeps the ORIGINAL result so the recorded cause stays truthful (OBS-196).
|
|
595
|
+
let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir, ctx.reviewHistory));
|
|
596
|
+
// OBS-193/574: an unparseable review verdict retries the REVIEW exactly once, preferring a
|
|
597
|
+
// different adapter. Only a single-adapter eligible pool may fall back to another channel on the
|
|
598
|
+
// flaked adapter. The flaked verdict never enters results; an exhausted pool preserves its cause.
|
|
584
599
|
if (rv.meta?.unparseable === true && typeof rv.meta.reviewer === "string") {
|
|
585
600
|
const flaked = rv.meta.reviewer;
|
|
586
601
|
const emptyOutput = rv.meta.cause === "empty-output";
|
|
@@ -594,15 +609,24 @@ export async function runGates(task, ctx) {
|
|
|
594
609
|
const retryVia = ctx.via
|
|
595
610
|
? { ...ctx.via, nameFor: (role, adapter) => ctx.via.nameFor(role, adapter) + "-r1" }
|
|
596
611
|
: undefined;
|
|
597
|
-
const
|
|
612
|
+
const priorExclusions = ctx.excludeReviewers ?? [];
|
|
613
|
+
const flakedAdapter = flaked.slice(0, flaked.indexOf(":"));
|
|
614
|
+
const adapterExclusions = ctx.channels.filter((c) => c.adapter === flakedAdapter).map(channelKey);
|
|
615
|
+
const crossAdapter = pickReviewer(ctx.author, ctx.channels, [...priorExclusions, ...adapterExclusions], ctx.cfg.review.prefer ?? [], task.routingHints?.floor);
|
|
616
|
+
const exclusion = crossAdapter ? "adapter" : "channel";
|
|
617
|
+
const retryExclusions = [...priorExclusions, ...(crossAdapter ? adapterExclusions : [flaked])];
|
|
618
|
+
const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, retryExclusions, ctx.artifactDir, ctx.reviewHistory));
|
|
598
619
|
if (second.meta?.noEligibleReviewer !== true) {
|
|
599
620
|
const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
|
|
621
|
+
const route = exclusion === "adapter"
|
|
622
|
+
? `different-adapter retry; excluded flaked adapter ${flakedAdapter}`
|
|
623
|
+
: `same-adapter fallback; excluded flaked channel ${flaked}`;
|
|
600
624
|
rv = {
|
|
601
625
|
...second,
|
|
602
626
|
// `details` is lifted onto the journal's gate-result row; meta.reviewRetry is not. Keep the
|
|
603
627
|
// re-route visible in the result text a reader actually opens, including on a red retry.
|
|
604
|
-
details: `review re-route: ${flaked} produced ${emptyOutput ? "EMPTY output" : "no parseable verdict"}; replaced by ${retried}\n${second.details}`,
|
|
605
|
-
meta: { ...second.meta, reviewRetry: { flaked, retried } },
|
|
628
|
+
details: `review re-route (${route}): ${flaked} produced ${emptyOutput ? "EMPTY output" : "no parseable verdict"}; replaced by ${retried}\n${second.details}`,
|
|
629
|
+
meta: { ...second.meta, reviewRetry: { flaked, retried, exclusion } },
|
|
606
630
|
};
|
|
607
631
|
}
|
|
608
632
|
else if (task.routingHints?.floor) {
|
|
@@ -623,11 +647,11 @@ export async function runGates(task, ctx) {
|
|
|
623
647
|
// The check runs BEFORE any gate, so on a clean tree it belongs to no gate: charging every round's
|
|
624
648
|
// first gate for it would inflate the one measurement the parked recalibrations key on. It becomes
|
|
625
649
|
// that gate's interval only on the path where it IS what the gate did — the refusal below.
|
|
626
|
-
const
|
|
627
|
-
const entryLoad = loadProvider();
|
|
650
|
+
const finishEntry = startMeasurement();
|
|
628
651
|
const entryDirt = sequence.length ? await dirtyWorktree() : undefined;
|
|
652
|
+
const entryMeasurement = finishEntry();
|
|
629
653
|
if (entryDirt) {
|
|
630
|
-
|
|
654
|
+
addMeasurement(sequence[0], entryMeasurement);
|
|
631
655
|
await emitStart(sequence[0]);
|
|
632
656
|
await record(dirtyRefusal(sequence[0], entryDirt));
|
|
633
657
|
return done();
|
|
@@ -1,4 +1,8 @@
|
|
|
1
1
|
export type VerdictDiscriminator = "approve" | "pass" | "ok" | "action";
|
|
2
|
-
export type VerdictUnparseableCause = "empty-output" | "no-verdict" | "malformed-verdict";
|
|
2
|
+
export type VerdictUnparseableCause = "empty-output" | "no-verdict" | "malformed-verdict" | "timeout" | "startup-failure";
|
|
3
|
+
export interface VerdictProcessFacts {
|
|
4
|
+
timedOut?: boolean;
|
|
5
|
+
exitCode?: number;
|
|
6
|
+
}
|
|
3
7
|
export declare function hasVerdictParticipationWitness(raw: string, nonce: string, discriminator: VerdictDiscriminator): boolean;
|
|
4
|
-
export declare function classifyVerdictCause(raw: string, nonce: string, discriminator: VerdictDiscriminator): VerdictUnparseableCause;
|
|
8
|
+
export declare function classifyVerdictCause(raw: string, nonce: string, discriminator: VerdictDiscriminator, process?: VerdictProcessFacts): VerdictUnparseableCause;
|
|
@@ -54,10 +54,14 @@ export function hasVerdictParticipationWitness(raw, nonce, discriminator) {
|
|
|
54
54
|
const discriminatorPattern = new RegExp(String.raw `"${discriminator}"\s*:\s*${valuePattern}\s*[,}]`);
|
|
55
55
|
return objectPrefixes(raw).some((prefix) => noncePattern.test(prefix) && discriminatorPattern.test(prefix));
|
|
56
56
|
}
|
|
57
|
-
export function classifyVerdictCause(raw, nonce, discriminator) {
|
|
57
|
+
export function classifyVerdictCause(raw, nonce, discriminator, process = {}) {
|
|
58
|
+
if (process.timedOut)
|
|
59
|
+
return "timeout";
|
|
58
60
|
if (raw.trim().length === 0)
|
|
59
61
|
return "empty-output";
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
62
|
+
if (hasVerdictParticipationWitness(raw, nonce, discriminator))
|
|
63
|
+
return "malformed-verdict";
|
|
64
|
+
if (process.exitCode !== undefined && process.exitCode !== 0)
|
|
65
|
+
return "startup-failure";
|
|
66
|
+
return "no-verdict";
|
|
63
67
|
}
|
|
@@ -5,6 +5,10 @@ export interface Disallowed {
|
|
|
5
5
|
entry: string;
|
|
6
6
|
}
|
|
7
7
|
export type PreferenceRole = "worker" | "judge" | "review" | "consult";
|
|
8
|
+
export declare function routingModelProvider(model: string, fallback?: string): string;
|
|
9
|
+
export declare const modelRouteIdentity: (model: string, fallback?: string) => string;
|
|
10
|
+
export declare const channelRouteIdentity: (key: string, fallback?: string) => string;
|
|
11
|
+
export declare function routingEntrySeatLines(cfg: TickmarkrConfig): string[];
|
|
8
12
|
export declare function excludedChannels(cfg: TickmarkrConfig, adapters: {
|
|
9
13
|
id: string;
|
|
10
14
|
}[] | string[], health: Record<string, AuthHealth>): {
|
package/dist/route/preference.js
CHANGED
|
@@ -1,6 +1,46 @@
|
|
|
1
1
|
import { channelKey, channelsFromConfig } from "../adapters/types.js";
|
|
2
2
|
import { validateGraph } from "../graph/schema.js";
|
|
3
3
|
import { route, RoutingError } from "./router.js";
|
|
4
|
+
const PREFERENCE_ROLES = ["worker", "judge", "review", "consult"];
|
|
5
|
+
// Keep routing retries on the same identity review diversity uses: the served provider plus the
|
|
6
|
+
// unprefixed model id. Gate review owns the original modelProvider policy; this dependency-leaf copy
|
|
7
|
+
// avoids importing gates back into route/run and must move with that helper when the scope permits.
|
|
8
|
+
export function routingModelProvider(model, fallback = "unknown") {
|
|
9
|
+
const id = model.toLowerCase();
|
|
10
|
+
const prefix = id.includes("/") ? id.slice(0, id.indexOf("/")) : "";
|
|
11
|
+
if (prefix === "openai" || prefix === "openai-codex" || /^(?:gpt|o\d)/.test(id))
|
|
12
|
+
return "openai";
|
|
13
|
+
if (prefix === "anthropic" || /^(?:claude|opus|sonnet|haiku|fable)(?:-|$)/.test(id))
|
|
14
|
+
return "anthropic";
|
|
15
|
+
if (prefix === "google" || /^gemini(?:-|$)/.test(id))
|
|
16
|
+
return "google";
|
|
17
|
+
if (prefix === "xai" || /^grok(?:-|$)/.test(id))
|
|
18
|
+
return "xai";
|
|
19
|
+
if (["zai", "zhipu", "zai-coding-plan"].includes(prefix) || /^glm(?:-|$)/.test(id))
|
|
20
|
+
return "zhipu";
|
|
21
|
+
if (["kimi-code", "moonshot"].includes(prefix) || /^kimi(?:-|$)/.test(id))
|
|
22
|
+
return "moonshot";
|
|
23
|
+
return fallback;
|
|
24
|
+
}
|
|
25
|
+
export const modelRouteIdentity = (model, fallback = "unknown") => `${routingModelProvider(model, fallback)}/${model.slice(model.lastIndexOf("/") + 1).toLowerCase()}`;
|
|
26
|
+
export const channelRouteIdentity = (key, fallback = "unknown") => {
|
|
27
|
+
const i = key.indexOf(":");
|
|
28
|
+
return i < 0 ? key : modelRouteIdentity(key.slice(i + 1), fallback);
|
|
29
|
+
};
|
|
30
|
+
export function routingEntrySeatLines(cfg) {
|
|
31
|
+
const lines = [];
|
|
32
|
+
const add = (path, entries, roles) => {
|
|
33
|
+
for (const entry of entries ?? [])
|
|
34
|
+
lines.push(`${path} '${entry}' reaches seats: ${roles.join(", ")}`);
|
|
35
|
+
};
|
|
36
|
+
add("routing.allow.adapters", cfg.routing.allow?.adapters, PREFERENCE_ROLES);
|
|
37
|
+
add("routing.allow.models", cfg.routing.allow?.models, PREFERENCE_ROLES);
|
|
38
|
+
add("routing.deny.adapters", cfg.routing.deny?.adapters, PREFERENCE_ROLES);
|
|
39
|
+
add("routing.deny.models", cfg.routing.deny?.models, PREFERENCE_ROLES);
|
|
40
|
+
add("routing.deny.workers.adapters", cfg.routing.deny?.workers?.adapters, ["worker"]);
|
|
41
|
+
add("routing.deny.workers.models", cfg.routing.deny?.workers?.models, ["worker"]);
|
|
42
|
+
return lines;
|
|
43
|
+
}
|
|
4
44
|
const adapterIds = (adapters) => typeof adapters[0] === "string" ? adapters : adapters.map((a) => a.id);
|
|
5
45
|
export function excludedChannels(cfg, adapters, health) {
|
|
6
46
|
const { allow, deny } = cfg.routing;
|
package/dist/route/router.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { channelKey, channelsFromConfig } from "../adapters/types.js";
|
|
2
2
|
import { TIER_RANK } from "../config/config.js";
|
|
3
|
-
import { disallowedBy } from "./preference.js";
|
|
3
|
+
import { channelRouteIdentity, disallowedBy, modelRouteIdentity, routingModelProvider } from "./preference.js";
|
|
4
4
|
import { cellOf, EXPLORE_CAP, explorationBonus, learnedScore, MIN_SAMPLES } from "./profile.js";
|
|
5
5
|
export const NO_EXPLORE_ENV = "TICKMARKR_NO_EXPLORE";
|
|
6
6
|
// OBS-89 (v1.60): the TICKMARKR_QUALITY variable is RETIRED — nothing in src reads it anymore and
|
|
@@ -337,7 +337,20 @@ export function nextChannel(current, task, cfg, channels, tried, profile, exclud
|
|
|
337
337
|
// profile-dependent filter. NO exploration bonus here (route():110 has one; a probe on the
|
|
338
338
|
// failure path would spend a real retry). Absent profile ⇒ every score is 0 ⇒ third key
|
|
339
339
|
// all-ties ⇒ the stable sort preserves the exact v1.7 candidate ORDER.
|
|
340
|
-
const
|
|
340
|
+
const triedKeys = new Set(tried);
|
|
341
|
+
const triedIdentities = new Set(tried.map((key) => {
|
|
342
|
+
const channel = channels.find((c) => channelKey(c) === key);
|
|
343
|
+
return channel ? modelRouteIdentity(channel.model, channel.vendor) : channelRouteIdentity(key);
|
|
344
|
+
}));
|
|
345
|
+
// excludeAdapter is expanded by the daemon into every channel key of the failed adapter. Once that
|
|
346
|
+
// complete set is present, the outage follows the current served provider across gateway aliases.
|
|
347
|
+
const currentAdapterExcluded = channels.some((c) => c.adapter === current.adapter)
|
|
348
|
+
&& channels.filter((c) => c.adapter === current.adapter).every((c) => triedKeys.has(channelKey(c)));
|
|
349
|
+
const currentChannel = channels.find((c) => c.adapter === current.adapter && c.model === current.model);
|
|
350
|
+
const excludedProvider = currentAdapterExcluded ? routingModelProvider(current.model, currentChannel?.vendor) : undefined;
|
|
351
|
+
const pool = channels.filter((c) => !triedIdentities.has(modelRouteIdentity(c.model, c.vendor))
|
|
352
|
+
&& (!excludedProvider || routingModelProvider(c.model, c.vendor) !== excludedProvider)
|
|
353
|
+
&& TIER_RANK[c.tier] >= TIER_RANK[current.tier]);
|
|
341
354
|
const scores = new Map(pool.map((c) => [channelKey(c), profile ? learnedScore(profile, task.shape, channelKey(c), c.channel, { availWeight: cfg.routing.learnedTuning?.availWeight }) : 0]));
|
|
342
355
|
const scoreOf = (c) => scores.get(channelKey(c));
|
|
343
356
|
const candidates = pool.sort((a, b) => TIER_RANK[a.tier] - TIER_RANK[b.tier] || marginalCostRank(a) - marginalCostRank(b) || scoreOf(b) - scoreOf(a));
|
package/dist/run/consult.d.ts
CHANGED
package/dist/run/consult.js
CHANGED
|
@@ -4,7 +4,7 @@ import { getAdapter } from "../adapters/registry.js";
|
|
|
4
4
|
import { bannerShell, paneDispatchCommand } from "../brand.js";
|
|
5
5
|
import { dewrapPaneVerdict, extractVerdictJson, gateExitTrailer, gatePaneName, generateVerdictNonce, verdictNonceLine } from "../gates/llm.js";
|
|
6
6
|
import { classifyVerdictCause } from "../gates/verdict-cause.js";
|
|
7
|
-
import { disallowedBy } from "../route/preference.js";
|
|
7
|
+
import { disallowedBy, routingModelProvider } from "../route/preference.js";
|
|
8
8
|
import { sh } from "./git.js";
|
|
9
9
|
import { redactSecrets } from "./redact.js";
|
|
10
10
|
import { filterLlmTranscript } from "./stall.js";
|
|
@@ -59,6 +59,22 @@ export function augmentRetryBrief(feedback, opts) {
|
|
|
59
59
|
return parts.join("\n\n");
|
|
60
60
|
}
|
|
61
61
|
const ACTIONS = ["retry", "reroute", "decompose", "human"];
|
|
62
|
+
function excludedProviderFromDossier(d, adapter) {
|
|
63
|
+
try {
|
|
64
|
+
const events = JSON.parse(d.journalTail);
|
|
65
|
+
const assignment = [...events].reverse().find((event) => event.event === "task-dispatch"
|
|
66
|
+
&& event.data?.assignment?.adapter === adapter
|
|
67
|
+
&& typeof event.data.assignment.model === "string")?.data?.assignment;
|
|
68
|
+
if (typeof assignment?.model !== "string")
|
|
69
|
+
return undefined;
|
|
70
|
+
const provider = routingModelProvider(assignment.model);
|
|
71
|
+
return provider === "unknown" ? undefined : provider;
|
|
72
|
+
}
|
|
73
|
+
catch {
|
|
74
|
+
// Legacy/non-JSON dossier tails retain the adapter exclusion without inventing a provider.
|
|
75
|
+
return undefined;
|
|
76
|
+
}
|
|
77
|
+
}
|
|
62
78
|
export function parseConsultVerdict(out, nonce) {
|
|
63
79
|
const v = extractVerdictJson(out, nonce);
|
|
64
80
|
if (!v)
|
|
@@ -110,10 +126,10 @@ ${d.journalTail}
|
|
|
110
126
|
Verdict meanings: retry = same assignment with your notes as feedback; reroute = different CLI/model;
|
|
111
127
|
decompose = task too big, needs human re-planning; human = a person must look at this.
|
|
112
128
|
|
|
113
|
-
On reroute only, optional excludeAdapter is
|
|
114
|
-
|
|
115
|
-
trust dialog, broken install) — not when a
|
|
116
|
-
reroutes so other models
|
|
129
|
+
On reroute only, optional excludeAdapter is the failed adapter id (e.g. "cursor-agent"). Tickmarkr
|
|
130
|
+
resolves the adapter's current model to its provider and bans that provider for this task. Use it for
|
|
131
|
+
environmental/provider failures ("the CLI is blocked", trust dialog, broken install) — not when a
|
|
132
|
+
single model produced bad code. Omit for model-level reroutes so other models remain eligible.
|
|
117
133
|
|
|
118
134
|
${verdictNonceLine(nonce)}
|
|
119
135
|
|
|
@@ -147,6 +163,10 @@ opts = {}) {
|
|
|
147
163
|
out = r.stdout + r.stderr;
|
|
148
164
|
}
|
|
149
165
|
else {
|
|
166
|
+
// Preserve the adapter contract for CLIs that cannot seed their TUI: supported adapters use
|
|
167
|
+
// the interactive form, while a declared null keeps the existing visible print fallback.
|
|
168
|
+
const command = adapter.interactiveCommand(promptFile, seatModel)
|
|
169
|
+
?? adapter.headlessCommand(promptFile, seatModel);
|
|
150
170
|
// T8: role-first pane name for fleet visibility (consult · T2); consultSeq stays on the dossier artifact only
|
|
151
171
|
const slot = await driver.slot(cwd, gatePaneName("consult", d.taskId), {
|
|
152
172
|
label: `CONSULT ${d.taskId}`,
|
|
@@ -158,7 +178,7 @@ opts = {}) {
|
|
|
158
178
|
writeFileSync(scriptPath, [
|
|
159
179
|
"export BASH_SILENCE_DEPRECATION_WARNING=1",
|
|
160
180
|
bannerShell(),
|
|
161
|
-
|
|
181
|
+
command,
|
|
162
182
|
gateExitTrailer(nonce),
|
|
163
183
|
].join("\n"));
|
|
164
184
|
try {
|
|
@@ -226,8 +246,19 @@ opts = {}) {
|
|
|
226
246
|
for (const [i, seat] of allowedSeats.entries()) {
|
|
227
247
|
try {
|
|
228
248
|
const parsed = await invokeSeat(seat.adapter, seat.model, i);
|
|
229
|
-
if (parsed.verdict)
|
|
230
|
-
|
|
249
|
+
if (parsed.verdict) {
|
|
250
|
+
const excludeProvider = parsed.verdict.excludeAdapter
|
|
251
|
+
? excludedProviderFromDossier(d, parsed.verdict.excludeAdapter)
|
|
252
|
+
: undefined;
|
|
253
|
+
return {
|
|
254
|
+
...parsed.verdict,
|
|
255
|
+
...(excludeProvider ? {
|
|
256
|
+
excludeProvider,
|
|
257
|
+
notes: `${parsed.verdict.notes} — excluded provider ${excludeProvider}`,
|
|
258
|
+
} : {}),
|
|
259
|
+
...seatIdentity(seat),
|
|
260
|
+
};
|
|
261
|
+
}
|
|
231
262
|
}
|
|
232
263
|
catch {
|
|
233
264
|
// failed seat (unknown adapter, dead driver/pane, shell error) — fall to the next entry
|
package/dist/run/daemon.d.ts
CHANGED
|
@@ -68,6 +68,7 @@ export declare function formatSummary(s: RunSummary): string;
|
|
|
68
68
|
* run id (cli/commands/status.ts positionalRunId), so naming it here is what stops the board from
|
|
69
69
|
* following the newest journal in a repo that already carries a second, newer run — a board showing
|
|
70
70
|
* the wrong run is a recorded incident (skills/tickmarkr-overseer/SKILL.md). */
|
|
71
|
+
export declare const daemonEntrypoint: string;
|
|
71
72
|
export declare const watchCommand: (runId: string) => string;
|
|
72
73
|
/**
|
|
73
74
|
* R3 (OBS-186): a gate that DECLINED to run is not a gate that failed. The review gate's skip branch
|
|
@@ -102,6 +103,11 @@ export declare const gateSatisfied: (g: GateResult) => boolean;
|
|
|
102
103
|
* A round is the gate-result span opened by each `gates` phase-start, per task.
|
|
103
104
|
*/
|
|
104
105
|
export declare function decisiveReviewRounds(events: JournalEvent[]): JournalEvent[];
|
|
106
|
+
export declare const SUITE_POLL_MS = 250;
|
|
107
|
+
export declare const SUITE_WAIT_CEILING_MS = 600000;
|
|
108
|
+
export declare const setSuiteWaitCeilingForTests: (ms: number) => void;
|
|
109
|
+
export declare const resetSuiteWaitCeilingForTests: () => void;
|
|
110
|
+
export declare const APPROVAL_POLL_MS = 250;
|
|
105
111
|
export declare const EARLY_LAUNCH_LIVENESS_MS = 60000;
|
|
106
112
|
/** Test seam — lowers the empty-pane liveness window without sleeping 60s per case. */
|
|
107
113
|
export declare function setEarlyLaunchLivenessMsForTests(ms: number): void;
|
|
@@ -141,6 +147,16 @@ export declare function verifyIntegrationTipCached(intWt: string, commands: Reco
|
|
|
141
147
|
lastMergedTask?: string;
|
|
142
148
|
baseline?: Baseline;
|
|
143
149
|
}): Promise<boolean>;
|
|
150
|
+
type SuitePidProbe = (pid: number) => number | undefined;
|
|
151
|
+
/** Count full-suite roots in one process-table snapshot. The probes are arguments so the ownership
|
|
152
|
+
* rules remain testable on hosts that forbid process inspection; production supplies cwd and the
|
|
153
|
+
* inherited TICKMARKR_SUITE_PARENT marker from the process itself. */
|
|
154
|
+
export declare function countLiveSuites(snapshot: string, repoRoot: string, daemonPid?: number, cwdForPid?: (pid: number) => string | undefined, suiteParentForPid?: SuitePidProbe): number;
|
|
155
|
+
export declare const setLiveSuiteCountForTests: (probe: (repoRoot: string) => Promise<number>) => void;
|
|
156
|
+
export declare const resetLiveSuiteCountForTests: () => void;
|
|
157
|
+
/** Live full-suite roots attributable to this repository or this daemon. Ancestors are excluded so
|
|
158
|
+
* a daemon invoked by vitest does not wait on its own test harness forever. */
|
|
159
|
+
export declare function liveSuiteCount(repoRoot: string): Promise<number>;
|
|
144
160
|
/** Test seam — exercise the production observer's total read bound with a small real tree. */
|
|
145
161
|
export declare function setObserveBudgetBytesForTests(bytes: number): void;
|
|
146
162
|
export declare function resetObserveBudgetBytesForTests(): void;
|