tickmarkr 1.85.0 → 1.87.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -2
- package/dist/adapters/catalog-remote.d.ts +64 -0
- package/dist/adapters/catalog-remote.js +287 -0
- package/dist/adapters/catalog.d.ts +108 -0
- package/dist/adapters/catalog.js +189 -0
- package/dist/adapters/claude-code.js +5 -3
- package/dist/adapters/fake.js +33 -4
- package/dist/adapters/model-lints.d.ts +25 -5
- package/dist/adapters/model-lints.js +184 -50
- package/dist/adapters/model-windows.d.ts +31 -0
- package/dist/adapters/model-windows.js +69 -0
- package/dist/adapters/prompt.d.ts +5 -1
- package/dist/adapters/prompt.js +13 -4
- package/dist/adapters/registry.d.ts +34 -27
- package/dist/adapters/registry.js +215 -112
- package/dist/adapters/types.js +15 -3
- package/dist/brand.d.ts +5 -1
- package/dist/brand.js +18 -2
- package/dist/cli/commands/doctor.d.ts +3 -0
- package/dist/cli/commands/doctor.js +43 -21
- package/dist/cli/commands/fleet.d.ts +7 -0
- package/dist/cli/commands/fleet.js +94 -74
- package/dist/cli/commands/init.js +118 -5
- package/dist/cli/commands/plan.js +11 -1
- package/dist/cli/commands/resume.js +7 -1
- package/dist/cli/commands/status.js +44 -18
- package/dist/compile/gsd.d.ts +2 -1
- package/dist/compile/gsd.js +68 -2
- package/dist/compile/native.d.ts +14 -0
- package/dist/compile/native.js +168 -12
- package/dist/config/config.d.ts +20 -5
- package/dist/config/config.js +96 -64
- package/dist/config/fleet-overlay.d.ts +25 -20
- package/dist/config/fleet-overlay.js +195 -77
- package/dist/config/fleet-why.d.ts +23 -0
- package/dist/config/fleet-why.js +42 -0
- package/dist/drivers/herdr.d.ts +1 -0
- package/dist/drivers/herdr.js +70 -19
- package/dist/gates/acceptance.js +7 -2
- package/dist/gates/llm.d.ts +0 -1
- package/dist/gates/llm.js +5 -30
- package/dist/gates/review.d.ts +2 -1
- package/dist/gates/review.js +9 -7
- package/dist/gates/run-gates.d.ts +1 -0
- package/dist/gates/run-gates.js +21 -2
- package/dist/gates/verdict-cause.d.ts +4 -0
- package/dist/gates/verdict-cause.js +63 -0
- package/dist/graph/schema.d.ts +6 -0
- package/dist/graph/schema.js +8 -5
- package/dist/route/preference.d.ts +1 -1
- package/dist/route/preference.js +8 -1
- package/dist/route/router.d.ts +0 -5
- package/dist/route/router.js +16 -20
- package/dist/run/consult.d.ts +6 -0
- package/dist/run/consult.js +49 -26
- package/dist/run/daemon.js +81 -20
- package/dist/run/journal.js +87 -7
- package/dist/tui/cockpit/capture.d.ts +12 -0
- package/dist/tui/cockpit/capture.js +37 -1
- package/dist/tui/cockpit/components.js +8 -8
- package/dist/tui/cockpit/theme.d.ts +32 -26
- package/dist/tui/cockpit/theme.js +11 -5
- package/dist/tui/ink/components.d.ts +0 -15
- package/dist/tui/ink/components.js +0 -17
- package/dist/tui/ink/fleet-app.d.ts +4 -1
- package/dist/tui/ink/fleet-app.js +134 -13
- package/fixtures/gateway-models.json +1 -0
- package/fixtures/sample.native.md +1 -1
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +464 -34
- package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +86 -0
- package/skills/tickmarkr-overseer/scripts/watch-panes.sh +1 -1
- package/dist/tui/ink/studio-app.d.ts +0 -59
- package/dist/tui/ink/studio-app.js +0 -320
- package/dist/tui/save.d.ts +0 -38
- package/dist/tui/save.js +0 -96
- package/dist/tui/staging.d.ts +0 -29
- package/dist/tui/staging.js +0 -78
package/dist/gates/llm.js
CHANGED
|
@@ -69,29 +69,6 @@ export function extractPromptNonce(prompt) {
|
|
|
69
69
|
export function gateExitTrailer(nonce) {
|
|
70
70
|
return `printf '\\nTICKMARKR_''EXIT_${nonce}:%s\\n' $?`;
|
|
71
71
|
}
|
|
72
|
-
// v1.64: scripted fake judge verdicts predate the required per-criterion evidence field — quote the
|
|
73
|
-
// first line of the prompt's own diff block into rows lacking one so zero-token fixtures keep their
|
|
74
|
-
// outcomes. Rows scripting an explicit evidence value pass through verbatim (tests exercise both paths).
|
|
75
|
-
function injectFakeEvidence(obj, prompt) {
|
|
76
|
-
if (!prompt.startsWith("TICKMARKR-JUDGE") || !Array.isArray(obj.criteria))
|
|
77
|
-
return obj;
|
|
78
|
-
const line = /```diff\n([\s\S]*?)```/.exec(prompt)?.[1].split("\n").find((l) => l.trim());
|
|
79
|
-
if (!line)
|
|
80
|
-
return obj;
|
|
81
|
-
const criteria = obj.criteria.map((row) => row && typeof row === "object" && !("evidence" in row) ? { ...row, evidence: line } : row);
|
|
82
|
-
return { ...obj, criteria };
|
|
83
|
-
}
|
|
84
|
-
// ponytail: fake adapter serves static verdict JSON without nonce; append a bound copy for zero-token tests.
|
|
85
|
-
export function augmentFakeVerdictOutput(adapter, out, nonce, prompt = "") {
|
|
86
|
-
if (adapter.id !== "fake")
|
|
87
|
-
return out;
|
|
88
|
-
const obj = extractJson(out);
|
|
89
|
-
if (!obj || typeof obj !== "object" || obj.nonce === nonce)
|
|
90
|
-
return out;
|
|
91
|
-
if (typeof obj.nonce === "string")
|
|
92
|
-
return out;
|
|
93
|
-
return `${out}\n${JSON.stringify(injectFakeEvidence({ ...obj, nonce }, prompt))}`;
|
|
94
|
-
}
|
|
95
72
|
/** T8: role-first pane name for fleet visibility — judge · T4, review · T3, consult · T2. */
|
|
96
73
|
export function gatePaneName(role, taskId, suffix = "") {
|
|
97
74
|
return `${role}${GATE_PANE_SEP}${taskId}${suffix}`;
|
|
@@ -130,11 +107,7 @@ export async function runHeadless(adapter, model, prompt, cwd, timeoutMs = 30000
|
|
|
130
107
|
const pf = join(mkdtempSync(join(tmpdir(), "tickmarkr-llm-")), "prompt.md");
|
|
131
108
|
writeFileSync(pf, prompt);
|
|
132
109
|
const r = await sh(adapter.headlessCommand(pf, model), cwd, timeoutMs);
|
|
133
|
-
|
|
134
|
-
let out = r.stdout + "\n" + r.stderr;
|
|
135
|
-
if (nonce)
|
|
136
|
-
out = augmentFakeVerdictOutput(adapter, out, nonce, prompt);
|
|
137
|
-
return out;
|
|
110
|
+
return r.stdout + "\n" + r.stderr;
|
|
138
111
|
}
|
|
139
112
|
// v1.1 default path: the same headless CLI call, but dispatched through the driver
|
|
140
113
|
// as a visible named agent (herdr pane), with the quote-split completion wrapper.
|
|
@@ -157,10 +130,9 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
|
|
|
157
130
|
// nonce-suffixed exit only: a displayed bare "TICKMARKR_EXIT:" or another call's marker must not
|
|
158
131
|
// false-complete — same guard the worker path uses (daemon.ts:330-331).
|
|
159
132
|
await via.driver.waitOutput(slot, `TICKMARKR_EXIT_${nonce}:\\d`, timeoutMs, { regex: true });
|
|
160
|
-
|
|
133
|
+
const out = await via.driver.read(slot, 400);
|
|
161
134
|
if (!via.keep)
|
|
162
135
|
await via.driver.close(slot);
|
|
163
|
-
out = augmentFakeVerdictOutput(adapter, out, nonce, prompt);
|
|
164
136
|
return dewrapPaneVerdict(out, nonce);
|
|
165
137
|
}
|
|
166
138
|
// OBS-155: a TUI renders the verdict as a bullet and HARD-wraps it at pane width with a 2-space
|
|
@@ -179,6 +151,9 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
|
|
|
179
151
|
export function dewrapPaneVerdict(out, nonce) {
|
|
180
152
|
if (!out.includes(nonce))
|
|
181
153
|
return out;
|
|
154
|
+
// Preserve already-readable responder bytes; only a genuinely wrapped verdict needs reconstruction.
|
|
155
|
+
if (extractVerdictJson(out, nonce))
|
|
156
|
+
return out;
|
|
182
157
|
const lines = out.split("\n");
|
|
183
158
|
// OBS-209: EVERY brace-start is a candidate, scanned newest-first. findIndex took only the first,
|
|
184
159
|
// so any earlier line beginning with `{` — a quoted snippet, a lone brace in the reviewer's own
|
package/dist/gates/review.d.ts
CHANGED
|
@@ -3,6 +3,7 @@ import { type TickmarkrConfig } from "../config/config.js";
|
|
|
3
3
|
import { type Task } from "../graph/schema.js";
|
|
4
4
|
import { type GateVia } from "./llm.js";
|
|
5
5
|
import type { GateResult } from "./types.js";
|
|
6
|
+
import { type VerdictUnparseableCause } from "./verdict-cause.js";
|
|
6
7
|
export type ReviewSeverity = "material" | "minor";
|
|
7
8
|
export interface ReviewFinding {
|
|
8
9
|
note: string;
|
|
@@ -44,5 +45,5 @@ export declare function diffCapParkReason(results: GateResult[]): string | null;
|
|
|
44
45
|
export declare function modelId(model: string): string;
|
|
45
46
|
export declare function pickReviewer(author: Assignment, channels: BillingChannel[], exclude?: string[], // v1.1 failover: reviewer channels that already produced garbage for this task
|
|
46
47
|
prefer?: string[]): BillingChannel | null;
|
|
47
|
-
export type ReviewUnparseableCause =
|
|
48
|
+
export type ReviewUnparseableCause = VerdictUnparseableCause;
|
|
48
49
|
export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string): Promise<GateResult>;
|
package/dist/gates/review.js
CHANGED
|
@@ -8,6 +8,7 @@ import { shOk } from "../run/git.js";
|
|
|
8
8
|
import { redactSecrets } from "../run/redact.js";
|
|
9
9
|
import { marginalCostRank } from "../route/router.js";
|
|
10
10
|
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
|
|
11
|
+
import { classifyVerdictCause } from "./verdict-cause.js";
|
|
11
12
|
// legacy flat `issues` shape — every issue blocks; the approve flag must agree with the list.
|
|
12
13
|
function classifyReviewIssues(approve, issues) {
|
|
13
14
|
const inconsistencies = [];
|
|
@@ -303,9 +304,9 @@ artifactDir) {
|
|
|
303
304
|
// "a law caps complexity at 3, the gate starts at 7" unreachability OBS-186 measured.
|
|
304
305
|
//
|
|
305
306
|
// COLLATERAL this rescoped task closed: the run-gates/daemon participation assertions are rewritten
|
|
306
|
-
// path-keyed, the NamedFake review fixtures
|
|
307
|
-
//
|
|
308
|
-
// weakened, the fixture is fixed), the merge decision reads `gateSatisfied`, and the daemon writes a
|
|
307
|
+
// path-keyed, the NamedFake review fixtures author their own nonce-bound verdict (a renamed fake is
|
|
308
|
+
// a distinct responder and does not inherit the registered fake's producer contract — the check is
|
|
309
|
+
// not weakened, the fixture is fixed), the merge decision reads `gateSatisfied`, and the daemon writes a
|
|
309
310
|
// parallel round's gate-result rows in GATE_NAMES order (src/run/daemon.ts).
|
|
310
311
|
//
|
|
311
312
|
// That last one is why: retiring the switch makes fixtures that used to SKIP review journal TWO
|
|
@@ -419,9 +420,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
419
420
|
if (!v || (findings === null && (typeof v.approve !== "boolean" || !Array.isArray(v.issues)))) {
|
|
420
421
|
// OBS-196: name the cause and persist the raw bytes — a ruled-on "unparseable" without its
|
|
421
422
|
// evidence cannot be audited, and a cutoff must never be indistinguishable from a parse defect.
|
|
422
|
-
const cause = raw
|
|
423
|
-
? "empty-output"
|
|
424
|
-
: !raw.includes(nonce) ? "no-verdict" : "malformed-verdict";
|
|
423
|
+
const cause = classifyVerdictCause(raw, nonce, "approve");
|
|
425
424
|
let saved;
|
|
426
425
|
if (artifactDir) {
|
|
427
426
|
try {
|
|
@@ -432,10 +431,13 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
432
431
|
saved = undefined; // persistence is evidence, not a gate input — never fail the gate on it
|
|
433
432
|
}
|
|
434
433
|
}
|
|
434
|
+
const failure = cause === "malformed-verdict"
|
|
435
|
+
? "review output unparseable"
|
|
436
|
+
: "review dispatch failed — no structurally valid nonce-bound response; output unparseable";
|
|
435
437
|
return {
|
|
436
438
|
gate: "review",
|
|
437
439
|
pass: false,
|
|
438
|
-
details:
|
|
440
|
+
details: `${failure} (reviewer ${reviewer.adapter}:${reviewer.model}; cause: ${cause}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
|
|
439
441
|
meta: { ...policyMeta, reviewer: channelKey(reviewer), unparseable: true, cause },
|
|
440
442
|
};
|
|
441
443
|
}
|
package/dist/gates/run-gates.js
CHANGED
|
@@ -8,6 +8,7 @@ import { acceptanceGate } from "./acceptance.js";
|
|
|
8
8
|
import { compareToBaseline } from "./baseline.js";
|
|
9
9
|
import { evidenceGate } from "./evidence.js";
|
|
10
10
|
import { captureLlmOutput } from "./llm.js";
|
|
11
|
+
import { disallowedBy } from "../route/preference.js";
|
|
11
12
|
import { marginalCostRank } from "../route/router.js";
|
|
12
13
|
import { reviewGate } from "./review.js";
|
|
13
14
|
import { scopeGate } from "./scope.js";
|
|
@@ -233,6 +234,21 @@ export async function runGates(task, ctx) {
|
|
|
233
234
|
};
|
|
234
235
|
// acceptance judge — LLM spend, so everything deterministic has already passed when this runs
|
|
235
236
|
const runAcceptance = async () => {
|
|
237
|
+
// v1.87 T2: the judge is a configured seat like any other — check it against the operator's
|
|
238
|
+
// policy BEFORE spending a dispatch on it. disallowedBy carries the whole deny grammar (adapter,
|
|
239
|
+
// model, or adapter:model), so a model-scoped deny cannot slip past an adapter-id-only read.
|
|
240
|
+
const judgeDenied = disallowedBy({ adapter: ctx.cfg.judge.adapter, model: ctx.cfg.judge.model }, ctx.cfg.routing, "judge");
|
|
241
|
+
if (judgeDenied) {
|
|
242
|
+
return {
|
|
243
|
+
result: {
|
|
244
|
+
gate: "acceptance",
|
|
245
|
+
pass: false,
|
|
246
|
+
details: `judge ${channelKey({ adapter: ctx.cfg.judge.adapter, model: ctx.cfg.judge.model })} is disallowed by routing.${judgeDenied.by} (${judgeDenied.entry}) — remove the ${judgeDenied.by} entry or re-point cfg.judge at an allowed channel`,
|
|
247
|
+
meta: { judgeDisallowed: { by: judgeDenied.by, entry: judgeDenied.entry } },
|
|
248
|
+
},
|
|
249
|
+
invocations: [],
|
|
250
|
+
};
|
|
251
|
+
}
|
|
236
252
|
const judgeAdapter = getAdapter(ctx.cfg.judge.adapter, ctx.adapters);
|
|
237
253
|
const jvia = ctx.via
|
|
238
254
|
? { driver: ctx.via.driver, keep: ctx.via.keep, onSlot: ctx.via.onSlot, name: ctx.via.nameFor("judge", judgeAdapter.id), label: ctx.via.labelFor("judge") }
|
|
@@ -280,8 +296,11 @@ export async function runGates(task, ctx) {
|
|
|
280
296
|
// pickReviewer's sort (review.ts:37): TIER_RANK desc, marginalCostRank asc — proven ordering; both
|
|
281
297
|
// symbols already imported by a sibling gate file.
|
|
282
298
|
.sort((x, y) => TIER_RANK[y.tier] - TIER_RANK[x.tier] || marginalCostRank(x) - marginalCostRank(y))[0];
|
|
283
|
-
|
|
284
|
-
|
|
299
|
+
// v1.87 T2: the failover seat obeys the same policy the primary judge just passed — a denied
|
|
300
|
+
// channel is refused here too, never reached by falling through the exclusion arms below.
|
|
301
|
+
const judgePool = (ctx.judgeChannels ?? ctx.channels).filter((c) => disallowedBy(c, ctx.cfg.routing, "judge") === null);
|
|
302
|
+
const crossAdapter = pick(judgePool.filter((c) => c.adapter !== flakedAdapter));
|
|
303
|
+
const sameAdapter = pick(judgePool.filter((c) => c.adapter === flakedAdapter && channelKey(c) !== flakedKey));
|
|
285
304
|
// Prefer a different adapter; if the fleet only has one adapter, retry on a different channel of
|
|
286
305
|
// that adapter; if the fleet has only one channel, fall back to the original judge config.
|
|
287
306
|
const retry = crossAdapter ?? sameAdapter ?? { adapter: ctx.cfg.judge.adapter, model: ctx.cfg.judge.model };
|
|
@@ -0,0 +1,4 @@
|
|
|
1
|
+
export type VerdictDiscriminator = "approve" | "pass" | "ok" | "action";
|
|
2
|
+
export type VerdictUnparseableCause = "empty-output" | "no-verdict" | "malformed-verdict";
|
|
3
|
+
export declare function hasVerdictParticipationWitness(raw: string, nonce: string, discriminator: VerdictDiscriminator): boolean;
|
|
4
|
+
export declare function classifyVerdictCause(raw: string, nonce: string, discriminator: VerdictDiscriminator): VerdictUnparseableCause;
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
const MAX_WITNESS_BYTES = 4096;
|
|
2
|
+
const JSON_STRING_VALUE = String.raw `"(?:\\(?:["\\/bfnrt]|u[0-9a-fA-F]{4})|[^"\\\u0000-\u001F])*"`;
|
|
3
|
+
function escapeRegex(value) {
|
|
4
|
+
return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
5
|
+
}
|
|
6
|
+
// Pane renderers hard-wrap at physical columns and prefix continuation lines with whitespace or box
|
|
7
|
+
// chrome. Joining those renderer lines is deliberately narrower than parsing: it restores split keys,
|
|
8
|
+
// values and delimiters, but it does not require the response to be complete or valid JSON.
|
|
9
|
+
function joinRendererLines(raw) {
|
|
10
|
+
return raw.replace(/\r\n?/g, "\n").split("\n")
|
|
11
|
+
.map((line) => line.replace(/^[\t │|]+/, "").replace(/[\t │|]+$/, ""))
|
|
12
|
+
.join("");
|
|
13
|
+
}
|
|
14
|
+
// Yield only the first structural prefix of each object candidate. Stopping at the next unquoted
|
|
15
|
+
// opening brace prevents a nonce from one object binding a discriminator from another; the prompts
|
|
16
|
+
// put their discriminator before any nested object, so no valid boundary shape is lost.
|
|
17
|
+
function objectPrefixes(raw) {
|
|
18
|
+
const joined = joinRendererLines(raw);
|
|
19
|
+
const prefixes = [];
|
|
20
|
+
for (let open = joined.indexOf("{"); open !== -1; open = joined.indexOf("{", open + 1)) {
|
|
21
|
+
const ceiling = Math.min(joined.length, open + MAX_WITNESS_BYTES);
|
|
22
|
+
let end = ceiling;
|
|
23
|
+
let quoted = false;
|
|
24
|
+
let escaped = false;
|
|
25
|
+
for (let i = open + 1; i < ceiling; i++) {
|
|
26
|
+
const char = joined[i];
|
|
27
|
+
if (quoted) {
|
|
28
|
+
if (escaped)
|
|
29
|
+
escaped = false;
|
|
30
|
+
else if (char === "\\")
|
|
31
|
+
escaped = true;
|
|
32
|
+
else if (char === '"')
|
|
33
|
+
quoted = false;
|
|
34
|
+
continue;
|
|
35
|
+
}
|
|
36
|
+
if (char === '"')
|
|
37
|
+
quoted = true;
|
|
38
|
+
else if (char === "{") {
|
|
39
|
+
end = i;
|
|
40
|
+
break;
|
|
41
|
+
}
|
|
42
|
+
else if (char === "}") {
|
|
43
|
+
end = i + 1;
|
|
44
|
+
break;
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
prefixes.push(joined.slice(open, end));
|
|
48
|
+
}
|
|
49
|
+
return prefixes;
|
|
50
|
+
}
|
|
51
|
+
export function hasVerdictParticipationWitness(raw, nonce, discriminator) {
|
|
52
|
+
const noncePattern = new RegExp(String.raw `"nonce"\s*:\s*${escapeRegex(JSON.stringify(nonce))}\s*[,]`);
|
|
53
|
+
const valuePattern = discriminator === "action" ? JSON_STRING_VALUE : "(?:true|false)";
|
|
54
|
+
const discriminatorPattern = new RegExp(String.raw `"${discriminator}"\s*:\s*${valuePattern}\s*[,}]`);
|
|
55
|
+
return objectPrefixes(raw).some((prefix) => noncePattern.test(prefix) && discriminatorPattern.test(prefix));
|
|
56
|
+
}
|
|
57
|
+
export function classifyVerdictCause(raw, nonce, discriminator) {
|
|
58
|
+
if (raw.trim().length === 0)
|
|
59
|
+
return "empty-output";
|
|
60
|
+
return hasVerdictParticipationWitness(raw, nonce, discriminator)
|
|
61
|
+
? "malformed-verdict"
|
|
62
|
+
: "no-verdict";
|
|
63
|
+
}
|
package/dist/graph/schema.d.ts
CHANGED
|
@@ -12,9 +12,11 @@ export type Oracle = (typeof ORACLES)[number];
|
|
|
12
12
|
export declare const AcceptanceItemSchema: z.ZodUnion<readonly [z.ZodString, z.ZodObject<{
|
|
13
13
|
oracle: z.ZodLiteral<"command">;
|
|
14
14
|
command: z.ZodString;
|
|
15
|
+
text: z.ZodOptional<z.ZodString>;
|
|
15
16
|
}, z.core.$strip>, z.ZodObject<{
|
|
16
17
|
oracle: z.ZodLiteral<"test">;
|
|
17
18
|
test: z.ZodString;
|
|
19
|
+
text: z.ZodOptional<z.ZodString>;
|
|
18
20
|
}, z.core.$strip>, z.ZodObject<{
|
|
19
21
|
oracle: z.ZodLiteral<"judge">;
|
|
20
22
|
text: z.ZodString;
|
|
@@ -43,9 +45,11 @@ export declare const TaskSchema: z.ZodObject<{
|
|
|
43
45
|
acceptance: z.ZodArray<z.ZodUnion<readonly [z.ZodString, z.ZodObject<{
|
|
44
46
|
oracle: z.ZodLiteral<"command">;
|
|
45
47
|
command: z.ZodString;
|
|
48
|
+
text: z.ZodOptional<z.ZodString>;
|
|
46
49
|
}, z.core.$strip>, z.ZodObject<{
|
|
47
50
|
oracle: z.ZodLiteral<"test">;
|
|
48
51
|
test: z.ZodString;
|
|
52
|
+
text: z.ZodOptional<z.ZodString>;
|
|
49
53
|
}, z.core.$strip>, z.ZodObject<{
|
|
50
54
|
oracle: z.ZodLiteral<"judge">;
|
|
51
55
|
text: z.ZodString;
|
|
@@ -133,9 +137,11 @@ export declare const RunGraphSchema: z.ZodObject<{
|
|
|
133
137
|
acceptance: z.ZodArray<z.ZodUnion<readonly [z.ZodString, z.ZodObject<{
|
|
134
138
|
oracle: z.ZodLiteral<"command">;
|
|
135
139
|
command: z.ZodString;
|
|
140
|
+
text: z.ZodOptional<z.ZodString>;
|
|
136
141
|
}, z.core.$strip>, z.ZodObject<{
|
|
137
142
|
oracle: z.ZodLiteral<"test">;
|
|
138
143
|
test: z.ZodString;
|
|
144
|
+
text: z.ZodOptional<z.ZodString>;
|
|
139
145
|
}, z.core.$strip>, z.ZodObject<{
|
|
140
146
|
oracle: z.ZodLiteral<"judge">;
|
|
141
147
|
text: z.ZodString;
|
package/dist/graph/schema.js
CHANGED
|
@@ -11,11 +11,14 @@ export const TIERS = ["cheap", "mid", "frontier"];
|
|
|
11
11
|
// A plain string is the read-old/write-new compat form — semantically a judge oracle (spec §2).
|
|
12
12
|
export const ORACLES = ["command", "test", "judge"];
|
|
13
13
|
// Typed acceptance oracle: command carries the thing to run, test the test name, judge free text.
|
|
14
|
-
//
|
|
14
|
+
// text is OPTIONAL non-empty on command/test (declared prose beside the oracle; judge requires it).
|
|
15
|
+
// Declaring it matters: z.object strips unknown keys, and loadGraph revalidates on every read —
|
|
16
|
+
// an undeclared text would be silently discarded at the next load. Anything else (a typed object
|
|
17
|
+
// naming an unknown oracle, or a text that is not a non-empty string) fails validation loudly here.
|
|
15
18
|
export const AcceptanceItemSchema = z.union([
|
|
16
19
|
z.string().min(1),
|
|
17
|
-
z.object({ oracle: z.literal("command"), command: z.string().min(1) }),
|
|
18
|
-
z.object({ oracle: z.literal("test"), test: z.string().min(1) }),
|
|
20
|
+
z.object({ oracle: z.literal("command"), command: z.string().min(1), text: z.string().min(1).optional() }),
|
|
21
|
+
z.object({ oracle: z.literal("test"), test: z.string().min(1), text: z.string().min(1).optional() }),
|
|
19
22
|
z.object({ oracle: z.literal("judge"), text: z.string().min(1) }),
|
|
20
23
|
]);
|
|
21
24
|
// Shared text rendering of one acceptance item — every consumer (worker prompt, acceptance gate,
|
|
@@ -24,9 +27,9 @@ export function renderAcceptanceItem(item) {
|
|
|
24
27
|
if (typeof item === "string")
|
|
25
28
|
return item;
|
|
26
29
|
if (item.oracle === "command")
|
|
27
|
-
return `$ ${item.command}`;
|
|
30
|
+
return item.text ?? `$ ${item.command}`;
|
|
28
31
|
if (item.oracle === "test")
|
|
29
|
-
return `test: ${item.test}`;
|
|
32
|
+
return item.text ?? `test: ${item.test}`;
|
|
30
33
|
return item.text; // judge — bare text, byte-identical to a plain-string judge criterion
|
|
31
34
|
}
|
|
32
35
|
export const TaskSchema = z.object({
|
|
@@ -33,5 +33,5 @@ export interface DenyPreferCollision {
|
|
|
33
33
|
disallowed: Disallowed;
|
|
34
34
|
}
|
|
35
35
|
export declare function preferEntryDenied(p: string, cfg: TickmarkrConfig): Disallowed | null;
|
|
36
|
-
export declare function denyPreferCollisions(cfg: TickmarkrConfig): DenyPreferCollision[];
|
|
36
|
+
export declare function denyPreferCollisions(cfg: TickmarkrConfig, shapes?: Iterable<string>): DenyPreferCollision[];
|
|
37
37
|
export declare function denyPreferCollisionLine({ kind, shape, detail, disallowed }: DenyPreferCollision): string;
|
package/dist/route/preference.js
CHANGED
|
@@ -81,11 +81,18 @@ export function preferEntryDenied(p, cfg) {
|
|
|
81
81
|
throw e;
|
|
82
82
|
}
|
|
83
83
|
}
|
|
84
|
-
|
|
84
|
+
// v1.87 T3 (OBS-162): graph-aware walk. `shapes` narrows it to the shapes the caller will actually
|
|
85
|
+
// route — resume hands it the loaded graph's shape set, so a collision on a shape no resumed task
|
|
86
|
+
// uses can no longer refuse the only crash-recovery path. Omitted ⇒ the whole routing map: doctor
|
|
87
|
+
// audits the config itself, which has no graph to be scoped by.
|
|
88
|
+
export function denyPreferCollisions(cfg, shapes) {
|
|
85
89
|
if (!cfg.routing.allow && !cfg.routing.deny)
|
|
86
90
|
return [];
|
|
91
|
+
const inGraph = shapes === undefined ? undefined : new Set(shapes);
|
|
87
92
|
const out = [];
|
|
88
93
|
for (const [shape, entry] of Object.entries(cfg.routing.map)) {
|
|
94
|
+
if (inGraph && !inGraph.has(shape))
|
|
95
|
+
continue;
|
|
89
96
|
if (entry.pin) {
|
|
90
97
|
const d = disallowedBy({ adapter: entry.pin.via, model: entry.pin.model }, cfg.routing);
|
|
91
98
|
if (d) {
|
package/dist/route/router.d.ts
CHANGED
|
@@ -19,11 +19,6 @@ export interface Route {
|
|
|
19
19
|
deviation?: RouteDeviation;
|
|
20
20
|
}
|
|
21
21
|
export interface RoutingPreferContext {
|
|
22
|
-
autoPrefer?: {
|
|
23
|
-
derivedAt: string;
|
|
24
|
-
[shape: string]: string[] | string;
|
|
25
|
-
};
|
|
26
|
-
doctorFresh: boolean;
|
|
27
22
|
overlayPreferShapes: ReadonlySet<string>;
|
|
28
23
|
}
|
|
29
24
|
export interface ExploreContext {
|
package/dist/route/router.js
CHANGED
|
@@ -28,19 +28,9 @@ const exploreOff = (task, cfg, exploreCtx) => {
|
|
|
28
28
|
return true;
|
|
29
29
|
return false;
|
|
30
30
|
};
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
};
|
|
35
|
-
const preferFromAuto = (shape, preferCtx) => !!preferCtx?.doctorFresh && !!preferCtx.autoPrefer && !preferCtx.overlayPreferShapes.has(shape) &&
|
|
36
|
-
autoPreferList(preferCtx.autoPrefer, shape) !== undefined;
|
|
37
|
-
const effectivePrefer = (shape, entry, preferCtx) => {
|
|
38
|
-
if (preferCtx?.overlayPreferShapes.has(shape))
|
|
39
|
-
return entry?.prefer;
|
|
40
|
-
if (preferCtx?.doctorFresh && preferCtx.autoPrefer)
|
|
41
|
-
return autoPreferList(preferCtx.autoPrefer, shape) ?? entry?.prefer;
|
|
42
|
-
return entry?.prefer;
|
|
43
|
-
};
|
|
31
|
+
// v1.86 T3: autoPrefer is deleted — prefer is operator-declared only. Nothing is ever derived
|
|
32
|
+
// from probe health, tier data, or a machine-built shape table.
|
|
33
|
+
const effectivePrefer = (entry) => entry?.prefer;
|
|
44
34
|
export class RoutingError extends Error {
|
|
45
35
|
constructor(msg) {
|
|
46
36
|
super(msg);
|
|
@@ -134,7 +124,7 @@ export function route(task, cfg, channels, profile, preferCtx, exclude, exploreC
|
|
|
134
124
|
// ponytail: sla is plan-time advisory only — never thread into learnedScore (would reroute warm rivals).
|
|
135
125
|
const scoreOpts = { availWeight: cfg.routing.learnedTuning?.availWeight };
|
|
136
126
|
const entry = cfg.routing.map[task.shape];
|
|
137
|
-
const prefer = effectivePrefer(
|
|
127
|
+
const prefer = effectivePrefer(entry);
|
|
138
128
|
const prefActive = !!(cfg.routing.allow || cfg.routing.deny);
|
|
139
129
|
const disallowedPin = (via, model, kind) => {
|
|
140
130
|
const d = disallowedBy({ adapter: via, model }, cfg.routing);
|
|
@@ -167,6 +157,14 @@ export function route(task, cfg, channels, profile, preferCtx, exclude, exploreC
|
|
|
167
157
|
}
|
|
168
158
|
};
|
|
169
159
|
const taskFloor = task.routingHints?.floor;
|
|
160
|
+
const mapPinFloor = taskFloor && (!advisoryFloor || TIER_RANK[taskFloor] >= TIER_RANK[advisoryFloor])
|
|
161
|
+
? { tier: taskFloor, source: "task" }
|
|
162
|
+
: advisoryFloor ? { tier: advisoryFloor, source: "config" } : undefined;
|
|
163
|
+
const lintMapPinFloor = (tier) => {
|
|
164
|
+
if (mapPinFloor && TIER_RANK[tier] < TIER_RANK[mapPinFloor.tier]) {
|
|
165
|
+
lints.push(`${task.id} (${task.shape}): map pin routes ${tier}, below ${mapPinFloor.source} floor ${mapPinFloor.tier} — map pins are supreme`);
|
|
166
|
+
}
|
|
167
|
+
};
|
|
170
168
|
const source = task.routingHints?.source;
|
|
171
169
|
const src = source ? `, ${source}` : ""; // never interpolate a possibly-undefined source
|
|
172
170
|
// task pin: planner-authored, try-first — degrades on miss or below-floor (D-05, research A3), never throws
|
|
@@ -188,12 +186,13 @@ export function route(task, cfg, channels, profile, preferCtx, exclude, exploreC
|
|
|
188
186
|
if (entry?.pin) {
|
|
189
187
|
disallowedPin(entry.pin.via, entry.pin.model, "map pin (config routing.map)");
|
|
190
188
|
const c = resolvePin(entry.pin, channels);
|
|
191
|
-
|
|
189
|
+
lintMapPinFloor(c.tier);
|
|
192
190
|
maybeSlaLint(lints, task, profile, slaMinutes, c);
|
|
193
191
|
return { assignment: toAssignment(c), ladder: ladderFor(task, entry), lints, provenance: `${degraded}pin ${entry.pin.via}:${entry.pin.model} (config routing.map)` };
|
|
194
192
|
}
|
|
195
193
|
const baseTier = floor ?? "cheap";
|
|
196
|
-
|
|
194
|
+
// D-04: a task floor is hard only on floor/auto paths; the map-pin branch above stays supreme.
|
|
195
|
+
const minTier = taskFloor && TIER_RANK[taskFloor] > TIER_RANK[baseTier] ? taskFloor : baseTier;
|
|
197
196
|
if (prefActive)
|
|
198
197
|
for (const p of prefer ?? [])
|
|
199
198
|
preflightPrefer(p);
|
|
@@ -293,13 +292,10 @@ export function route(task, cfg, channels, profile, preferCtx, exclude, exploreC
|
|
|
293
292
|
"tier cheap (default)";
|
|
294
293
|
// name the key that actually broke the tie: prefer outranks the marginal-cost/tier keys, so if the
|
|
295
294
|
// winner matched a prefer entry, prefer decided it — not "cheapest sufficient tier" (ROUTE-03, WR-01)
|
|
296
|
-
const preferVia = preferFromAuto(task.shape, preferCtx)
|
|
297
|
-
? `via prefer (auto-modernized ${preferCtx.autoPrefer.derivedAt.slice(0, 10)})`
|
|
298
|
-
: "via prefer";
|
|
299
295
|
// a spread-decided winner is never inside a prefer band (the spread skips those runs), so the
|
|
300
296
|
// three arms below are mutually exclusive by construction
|
|
301
297
|
const chosenBy = learnedChosen || (spreadDecided ? "via frontier spread"
|
|
302
|
-
: prefer && preferIndex(eligible[0], prefer) < prefer.length ?
|
|
298
|
+
: prefer && preferIndex(eligible[0], prefer) < prefer.length ? "via prefer" : "cheapest sufficient tier");
|
|
303
299
|
maybeSlaLint(lints, task, profile, slaMinutes, eligible[0]);
|
|
304
300
|
return { assignment: toAssignment(eligible[0]), ladder: ladderFor(task, entry), lints, provenance: `${degraded}${bound}, marginal-cost auto (${chosenBy})`, ...(deviation ? { deviation } : {}) };
|
|
305
301
|
}
|
package/dist/run/consult.d.ts
CHANGED
|
@@ -2,6 +2,7 @@ import type { WorkerAdapter } from "../adapters/types.js";
|
|
|
2
2
|
import type { TickmarkrConfig } from "../config/config.js";
|
|
3
3
|
import type { ExecutorDriver, Slot } from "../drivers/types.js";
|
|
4
4
|
import type { GateResult } from "../gates/types.js";
|
|
5
|
+
import { type VerdictUnparseableCause } from "../gates/verdict-cause.js";
|
|
5
6
|
export interface ConsultVerdict {
|
|
6
7
|
action: "retry" | "reroute" | "decompose" | "human";
|
|
7
8
|
notes: string;
|
|
@@ -23,6 +24,11 @@ export interface Dossier {
|
|
|
23
24
|
diff: string;
|
|
24
25
|
gates: GateResult[];
|
|
25
26
|
}
|
|
27
|
+
export interface ConsultParseResult {
|
|
28
|
+
verdict: ConsultVerdict | null;
|
|
29
|
+
cause?: VerdictUnparseableCause;
|
|
30
|
+
}
|
|
31
|
+
export declare function parseConsultVerdict(out: string, nonce: string): ConsultParseResult;
|
|
26
32
|
export declare function buildDossierPrompt(d: Dossier, nonce: string): string;
|
|
27
33
|
export declare function consult(d: Dossier, cfg: TickmarkrConfig, adapters: WorkerAdapter[], driver: ExecutorDriver, cwd: string, runDir: string, opts?: {
|
|
28
34
|
keep?: boolean;
|
package/dist/run/consult.js
CHANGED
|
@@ -2,7 +2,9 @@ import { mkdirSync, writeFileSync } from "node:fs";
|
|
|
2
2
|
import { join } from "node:path";
|
|
3
3
|
import { getAdapter } from "../adapters/registry.js";
|
|
4
4
|
import { bannerShell, paneDispatchCommand } from "../brand.js";
|
|
5
|
-
import {
|
|
5
|
+
import { extractVerdictJson, gateExitTrailer, gatePaneName, generateVerdictNonce, verdictNonceLine } from "../gates/llm.js";
|
|
6
|
+
import { classifyVerdictCause } from "../gates/verdict-cause.js";
|
|
7
|
+
import { disallowedBy } from "../route/preference.js";
|
|
6
8
|
import { sh } from "./git.js";
|
|
7
9
|
import { redactSecrets } from "./redact.js";
|
|
8
10
|
import { filterLlmTranscript } from "./stall.js";
|
|
@@ -57,6 +59,33 @@ export function augmentRetryBrief(feedback, opts) {
|
|
|
57
59
|
return parts.join("\n\n");
|
|
58
60
|
}
|
|
59
61
|
const ACTIONS = ["retry", "reroute", "decompose", "human"];
|
|
62
|
+
export function parseConsultVerdict(out, nonce) {
|
|
63
|
+
const v = extractVerdictJson(out, nonce);
|
|
64
|
+
if (!v)
|
|
65
|
+
return { verdict: null, cause: classifyVerdictCause(out, nonce, "action") };
|
|
66
|
+
// A parsed object with an unknown action is a content rejection, not silence. It keeps the existing
|
|
67
|
+
// null result without manufacturing a classifier cause for a state the classifier never saw.
|
|
68
|
+
if (!ACTIONS.includes(v.action))
|
|
69
|
+
return { verdict: null };
|
|
70
|
+
// fail-closed exclusion: only a non-empty string survives. Malformed values (number/array/object/
|
|
71
|
+
// empty) are dropped — the verdict stays a normal channel-level reroute/retry/…, never a crash
|
|
72
|
+
// and never silently forced to human. Unknown adapter ids pass through; the daemon treats a
|
|
73
|
+
// zero-match expansion as channel-level reroute.
|
|
74
|
+
const raw = v.excludeAdapter;
|
|
75
|
+
const excludeAdapter = typeof raw === "string" && raw.length > 0 ? raw : undefined;
|
|
76
|
+
const reason = typeof v.reason === "string" ? v.reason : undefined;
|
|
77
|
+
const guidance = typeof v.guidance === "string" ? v.guidance : undefined;
|
|
78
|
+
const notes = String(v.notes ?? guidance ?? reason ?? "");
|
|
79
|
+
return {
|
|
80
|
+
verdict: {
|
|
81
|
+
action: v.action,
|
|
82
|
+
notes,
|
|
83
|
+
...(reason ? { reason } : {}),
|
|
84
|
+
...(guidance ? { guidance } : {}),
|
|
85
|
+
...(excludeAdapter ? { excludeAdapter } : {}),
|
|
86
|
+
},
|
|
87
|
+
};
|
|
88
|
+
}
|
|
60
89
|
// v1.65 T2: transcript noise (spinner repaints, CR churn, pass-run spam) is squashed at build time —
|
|
61
90
|
// the filtered form is what persists to the consults/ artifact AND what the model reads; fail-open.
|
|
62
91
|
export function buildDossierPrompt(d, nonce) {
|
|
@@ -108,7 +137,8 @@ opts = {}) {
|
|
|
108
137
|
// consult model reads) is masked at this seam; the in-memory Dossier stays untouched.
|
|
109
138
|
writeFileSync(promptFile, redactSecrets(buildDossierPrompt(d, nonce)));
|
|
110
139
|
// One seat = the WHOLE invoke-and-parse unit, both visibility branches (OBS-69 class: a headless-only
|
|
111
|
-
// failover would leave the production pane path hard-failing on seat one). null
|
|
140
|
+
// failover would leave the production pane path hard-failing on seat one). A null verdict retains
|
|
141
|
+
// the classifier's cause when extraction failed; parsed content rejections deliberately have none.
|
|
112
142
|
const invokeSeat = async (seatAdapter, seatModel, seatIdx) => {
|
|
113
143
|
const adapter = getAdapter(seatAdapter, adapters);
|
|
114
144
|
let out;
|
|
@@ -143,26 +173,7 @@ opts = {}) {
|
|
|
143
173
|
await driver.close(slot);
|
|
144
174
|
}
|
|
145
175
|
}
|
|
146
|
-
|
|
147
|
-
const v = extractVerdictJson(out, nonce);
|
|
148
|
-
if (!v || !ACTIONS.includes(v.action))
|
|
149
|
-
return null;
|
|
150
|
-
// fail-closed exclusion: only a non-empty string survives. Malformed values (number/array/object/
|
|
151
|
-
// empty) are dropped — the verdict stays a normal channel-level reroute/retry/…, never a crash
|
|
152
|
-
// and never silently forced to human. Unknown adapter ids pass through; the daemon treats a
|
|
153
|
-
// zero-match expansion as channel-level reroute.
|
|
154
|
-
const raw = v.excludeAdapter;
|
|
155
|
-
const excludeAdapter = typeof raw === "string" && raw.length > 0 ? raw : undefined;
|
|
156
|
-
const reason = typeof v.reason === "string" ? v.reason : undefined;
|
|
157
|
-
const guidance = typeof v.guidance === "string" ? v.guidance : undefined;
|
|
158
|
-
const notes = String(v.notes ?? guidance ?? reason ?? "");
|
|
159
|
-
return {
|
|
160
|
-
action: v.action,
|
|
161
|
-
notes,
|
|
162
|
-
...(reason ? { reason } : {}),
|
|
163
|
-
...(guidance ? { guidance } : {}),
|
|
164
|
-
...(excludeAdapter ? { excludeAdapter } : {}),
|
|
165
|
-
};
|
|
176
|
+
return parseConsultVerdict(out, nonce);
|
|
166
177
|
};
|
|
167
178
|
// v1.54 T1: ranked seat failover. Walk consult.prefer (adapter:model entries) to the first entry
|
|
168
179
|
// whose adapter is in the live channel set; a failed seat or unparseable verdict falls to the next;
|
|
@@ -174,11 +185,23 @@ opts = {}) {
|
|
|
174
185
|
.map((entry) => ({ adapter: entry.slice(0, entry.indexOf(":")), model: entry.slice(entry.indexOf(":") + 1) }))
|
|
175
186
|
.filter((s) => live.has(s.adapter));
|
|
176
187
|
seats.push({ adapter: cfg.consult.adapter, model: cfg.consult.model });
|
|
177
|
-
|
|
188
|
+
// v1.87 T2: a consult seat reads the code, so it passes through the operator's policy exactly like
|
|
189
|
+
// a worker does. The filter runs AFTER the pin is pushed, so the final pinned seat is checked by
|
|
190
|
+
// the same rule as every prefer entry — and disallowedBy carries the full deny grammar (adapter,
|
|
191
|
+
// model, or adapter:model), so a model-scoped deny cannot slip past an adapter-id-only read.
|
|
192
|
+
const allowedSeats = seats.filter((s) => disallowedBy(s, cfg.routing, "consult") === null);
|
|
193
|
+
if (!allowedSeats.length) {
|
|
194
|
+
const d = disallowedBy(seats[seats.length - 1], cfg.routing, "consult");
|
|
195
|
+
return {
|
|
196
|
+
action: "human",
|
|
197
|
+
notes: `every consult seat is disallowed by routing.${d.by} (${d.entry}) — failing safe to human`,
|
|
198
|
+
};
|
|
199
|
+
}
|
|
200
|
+
for (const [i, seat] of allowedSeats.entries()) {
|
|
178
201
|
try {
|
|
179
|
-
const
|
|
180
|
-
if (
|
|
181
|
-
return
|
|
202
|
+
const parsed = await invokeSeat(seat.adapter, seat.model, i);
|
|
203
|
+
if (parsed.verdict)
|
|
204
|
+
return parsed.verdict;
|
|
182
205
|
}
|
|
183
206
|
catch {
|
|
184
207
|
// failed seat (unknown adapter, dead driver/pane, shell error) — fall to the next entry
|