tickmarkr 2.5.2 → 2.5.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/qwen.d.ts +1 -0
- package/dist/adapters/qwen.js +8 -0
- package/dist/adapters/types.d.ts +1 -0
- package/dist/cli/commands/doctor.d.ts +10 -0
- package/dist/cli/commands/doctor.js +50 -1
- package/dist/cli/commands/plan.js +144 -3
- package/dist/cli/commands/resume.js +2 -0
- package/dist/cli/commands/run.js +12 -0
- package/dist/cli/commands/status.js +8 -14
- package/dist/compile/collateral.d.ts +2 -0
- package/dist/compile/collateral.js +50 -9
- package/dist/compile/ownership.d.ts +16 -0
- package/dist/compile/ownership.js +113 -13
- package/dist/config/config.d.ts +9 -0
- package/dist/config/config.js +13 -2
- package/dist/drivers/herdr.d.ts +5 -0
- package/dist/drivers/herdr.js +14 -3
- package/dist/drivers/types.d.ts +1 -0
- package/dist/gates/baseline.d.ts +5 -1
- package/dist/gates/baseline.js +18 -5
- package/dist/gates/llm.d.ts +1 -1
- package/dist/gates/llm.js +34 -12
- package/dist/gates/review.d.ts +46 -2
- package/dist/gates/review.js +111 -21
- package/dist/gates/run-gates.d.ts +2 -0
- package/dist/gates/run-gates.js +19 -10
- package/dist/route/router.d.ts +13 -0
- package/dist/route/router.js +64 -11
- package/dist/run/daemon.d.ts +3 -1
- package/dist/run/daemon.js +332 -53
- package/dist/run/git.d.ts +25 -0
- package/dist/run/git.js +68 -1
- package/dist/run/journal.js +16 -7
- package/dist/run/operator-state.d.ts +1 -1
- package/dist/run/operator-state.js +5 -9
- package/dist/tui/cockpit/live-runtime.js +12 -10
- package/package.json +1 -1
- package/skills/tickmarkr-auto/SKILL.md +1 -1
- package/skills/tickmarkr-loop/SKILL.md +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +82 -4
- package/skills/tickmarkr-overseer/scripts/watch-launch.sh +38 -0
package/dist/gates/review.js
CHANGED
|
@@ -1,9 +1,9 @@
|
|
|
1
|
-
import { writeFileSync } from "node:fs";
|
|
1
|
+
import { existsSync, writeFileSync } from "node:fs";
|
|
2
2
|
import { join } from "node:path";
|
|
3
3
|
import { channelKey, shq } from "../adapters/types.js";
|
|
4
4
|
import { criticalPathHits, DEFAULT_DIFF_CAP, DEFAULT_REVIEW_CRITICAL_PATHS, declaredReviewPolicy, isReviewLeafPath, raiseReviewPolicy, REVIEW_VERSION_MIRRORS, TIER_RANK, } from "../config/config.js";
|
|
5
5
|
import { filesGlob } from "../graph/files-glob.js";
|
|
6
|
-
import { renderAcceptanceItem } from "../graph/schema.js";
|
|
6
|
+
import { renderAcceptanceItem, TIERS } from "../graph/schema.js";
|
|
7
7
|
import { getAdapter } from "../adapters/registry.js";
|
|
8
8
|
import { shOk } from "../run/git.js";
|
|
9
9
|
import { structuredFindings } from "../run/journal.js";
|
|
@@ -168,6 +168,31 @@ export function modelId(model) {
|
|
|
168
168
|
return model.slice(model.lastIndexOf("/") + 1);
|
|
169
169
|
}
|
|
170
170
|
export { modelProvider };
|
|
171
|
+
export function matchClosureId(candidate, target) {
|
|
172
|
+
if (typeof candidate !== "string")
|
|
173
|
+
return typeof target === "string" ? false : undefined;
|
|
174
|
+
const normCandidate = candidate.replace(/\s+/g, "");
|
|
175
|
+
if (typeof target === "string") {
|
|
176
|
+
return normCandidate === target.replace(/\s+/g, "");
|
|
177
|
+
}
|
|
178
|
+
for (const fp of target) {
|
|
179
|
+
if (typeof fp === "string" && normCandidate === fp.replace(/\s+/g, ""))
|
|
180
|
+
return fp;
|
|
181
|
+
}
|
|
182
|
+
return undefined;
|
|
183
|
+
}
|
|
184
|
+
/**
|
|
185
|
+
* Validates closure ids in a review verdict: membership, duplication, and coverage of every prior id
|
|
186
|
+
* all route through matchClosureId.
|
|
187
|
+
*/
|
|
188
|
+
export function isReviewClosureInvalid(v, priorIds) {
|
|
189
|
+
const priors = priorIds instanceof Set ? priorIds : new Set(priorIds);
|
|
190
|
+
const closureLists = [v?.resolved, v?.reraised];
|
|
191
|
+
const allCandidateIds = [...(v?.resolved ?? []), ...(v?.reraised ?? [])];
|
|
192
|
+
return !!v && (priors.size > 0 || closureLists.some((list) => list !== undefined)) && (closureLists.some((list) => !Array.isArray(list) || list.some((id) => !matchClosureId(id, priors)))
|
|
193
|
+
|| new Set(allCandidateIds.map((id) => matchClosureId(id, priors) ?? id)).size !== allCandidateIds.length
|
|
194
|
+
|| [...priors].some((id) => !allCandidateIds.some((candidate) => matchClosureId(candidate, id))));
|
|
195
|
+
}
|
|
171
196
|
// v1.53 T2: same entry grammar as routing.map.prefer (router.ts preferIndex — router is out of this
|
|
172
197
|
// module's dependency direction for a private fn, so the 3 lines live here too): `adapter` matches
|
|
173
198
|
// every channel of that adapter, `adapter:model` exactly one; unmatched channels sort after all entries.
|
|
@@ -175,9 +200,52 @@ function reviewPreferIndex(c, prefer) {
|
|
|
175
200
|
const i = prefer.findIndex((p) => p === c.adapter || p === channelKey(c));
|
|
176
201
|
return i === -1 ? prefer.length : i;
|
|
177
202
|
}
|
|
203
|
+
/**
|
|
204
|
+
* RF-1 (OBS-922 add.2/3): the tier a reviewer must meet is the maximum of the author's tier, the
|
|
205
|
+
* task-declared floor, a configured `review.floor` tier and, on a second round or a retry, the prior
|
|
206
|
+
* reviewer's tier. The cause names the input that reached the maximum (earlier inputs win a tie, so a
|
|
207
|
+
* floor the author's tier already satisfies is attributed to the author).
|
|
208
|
+
*/
|
|
209
|
+
export function resolveReviewerFloor(authorTier, taskFloor, configFloor, priorReviewerTier) {
|
|
210
|
+
const inputs = [
|
|
211
|
+
[authorTier, "author-tier"], [taskFloor, "task-floor"], [configFloor, "config"], [priorReviewerTier, "prior-reviewer"],
|
|
212
|
+
];
|
|
213
|
+
let best = { floor: authorTier, cause: "author-tier" };
|
|
214
|
+
for (const [tier, cause] of inputs) {
|
|
215
|
+
if (tier !== undefined && TIER_RANK[tier] > TIER_RANK[best.floor])
|
|
216
|
+
best = { floor: tier, cause };
|
|
217
|
+
}
|
|
218
|
+
return best;
|
|
219
|
+
}
|
|
220
|
+
/**
|
|
221
|
+
* The highest tier among the task's prior reviewers — the prior reviewer's tier for RF-1. A recorded
|
|
222
|
+
* dispatch tier is historical evidence and wins over the current pool; an unrecorded one falls back to
|
|
223
|
+
* the seat's channel; a seat neither establishes (it left the pool on resume, or the journal holds
|
|
224
|
+
* garbage) holds frontier — fail closed, never silently dropped.
|
|
225
|
+
*/
|
|
226
|
+
export function priorReviewerTier(channels, priorReviewers = []) {
|
|
227
|
+
let top;
|
|
228
|
+
for (const p of priorReviewers) {
|
|
229
|
+
const key = typeof p === "string" ? p : p.reviewer;
|
|
230
|
+
const seen = (typeof p === "string" ? undefined : p.tier) ?? channels.find((ch) => channelKey(ch) === key)?.tier;
|
|
231
|
+
const tier = TIERS.includes(seen) ? seen : "frontier";
|
|
232
|
+
if (top === undefined || TIER_RANK[tier] > TIER_RANK[top])
|
|
233
|
+
top = tier;
|
|
234
|
+
}
|
|
235
|
+
return top;
|
|
236
|
+
}
|
|
237
|
+
/**
|
|
238
|
+
* The gate's floor: author tier, task floor, review.floor (a tier — `worker` names none) and the seats
|
|
239
|
+
* the caller names as THIS TASK's prior reviewers (earlier rounds' seats, a flaked seat). Eligibility
|
|
240
|
+
* exclusions are NOT evidence — a retry bans a flaked seat's whole adapter, and those sibling channels
|
|
241
|
+
* never reviewed — and neither is the run-scoped LRU rotation history, which names unrelated tasks' seats.
|
|
242
|
+
*/
|
|
243
|
+
export function gateReviewerFloor(task, cfg, author, channels, priorReviewers = []) {
|
|
244
|
+
return resolveReviewerFloor(author.tier, task.routingHints?.floor, cfg.review.floor === "worker" ? undefined : cfg.review.floor, priorReviewerTier(channels, priorReviewers));
|
|
245
|
+
}
|
|
178
246
|
export function pickReviewer(author, channels, exclude = [], // v1.1 failover: reviewer channels that already produced garbage for this task
|
|
179
247
|
prefer = [], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
|
|
180
|
-
floor, // task
|
|
248
|
+
floor, // task/config/prior floor from the caller; the author's own tier is ALWAYS applied here (RF-1)
|
|
181
249
|
history = [], // run-scoped picks, oldest to newest; empty preserves the established ranking
|
|
182
250
|
onSeat, demoted = new Set()) {
|
|
183
251
|
// FLEET-05 success criterion 2: an author not resolvable in the channel list yields NO reviewer.
|
|
@@ -188,6 +256,8 @@ onSeat, demoted = new Set()) {
|
|
|
188
256
|
if (!authorChannel)
|
|
189
257
|
return null;
|
|
190
258
|
const authorProvider = modelProvider(author.model, authorChannel.vendor);
|
|
259
|
+
// RF-1: every caller inherits the author-tier floor — a reviewer is never seated below its author.
|
|
260
|
+
const effectiveFloor = resolveReviewerFloor(author.tier, floor).floor;
|
|
191
261
|
const ranked = channels
|
|
192
262
|
// Three independent axes: different vendor, different resolved provider identity (OBS-946: on initial pick
|
|
193
263
|
// as well as failover, so an aggregator channel stamped "mixed" never seats the author's own provider),
|
|
@@ -197,7 +267,7 @@ onSeat, demoted = new Set()) {
|
|
|
197
267
|
&& modelProvider(c.model, c.vendor) !== authorProvider
|
|
198
268
|
&& modelId(c.model) !== modelId(author.model)
|
|
199
269
|
&& !exclude.includes(channelKey(c))
|
|
200
|
-
&&
|
|
270
|
+
&& TIER_RANK[c.tier] >= TIER_RANK[effectiveFloor])
|
|
201
271
|
.sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b));
|
|
202
272
|
const reviewer = [...ranked].sort((a, b) => Number(demoted.has(channelKey(a))) - Number(demoted.has(channelKey(b)))
|
|
203
273
|
|| history.lastIndexOf(channelKey(a)) - history.lastIndexOf(channelKey(b))
|
|
@@ -222,7 +292,10 @@ ${files.map((path) => `- ${path}`).join("\n")}`;
|
|
|
222
292
|
export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
|
|
223
293
|
// OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
|
|
224
294
|
// direct tests) skips persistence and changes nothing else.
|
|
225
|
-
artifactDir, reviewHistory, demotedReviewers, carriedFindings = []
|
|
295
|
+
artifactDir, reviewHistory, demotedReviewers, carriedFindings = [],
|
|
296
|
+
// RF-1: channel keys of THIS task's prior reviewers (earlier rounds, a flaked seat) — task-scoped,
|
|
297
|
+
// never the run-wide rotation history nor excludeReviewers; the seat holds the highest of their tiers.
|
|
298
|
+
priorReviewers = []) {
|
|
226
299
|
// R3 (OBS-186): participation is keyed on PATHS. The compiler's assignment comes from the DECLARED
|
|
227
300
|
// files[]; the operator's floor may RAISE it to full and can never lower it. `complexityThreshold` is
|
|
228
301
|
// retired — the branch that returned a green skip on a complexity comparison is gone, and with it the
|
|
@@ -292,20 +365,19 @@ artifactDir, reviewHistory, demotedReviewers, carriedFindings = []) {
|
|
|
292
365
|
policy: "full",
|
|
293
366
|
...(promotedBy ? { promotedFrom: declaredPolicy, promotedBy } : {}),
|
|
294
367
|
};
|
|
295
|
-
//
|
|
296
|
-
//
|
|
297
|
-
const reviewerFloor = task
|
|
368
|
+
// RF-1: the floor is max(author tier, task floor, review.floor tier, prior reviewer's tier). Only
|
|
369
|
+
// review.floor is read from config — cfg.routing.floors governs workers and never moves review seats.
|
|
370
|
+
const { floor: reviewerFloor, cause: reviewerFloorCause } = gateReviewerFloor(task, cfg, author, channels, priorReviewers);
|
|
371
|
+
const floorMeta = { reviewerFloor, reviewerFloorCause };
|
|
298
372
|
let rotationSeat;
|
|
299
373
|
const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor, reviewHistory, reviewHistory ? (seat) => { rotationSeat = seat; } : undefined, demotedReviewers);
|
|
300
374
|
if (!reviewer) {
|
|
301
375
|
// meta.noEligibleReviewer lets run-gates' review-retry keep the ORIGINAL unparseable result when
|
|
302
376
|
// the retry finds no second seat — a truthful cause beats a synthetic no-reviewer failure.
|
|
303
|
-
const reason = reviewerFloor
|
|
304
|
-
? `no cross-vendor reviewer available at or above task-declared ${reviewerFloor} floor (diversity rule)`
|
|
305
|
-
: "no cross-vendor reviewer available (diversity rule)";
|
|
377
|
+
const reason = `no cross-vendor reviewer available at or above ${reviewerFloor} floor (${reviewerFloorCause}; diversity rule)`;
|
|
306
378
|
return cfg.review.required || priorMaterials.length > 0
|
|
307
|
-
? { gate: "review", pass: false, details: `unreadable — ${reason}; ${priorMaterials.length ? "carried materials require a review verdict" : "set review.required:false to waive"}`, meta: { noEligibleReviewer: true, unreadable: true, ...
|
|
308
|
-
: { gate: "review", pass: true, details: `WARNING: ${reason} — review waived by config`, meta: { noEligibleReviewer: true, ...
|
|
379
|
+
? { gate: "review", pass: false, details: `unreadable — ${reason}; ${priorMaterials.length ? "carried materials require a review verdict" : "set review.required:false to waive"}`, meta: { noEligibleReviewer: true, unreadable: true, ...floorMeta } }
|
|
380
|
+
: { gate: "review", pass: true, details: `WARNING: ${reason} — review waived by config`, meta: { noEligibleReviewer: true, ...floorMeta } };
|
|
309
381
|
}
|
|
310
382
|
reviewHistory?.push(channelKey(reviewer));
|
|
311
383
|
const rotationMeta = rotationSeat === undefined ? {} : { rotationSeat };
|
|
@@ -344,6 +416,7 @@ Classify every concern as "material" (a correctness, security, or acceptance-cri
|
|
|
344
416
|
block the merge) or "minor" (style, naming, or preference that should not block). ONLY material findings
|
|
345
417
|
block approval. For a minor concern you have decided not to block on, set "defer": true and give a
|
|
346
418
|
one-line "rationale" — it is recorded in the review, never dropped.
|
|
419
|
+
A fix you prescribe that would break suites outside the task's declared write scope (files[]) is a scope finding, never a material one.
|
|
347
420
|
|
|
348
421
|
Respond with ONLY this JSON:
|
|
349
422
|
{"nonce": "${nonce}", "approve": true|false, "resolved": [], "reraised": [], "findings": [{"note": "...", "severity": "material"|"minor", "defer": false, "rationale": ""}], "comments": [{"path": "path/to/file", "line": 42, "body": "actionable feedback"}]}
|
|
@@ -357,7 +430,19 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
357
430
|
// make two otherwise-identical runs diverge in their journal bytes. The reviewer channel already
|
|
358
431
|
// disambiguates every call that matters: a retry always excludes the flaked channel (run-gates.ts),
|
|
359
432
|
// so it can never collide with the attempt it replaces.
|
|
360
|
-
const
|
|
433
|
+
const baseArtifactId = `${task.id}-${channelKey(reviewer).replace(/[^a-zA-Z0-9_.-]/g, "-")}`;
|
|
434
|
+
let artifactId = baseArtifactId;
|
|
435
|
+
if (artifactDir) {
|
|
436
|
+
if (existsSync(join(artifactDir, `review-brief-${baseArtifactId}.md`)) ||
|
|
437
|
+
existsSync(join(artifactDir, `review-raw-${baseArtifactId}.txt`))) {
|
|
438
|
+
let counter = 2;
|
|
439
|
+
while (existsSync(join(artifactDir, `review-brief-${baseArtifactId}-${counter}.md`)) ||
|
|
440
|
+
existsSync(join(artifactDir, `review-raw-${baseArtifactId}-${counter}.txt`))) {
|
|
441
|
+
counter++;
|
|
442
|
+
}
|
|
443
|
+
artifactId = `${baseArtifactId}-${counter}`;
|
|
444
|
+
}
|
|
445
|
+
}
|
|
361
446
|
const briefPath = artifactDir ? join(artifactDir, `review-brief-${artifactId}.md`) : undefined;
|
|
362
447
|
// Persistence is evidence, not a gate input: a full disk or a removed run dir never fails the gate.
|
|
363
448
|
let savedBrief;
|
|
@@ -397,10 +482,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
397
482
|
const v = extractVerdictJson(raw, nonce);
|
|
398
483
|
const findings = v && Array.isArray(v.findings) ? v.findings : null;
|
|
399
484
|
const priorIds = new Set(priorMaterials.map((finding) => finding.fingerprint));
|
|
400
|
-
const
|
|
401
|
-
const closureInvalid = !!v && (priorIds.size > 0 || closureLists.some((list) => list !== undefined)) && (closureLists.some((list) => !Array.isArray(list) || list.some((id) => typeof id !== "string" || !priorIds.has(id)))
|
|
402
|
-
|| new Set([...(v?.resolved ?? []), ...(v?.reraised ?? [])]).size !== (v?.resolved?.length ?? 0) + (v?.reraised?.length ?? 0)
|
|
403
|
-
|| [...priorIds].some((id) => !v?.resolved?.includes(id) && !v?.reraised?.includes(id)));
|
|
485
|
+
const closureInvalid = isReviewClosureInvalid(v, priorIds);
|
|
404
486
|
// findings decides the verdict on its own; the legacy path still needs approve + issues to parse.
|
|
405
487
|
if (!v || closureInvalid || (findings === null && (typeof v.approve !== "boolean" || !Array.isArray(v.issues)))) {
|
|
406
488
|
// OBS-196: name the cause and persist the raw bytes — a ruled-on "unparseable" without its
|
|
@@ -419,7 +501,9 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
419
501
|
meta: {
|
|
420
502
|
...policyMeta,
|
|
421
503
|
...rotationMeta,
|
|
504
|
+
...floorMeta,
|
|
422
505
|
reviewer: channelKey(reviewer),
|
|
506
|
+
reviewerTier: reviewer.tier,
|
|
423
507
|
vendor: reviewer.vendor,
|
|
424
508
|
provider,
|
|
425
509
|
...(cause === "malformed-verdict" ? { unparseable: true } : { noVerdict: true, classification: "infra", infra: true }),
|
|
@@ -434,7 +518,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
434
518
|
const decided = findings !== null
|
|
435
519
|
? classifyReviewFindings(findings)
|
|
436
520
|
: classifyReviewIssues(v.approve, v.issues);
|
|
437
|
-
const reraised = priorMaterials.filter((finding) => v.reraised?.
|
|
521
|
+
const reraised = priorMaterials.filter((finding) => v.reraised?.some((id) => matchClosureId(id, finding.fingerprint)));
|
|
438
522
|
if (reraised.length) {
|
|
439
523
|
if (decided.pass)
|
|
440
524
|
decided.headline = "requested changes";
|
|
@@ -454,8 +538,14 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
454
538
|
pass: decided.pass,
|
|
455
539
|
details,
|
|
456
540
|
meta: {
|
|
457
|
-
...policyMeta, ...rotationMeta, reviewer: channelKey(reviewer), vendor: reviewer.vendor, provider,
|
|
458
|
-
|
|
541
|
+
...policyMeta, ...rotationMeta, ...floorMeta, reviewer: channelKey(reviewer), reviewerTier: reviewer.tier, vendor: reviewer.vendor, provider,
|
|
542
|
+
// OBS-990 b: the verbatim ids plus ONE normalised copy of each list — never a third alias.
|
|
543
|
+
...(priorMaterials.length ? {
|
|
544
|
+
resolved: v.resolved,
|
|
545
|
+
reraised: v.reraised,
|
|
546
|
+
resolvedMatches: (v.resolved ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
|
|
547
|
+
reraisedMatches: (v.reraised ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
|
|
548
|
+
} : {}),
|
|
459
549
|
...(reraised.length ? { findings: [
|
|
460
550
|
...structuredFindings("review", details).filter((finding) => !reraised.some((prior) => prior.note === finding.note)),
|
|
461
551
|
...reraised,
|
|
@@ -3,6 +3,7 @@ import { type TickmarkrConfig } from "../config/config.js";
|
|
|
3
3
|
import { type GateName, type Task } from "../graph/schema.js";
|
|
4
4
|
import { type Baseline } from "./baseline.js";
|
|
5
5
|
import { type GateVia } from "./llm.js";
|
|
6
|
+
import { type PriorReviewer } from "./review.js";
|
|
6
7
|
import type { GateResult } from "./types.js";
|
|
7
8
|
import { type StructuredFinding } from "../run/journal.js";
|
|
8
9
|
export type LoadProvider = () => number;
|
|
@@ -57,6 +58,7 @@ export interface GateContext {
|
|
|
57
58
|
excludeReviewers?: string[];
|
|
58
59
|
demotedReviewers?: Set<string>;
|
|
59
60
|
reviewHistory?: string[];
|
|
61
|
+
priorReviewers?: PriorReviewer[];
|
|
60
62
|
artifactDir?: string;
|
|
61
63
|
pipeline?: "v185" | "legacy";
|
|
62
64
|
selectTests?: boolean;
|
package/dist/gates/run-gates.js
CHANGED
|
@@ -11,7 +11,7 @@ import { evidenceGate } from "./evidence.js";
|
|
|
11
11
|
import { captureLlmOutput } from "./llm.js";
|
|
12
12
|
import { disallowedBy } from "../route/preference.js";
|
|
13
13
|
import { marginalCostRank } from "../route/router.js";
|
|
14
|
-
import { pickReviewer, reviewGate } from "./review.js";
|
|
14
|
+
import { gateReviewerFloor, pickReviewer, reviewGate } from "./review.js";
|
|
15
15
|
import { scopeGate } from "./scope.js";
|
|
16
16
|
import { shGit } from "../run/git.js";
|
|
17
17
|
import { withJudgeInvocationEvidence } from "../run/journal.js";
|
|
@@ -366,7 +366,7 @@ export async function runGates(task, ctx) {
|
|
|
366
366
|
// ponytail: legacy runs adjacent tools in ONE compareToBaseline call, so there is one interval
|
|
367
367
|
// to measure and each of its gates carries it. Split it only if this branch ever stops batching.
|
|
368
368
|
const finish = startMeasurement();
|
|
369
|
-
const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [...gates]);
|
|
369
|
+
const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [...gates], selected ? { selected } : {});
|
|
370
370
|
const batch = finish();
|
|
371
371
|
for (const g of gates)
|
|
372
372
|
addMeasurement(g, batch);
|
|
@@ -386,7 +386,7 @@ export async function runGates(task, ctx) {
|
|
|
386
386
|
// any later tool before anyone reads its verdict.
|
|
387
387
|
for (const g of gates) {
|
|
388
388
|
await emitStart(g);
|
|
389
|
-
const [r] = await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g]));
|
|
389
|
+
const [r] = await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], g === "test" && selected ? { selected } : {}));
|
|
390
390
|
// the screen's interval IS the test gate's first interval, so the split needs no second clock
|
|
391
391
|
if (g === "test" && selected)
|
|
392
392
|
selectedDurationMs = spans.get("test").durationMs;
|
|
@@ -602,7 +602,11 @@ export async function runGates(task, ctx) {
|
|
|
602
602
|
}
|
|
603
603
|
return rv;
|
|
604
604
|
};
|
|
605
|
-
|
|
605
|
+
// RF-1: THIS task's prior reviewers — earlier rounds' seats plus the seats that produced garbage for
|
|
606
|
+
// it (excludeReviewers names only dispatched seats). Kept apart from the eligibility exclusions the
|
|
607
|
+
// retry below adds for a flaked seat's whole adapter: those sibling channels never reviewed.
|
|
608
|
+
const priorReviewers = [...(ctx.priorReviewers ?? []), ...(ctx.excludeReviewers ?? [])];
|
|
609
|
+
let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, priorReviewers));
|
|
606
610
|
// OBS-193/574: an unparseable review verdict retries the REVIEW exactly once, preferring a
|
|
607
611
|
// different adapter. Only a single-adapter eligible pool may fall back to another channel on the
|
|
608
612
|
// flaked adapter. The flaked verdict never enters results; an exhausted pool preserves its cause.
|
|
@@ -622,10 +626,14 @@ export async function runGates(task, ctx) {
|
|
|
622
626
|
const priorExclusions = ctx.excludeReviewers ?? [];
|
|
623
627
|
const flakedAdapter = flaked.slice(0, flaked.indexOf(":"));
|
|
624
628
|
const adapterExclusions = ctx.channels.filter((c) => c.adapter === flakedAdapter).map(channelKey);
|
|
625
|
-
|
|
629
|
+
// RF-1: the retry filters by the floor reviewGate resolves — author tier, task floor, review.floor
|
|
630
|
+
// and the prior reviewers' tiers, the flaked seat's own included, so a retry never drops a tier.
|
|
631
|
+
const retryPrior = [...priorReviewers, flaked];
|
|
632
|
+
const retryFloor = gateReviewerFloor(task, ctx.cfg, ctx.author, ctx.channels, retryPrior).floor;
|
|
633
|
+
const crossAdapter = pickReviewer(ctx.author, ctx.channels, [...priorExclusions, ...adapterExclusions], ctx.cfg.review.prefer ?? [], retryFloor);
|
|
626
634
|
const exclusion = crossAdapter ? "adapter" : "channel";
|
|
627
635
|
const retryExclusions = [...priorExclusions, ...(crossAdapter ? adapterExclusions : [flaked])];
|
|
628
|
-
const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, retryExclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings));
|
|
636
|
+
const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, retryExclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, retryPrior));
|
|
629
637
|
if (second.meta?.noEligibleReviewer !== true) {
|
|
630
638
|
const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
|
|
631
639
|
const route = exclusion === "adapter"
|
|
@@ -639,10 +647,11 @@ export async function runGates(task, ctx) {
|
|
|
639
647
|
meta: { ...second.meta, reviewRetry: { flaked, retried, exclusion } },
|
|
640
648
|
};
|
|
641
649
|
}
|
|
642
|
-
else
|
|
643
|
-
// Preserve the original no-answer cause when no replacement exists, but name the
|
|
644
|
-
// floor that correctly refused a lower-tier fallback.
|
|
645
|
-
|
|
650
|
+
else {
|
|
651
|
+
// Preserve the original no-answer cause when no replacement exists, but name the resolved
|
|
652
|
+
// floor that correctly refused a lower-tier fallback — in details AND in the row's meta.
|
|
653
|
+
const { reviewerFloor, reviewerFloorCause } = second.meta ?? {};
|
|
654
|
+
rv = { ...rv, details: `${rv.details}\nreview re-route refused: ${second.details}`, meta: { ...rv.meta, reviewerFloor, reviewerFloorCause } };
|
|
646
655
|
}
|
|
647
656
|
}
|
|
648
657
|
return invocations.length ? { ...rv, meta: { ...rv.meta, invocations } } : rv;
|
package/dist/route/router.d.ts
CHANGED
|
@@ -33,3 +33,16 @@ export declare class RoutingError extends Error {
|
|
|
33
33
|
export declare function marginalCostRank(c: BillingChannel): number;
|
|
34
34
|
export declare function route(task: Task, cfg: TickmarkrConfig, channels: BillingChannel[], profile?: RoutingProfile, preferCtx?: RoutingPreferContext, exclude?: ReadonlySet<string>, exploreCtx?: ExploreContext): Route;
|
|
35
35
|
export declare function nextChannel(current: Assignment, task: Task, cfg: TickmarkrConfig, channels: BillingChannel[], tried: string[], profile?: RoutingProfile, exclude?: ReadonlySet<string>): Assignment | null;
|
|
36
|
+
export type ClimbPick = Assignment & {
|
|
37
|
+
climbed: boolean;
|
|
38
|
+
reason?: string;
|
|
39
|
+
};
|
|
40
|
+
export type ClimbResult = ClimbPick;
|
|
41
|
+
/**
|
|
42
|
+
* v2.5.4 OBS-986 (ES-1): climb one tier on request when untried channels exist in higher tiers.
|
|
43
|
+
* With routing.escalateTier on, returns the cheapest untried live channel strictly above current.tier,
|
|
44
|
+
* ordered as nextChannel orders (tier, then marginal cost, learned score strictly last), flagged climbed: true.
|
|
45
|
+
* When the knob is off, when current is frontier, or when no higher tier has an untried channel,
|
|
46
|
+
* returns nextChannel's same-or-higher pick flagged climbed: false naming the reason.
|
|
47
|
+
*/
|
|
48
|
+
export declare function climbChannel(current: Assignment, task: Task, cfg: TickmarkrConfig, channels: BillingChannel[], tried: string[], profile?: RoutingProfile, exclude?: ReadonlySet<string>): ClimbPick | null;
|
package/dist/route/router.js
CHANGED
|
@@ -328,15 +328,8 @@ export function route(task, cfg, channels, profile, preferCtx, exclude, exploreC
|
|
|
328
328
|
maybeSlaLint(lints, task, profile, slaMinutes, eligible[0]);
|
|
329
329
|
return { assignment: toAssignment(eligible[0]), ladder: ladderFor(task, entry), lints, provenance: `${degraded}${bound}, marginal-cost auto (${chosenBy})`, ...(deviation ? { deviation } : {}) };
|
|
330
330
|
}
|
|
331
|
-
|
|
331
|
+
function candidatePool(current, channels, tried, tierPredicate, exclude) {
|
|
332
332
|
channels = withoutExcluded(channels, exclude);
|
|
333
|
-
// already cheapest-sufficient: TIER_RANK asc is the PRIMARY key so escalation climbs one band at a time.
|
|
334
|
-
// Do NOT "unify" this onto route()'s key order (marginal-cost first) — that reverses climb-one-band on mixed fleets (ROUTE-02, D2).
|
|
335
|
-
// ROUTE-13: learnedScore is the STRICTLY-LAST key — within-band tiebreak only. Precomputed
|
|
336
|
-
// outside the comparator (Pitfall 1), never arithmetic-combined with the band keys, no
|
|
337
|
-
// profile-dependent filter. NO exploration bonus here (route():110 has one; a probe on the
|
|
338
|
-
// failure path would spend a real retry). Absent profile ⇒ every score is 0 ⇒ third key
|
|
339
|
-
// all-ties ⇒ the stable sort preserves the exact v1.7 candidate ORDER.
|
|
340
333
|
const triedKeys = new Set(tried);
|
|
341
334
|
const triedIdentities = new Set(tried.map((key) => {
|
|
342
335
|
const channel = channels.find((c) => channelKey(c) === key);
|
|
@@ -348,11 +341,71 @@ export function nextChannel(current, task, cfg, channels, tried, profile, exclud
|
|
|
348
341
|
&& channels.filter((c) => c.adapter === current.adapter).every((c) => triedKeys.has(channelKey(c)));
|
|
349
342
|
const currentChannel = channels.find((c) => c.adapter === current.adapter && c.model === current.model);
|
|
350
343
|
const excludedProvider = currentAdapterExcluded ? routingModelProvider(current.model, currentChannel?.vendor) : undefined;
|
|
351
|
-
|
|
344
|
+
return channels.filter((c) => !triedIdentities.has(modelRouteIdentity(c.model, c.vendor))
|
|
352
345
|
&& (!excludedProvider || routingModelProvider(c.model, c.vendor) !== excludedProvider)
|
|
353
|
-
&&
|
|
346
|
+
&& tierPredicate(c.tier));
|
|
347
|
+
}
|
|
348
|
+
function rankFailoverCandidates(pool, task, cfg, profile) {
|
|
354
349
|
const scores = new Map(pool.map((c) => [channelKey(c), profile ? learnedScore(profile, task.shape, channelKey(c), c.channel, { availWeight: cfg.routing.learnedTuning?.availWeight }) : 0]));
|
|
355
350
|
const scoreOf = (c) => scores.get(channelKey(c));
|
|
356
|
-
|
|
351
|
+
return pool.sort((a, b) => TIER_RANK[a.tier] - TIER_RANK[b.tier] || marginalCostRank(a) - marginalCostRank(b) || scoreOf(b) - scoreOf(a));
|
|
352
|
+
}
|
|
353
|
+
export function nextChannel(current, task, cfg, channels, tried, profile, exclude) {
|
|
354
|
+
// already cheapest-sufficient: TIER_RANK asc is the PRIMARY key so escalation climbs one band at a time.
|
|
355
|
+
// Do NOT "unify" this onto route()'s key order (marginal-cost first) — that reverses climb-one-band on mixed fleets (ROUTE-02, D2).
|
|
356
|
+
// ROUTE-13: learnedScore is the STRICTLY-LAST key — within-band tiebreak only. Precomputed
|
|
357
|
+
// outside the comparator (Pitfall 1), never arithmetic-combined with the band keys, no
|
|
358
|
+
// profile-dependent filter. NO exploration bonus here (route():110 has one; a probe on the
|
|
359
|
+
// failure path would spend a real retry). Absent profile ⇒ every score is 0 ⇒ third key
|
|
360
|
+
// all-ties ⇒ the stable sort preserves the exact v1.7 candidate ORDER.
|
|
361
|
+
const pool = candidatePool(current, channels, tried, (tier) => TIER_RANK[tier] >= TIER_RANK[current.tier], exclude);
|
|
362
|
+
const candidates = rankFailoverCandidates(pool, task, cfg, profile);
|
|
357
363
|
return candidates.length ? toAssignment(candidates[0]) : null;
|
|
358
364
|
}
|
|
365
|
+
function makeClimbPick(assignment, climbed, reason) {
|
|
366
|
+
const pick = {
|
|
367
|
+
adapter: assignment.adapter,
|
|
368
|
+
model: assignment.model,
|
|
369
|
+
channel: assignment.channel,
|
|
370
|
+
tier: assignment.tier,
|
|
371
|
+
climbed,
|
|
372
|
+
...(climbed ? {} : { reason }),
|
|
373
|
+
};
|
|
374
|
+
Object.defineProperty(pick, "assignment", {
|
|
375
|
+
get() {
|
|
376
|
+
return {
|
|
377
|
+
adapter: this.adapter,
|
|
378
|
+
model: this.model,
|
|
379
|
+
channel: this.channel,
|
|
380
|
+
tier: this.tier,
|
|
381
|
+
};
|
|
382
|
+
},
|
|
383
|
+
enumerable: false,
|
|
384
|
+
});
|
|
385
|
+
return pick;
|
|
386
|
+
}
|
|
387
|
+
/**
|
|
388
|
+
* v2.5.4 OBS-986 (ES-1): climb one tier on request when untried channels exist in higher tiers.
|
|
389
|
+
* With routing.escalateTier on, returns the cheapest untried live channel strictly above current.tier,
|
|
390
|
+
* ordered as nextChannel orders (tier, then marginal cost, learned score strictly last), flagged climbed: true.
|
|
391
|
+
* When the knob is off, when current is frontier, or when no higher tier has an untried channel,
|
|
392
|
+
* returns nextChannel's same-or-higher pick flagged climbed: false naming the reason.
|
|
393
|
+
*/
|
|
394
|
+
export function climbChannel(current, task, cfg, channels, tried, profile, exclude) {
|
|
395
|
+
const escalateOn = cfg.routing.escalateTier !== "off";
|
|
396
|
+
if (!escalateOn) {
|
|
397
|
+
const pick = nextChannel(current, task, cfg, channels, tried, profile, exclude);
|
|
398
|
+
return pick ? makeClimbPick(pick, false, "routing.escalateTier knob is off") : null;
|
|
399
|
+
}
|
|
400
|
+
if (current.tier === "frontier") {
|
|
401
|
+
const pick = nextChannel(current, task, cfg, channels, tried, profile, exclude);
|
|
402
|
+
return pick ? makeClimbPick(pick, false, "no higher tier") : null;
|
|
403
|
+
}
|
|
404
|
+
const higherPool = candidatePool(current, channels, tried, (tier) => TIER_RANK[tier] > TIER_RANK[current.tier], exclude);
|
|
405
|
+
if (higherPool.length > 0) {
|
|
406
|
+
const candidates = rankFailoverCandidates(higherPool, task, cfg, profile);
|
|
407
|
+
return makeClimbPick(toAssignment(candidates[0]), true);
|
|
408
|
+
}
|
|
409
|
+
const pick = nextChannel(current, task, cfg, channels, tried, profile, exclude);
|
|
410
|
+
return pick ? makeClimbPick(pick, false, "no higher tier has an untried channel") : null;
|
|
411
|
+
}
|
package/dist/run/daemon.d.ts
CHANGED
|
@@ -109,7 +109,9 @@ export declare const SUITE_WAIT_CEILING_MS = 600000;
|
|
|
109
109
|
export declare const setSuiteWaitCeilingForTests: (ms: number) => void;
|
|
110
110
|
export declare const resetSuiteWaitCeilingForTests: () => void;
|
|
111
111
|
export declare const APPROVAL_POLL_MS = 250;
|
|
112
|
-
export declare const APPROVAL_WINDOW_MS =
|
|
112
|
+
export declare const APPROVAL_WINDOW_MS = 120000;
|
|
113
|
+
export declare const setApprovalWindowForTests: (ms: number) => void;
|
|
114
|
+
export declare const resetApprovalWindowForTests: () => void;
|
|
113
115
|
export declare const EARLY_LAUNCH_LIVENESS_MS = 60000;
|
|
114
116
|
/** Test seam — lowers the empty-pane liveness window without sleeping 60s per case. */
|
|
115
117
|
export declare function setEarlyLaunchLivenessMsForTests(ms: number): void;
|