tickmarkr 2.5.4 → 2.5.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. package/README.md +3 -1
  2. package/dist/adapters/registry.js +6 -1
  3. package/dist/adapters/types.d.ts +3 -0
  4. package/dist/cli/commands/approve.d.ts +1 -0
  5. package/dist/cli/commands/approve.js +118 -9
  6. package/dist/cli/commands/doctor.d.ts +10 -0
  7. package/dist/cli/commands/doctor.js +44 -29
  8. package/dist/cli/commands/fleet.js +199 -33
  9. package/dist/cli/commands/init.js +196 -6
  10. package/dist/cli/commands/plan.js +20 -4
  11. package/dist/cli/commands/resume.js +4 -2
  12. package/dist/cli/commands/run.js +11 -2
  13. package/dist/cli/commands/verify.js +102 -81
  14. package/dist/cli/help.d.ts +2 -0
  15. package/dist/cli/help.js +3 -1
  16. package/dist/config/config.d.ts +25 -8
  17. package/dist/config/config.js +41 -29
  18. package/dist/config/fleet-overlay.d.ts +3 -9
  19. package/dist/config/fleet-overlay.js +114 -19
  20. package/dist/config/fleet-why.d.ts +7 -0
  21. package/dist/config/fleet-why.js +5 -0
  22. package/dist/drivers/index.d.ts +15 -1
  23. package/dist/drivers/index.js +39 -10
  24. package/dist/drivers/orca.d.ts +119 -10
  25. package/dist/drivers/orca.js +781 -110
  26. package/dist/gates/baseline.d.ts +17 -3
  27. package/dist/gates/baseline.js +63 -15
  28. package/dist/gates/cache.d.ts +101 -0
  29. package/dist/gates/cache.js +401 -0
  30. package/dist/gates/llm.d.ts +3 -0
  31. package/dist/gates/llm.js +11 -0
  32. package/dist/gates/review.d.ts +28 -3
  33. package/dist/gates/review.js +106 -14
  34. package/dist/gates/run-gates.d.ts +9 -1
  35. package/dist/gates/run-gates.js +407 -102
  36. package/dist/gates/test-manifest.d.ts +128 -0
  37. package/dist/gates/test-manifest.js +463 -0
  38. package/dist/gates/test-reporter.d.ts +4 -0
  39. package/dist/gates/test-reporter.js +57 -0
  40. package/dist/graph/graph.d.ts +2 -0
  41. package/dist/graph/graph.js +45 -1
  42. package/dist/route/preference.d.ts +22 -1
  43. package/dist/route/preference.js +123 -25
  44. package/dist/route/router.js +31 -6
  45. package/dist/run/consult.js +5 -4
  46. package/dist/run/daemon.d.ts +15 -0
  47. package/dist/run/daemon.js +638 -167
  48. package/dist/run/execution-budget.d.ts +25 -0
  49. package/dist/run/execution-budget.js +142 -0
  50. package/dist/run/git.d.ts +50 -1
  51. package/dist/run/git.js +131 -12
  52. package/dist/run/journal.d.ts +16 -2
  53. package/dist/run/journal.js +115 -25
  54. package/dist/run/lease.d.ts +58 -0
  55. package/dist/run/lease.js +310 -0
  56. package/dist/run/merge.d.ts +2 -0
  57. package/dist/run/merge.js +91 -3
  58. package/dist/run/operator-state.d.ts +11 -0
  59. package/dist/run/operator-state.js +17 -3
  60. package/dist/run/recovery.d.ts +8 -0
  61. package/dist/run/recovery.js +25 -0
  62. package/dist/run/repair-selection.d.ts +12 -0
  63. package/dist/run/repair-selection.js +56 -0
  64. package/dist/run/stall.d.ts +6 -1
  65. package/dist/run/stall.js +60 -3
  66. package/dist/tui/cockpit/board.d.ts +96 -0
  67. package/dist/tui/cockpit/board.js +346 -0
  68. package/dist/tui/cockpit/decision-actions.js +2 -0
  69. package/dist/tui/cockpit/layout.d.ts +5 -1
  70. package/dist/tui/cockpit/layout.js +8 -3
  71. package/dist/tui/cockpit/live-runtime.js +71 -21
  72. package/dist/tui/cockpit/run-view.d.ts +7 -5
  73. package/dist/tui/cockpit/run-view.js +12 -11
  74. package/dist/tui/ink/fleet-app.d.ts +49 -27
  75. package/dist/tui/ink/fleet-app.js +229 -38
  76. package/package.json +2 -2
  77. package/skills/tickmarkr-overseer/SKILL.md +173 -113
  78. package/skills/tickmarkr-overseer/scripts/grade-ci.sh +29 -2
@@ -1,5 +1,5 @@
1
1
  import { existsSync, writeFileSync } from "node:fs";
2
- import { join } from "node:path";
2
+ import { dirname, join } from "node:path";
3
3
  import { channelKey, shq } from "../adapters/types.js";
4
4
  import { criticalPathHits, DEFAULT_DIFF_CAP, DEFAULT_REVIEW_CRITICAL_PATHS, declaredReviewPolicy, isReviewLeafPath, raiseReviewPolicy, REVIEW_VERSION_MIRRORS, TIER_RANK, } from "../config/config.js";
5
5
  import { filesGlob } from "../graph/files-glob.js";
@@ -10,6 +10,7 @@ import { structuredFindings } from "../run/journal.js";
10
10
  import { redactSecrets } from "../run/redact.js";
11
11
  import { marginalCostRank } from "../route/router.js";
12
12
  import { modelProvider } from "../route/preference.js";
13
+ import { resolveStateDir } from "./cache.js";
13
14
  import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlmDetailed, verdictNonceLine } from "./llm.js";
14
15
  import { classifyVerdictCause } from "./verdict-cause.js";
15
16
  import { captureDiffCapFor, measureArtifactDiff, reviewableLogicDiff, } from "./artifact-manifest.js";
@@ -130,8 +131,14 @@ export function checkDiffCap(gate, measured, cap, prefix = "") {
130
131
  gate,
131
132
  pass: false,
132
133
  details: prefix + `diff exceeds verifiable cap (${measured} > ${cap}) — ${DIFF_CAP_REMEDY}`,
133
- // daemon/run-gates: park('human') immediately — the diff cannot shrink by retrying (OBS-48).
134
- meta: { park: "human" },
134
+ meta: {
135
+ park: "diff-cap",
136
+ parkKind: "diff-cap",
137
+ measuredBytes: measured,
138
+ permittedBytes: cap,
139
+ measured,
140
+ permitted: cap,
141
+ },
135
142
  };
136
143
  }
137
144
  /** Apply the strict reviewable-logic cap and the finite, larger capture cap independently. */
@@ -147,12 +154,19 @@ export function checkTaskDiffCaps(gate, measured, logicCap, prefix = "") {
147
154
  pass: false,
148
155
  details: prefix
149
156
  + `captured artifact diff exceeds verifiable capture cap (${measured.captureBytes} > ${captureCap}) — ${DIFF_CAP_REMEDY}`,
150
- meta: { park: "human" },
157
+ meta: {
158
+ park: "diff-cap",
159
+ parkKind: "diff-cap",
160
+ measuredBytes: measured.captureBytes,
161
+ permittedBytes: captureCap,
162
+ measured: measured.captureBytes,
163
+ permitted: captureCap,
164
+ },
151
165
  };
152
166
  }
153
167
  export function isDiffCapPark(result) {
154
168
  return result.pass === false
155
- && result.meta?.park === "human"
169
+ && result.meta?.parkKind === "diff-cap"
156
170
  && /diff exceeds verifiable (?:capture )?cap/i.test(result.details);
157
171
  }
158
172
  // ponytail: single policy hook for callers after runGates — skips the escalation ladder on diff-cap trips.
@@ -193,6 +207,21 @@ export function isReviewClosureInvalid(v, priorIds) {
193
207
  || new Set(allCandidateIds.map((id) => matchClosureId(id, priors) ?? id)).size !== allCandidateIds.length
194
208
  || [...priors].some((id) => !allCandidateIds.some((candidate) => matchClosureId(candidate, id))));
195
209
  }
210
+ /**
211
+ * OBS-1013 add.3: the reviewer ECHOED closure ids and at least one matches no carried fingerprint —
212
+ * it answered about the materials and missed the id (a retyped, truncated or paraphrased fingerprint).
213
+ * That is a no-verdict about the carried work (re-route), not a parse defect. A verdict that omits a
214
+ * list, carries a non-string or duplicates an id stays malformed: its shape, not its ids, is wrong.
215
+ */
216
+ export function isReviewClosureMismatch(v, priorIds) {
217
+ if (!v || !Array.isArray(v.resolved) || !Array.isArray(v.reraised))
218
+ return false;
219
+ const ids = [...v.resolved, ...v.reraised];
220
+ if (!ids.every((id) => typeof id === "string"))
221
+ return false;
222
+ const priors = priorIds instanceof Set ? priorIds : new Set(priorIds);
223
+ return ids.some((id) => matchClosureId(id, priors) === undefined);
224
+ }
196
225
  // v1.53 T2: same entry grammar as routing.map.prefer (router.ts preferIndex — router is out of this
197
226
  // module's dependency direction for a private fn, so the 3 lines live here too): `adapter` matches
198
227
  // every channel of that adapter, `adapter:model` exactly one; unmatched channels sort after all entries.
@@ -247,7 +276,9 @@ export function pickReviewer(author, channels, exclude = [], // v1.1 failover: r
247
276
  prefer = [], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
248
277
  floor, // task/config/prior floor from the caller; the author's own tier is ALWAYS applied here (RF-1)
249
278
  history = [], // run-scoped picks, oldest to newest; empty preserves the established ranking
250
- onSeat, demoted = new Set()) {
279
+ onSeat, demoted = new Set(),
280
+ // OBS-1033: vendors that authored a carried commit inside the accumulated diff — excluded for the round.
281
+ excludeVendors = new Set()) {
251
282
  // FLEET-05 success criterion 2: an author not resolvable in the channel list yields NO reviewer.
252
283
  // The old `?? author.adapter` fallback compared an adapter id to vendor names, matched nothing, and
253
284
  // admitted every reviewer — including the author's own channel (fail-OPEN). null lands on reviewGate's
@@ -267,6 +298,7 @@ onSeat, demoted = new Set()) {
267
298
  && modelProvider(c.model, c.vendor) !== authorProvider
268
299
  && modelId(c.model) !== modelId(author.model)
269
300
  && !exclude.includes(channelKey(c))
301
+ && !excludeVendors.has(c.vendor)
270
302
  && TIER_RANK[c.tier] >= TIER_RANK[effectiveFloor])
271
303
  .sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b));
272
304
  const reviewer = [...ranked].sort((a, b) => Number(demoted.has(channelKey(a))) - Number(demoted.has(channelKey(b)))
@@ -289,13 +321,66 @@ export function renderDeclaredWriteScope(files) {
289
321
  The task DECLARED these write-scope patterns:
290
322
  ${files.map((path) => `- ${path}`).join("\n")}`;
291
323
  }
324
+ /**
325
+ * OBS-1033: the vendors of the seats that authored commits inside the accumulated diff. A seat is
326
+ * never handed its own earlier work to approve. A prior author not resolvable in the pool excludes
327
+ * its adapter's vendors instead (fail closed: the seat is known, its vendor is whatever it bills as).
328
+ */
329
+ export function carriedAuthorVendors(channels, carriedAuthors = []) {
330
+ const vendors = new Set();
331
+ for (const key of carriedAuthors) {
332
+ const adapter = key.split(":")[0];
333
+ const exact = channels.filter((c) => channelKey(c) === key);
334
+ for (const c of exact.length ? exact : channels.filter((c) => c.adapter === adapter))
335
+ vendors.add(c.vendor);
336
+ }
337
+ return vendors;
338
+ }
339
+ /**
340
+ * OBS-1020: the compiled goal is the contract. After `resume --graph-changed` the worktree's copy of
341
+ * the spec is the pre-change text on the integration branch, so a reviewer that reads it grades a
342
+ * superseded contract. The daemon's repository root is where specs and planning records are current.
343
+ */
344
+ export function renderGoalSection(goal, repoRoot) {
345
+ return `## Goal (authoritative — compiled from the sealed graph; the worktree's spec file may be stale after resume --graph-changed)
346
+ ${goal}
347
+ ${repoRoot ? `Specs and planning records are read in the daemon's repository root ${repoRoot} (its specs/ and .planning/), never this worktree's copies.` : ""}`;
348
+ }
349
+ /**
350
+ * The daemon's repository root: the parent of the state dir the run's artifacts live under. Named
351
+ * only when that state dir exists — a guessed one would send the reviewer to a path that holds nothing.
352
+ */
353
+ function daemonRepoRoot(worktree, artifactDir) {
354
+ try {
355
+ const stateDir = resolveStateDir(worktree, artifactDir);
356
+ return stateDir.endsWith("/.tickmarkr") && existsSync(stateDir) ? dirname(stateDir) : undefined;
357
+ }
358
+ catch {
359
+ return undefined;
360
+ }
361
+ }
362
+ /**
363
+ * OBS-1013 add.3: each carried id is printed ONCE, verbatim, inside a fenced block the reviewer can
364
+ * copy; the notes follow in the same order. A reviewer that retyped a 600-byte id from prose lost
365
+ * closure on a typo and that read as malformed — the block is what a closure list is copied from.
366
+ */
367
+ export function renderPriorMaterials(priorMaterials) {
368
+ return `## Prior materials this attempt must close
369
+ Copy each fingerprint below EXACTLY (they appear once, in this block) into resolved or reraised:
370
+ \`\`\`text
371
+ ${priorMaterials.map((finding) => `Fingerprint: ${finding.fingerprint}`).join("\n")}
372
+ \`\`\`
373
+ ${priorMaterials.map((finding, i) => `${i + 1}. ${finding.note}`).join("\n\n")}`;
374
+ }
292
375
  export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
293
376
  // OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
294
377
  // direct tests) skips persistence and changes nothing else.
295
378
  artifactDir, reviewHistory, demotedReviewers, carriedFindings = [],
296
379
  // RF-1: channel keys of THIS task's prior reviewers (earlier rounds, a flaked seat) — task-scoped,
297
380
  // never the run-wide rotation history nor excludeReviewers; the seat holds the highest of their tiers.
298
- priorReviewers = []) {
381
+ priorReviewers = [],
382
+ // OBS-1033: channel keys of the seats that authored the carried commits (the daemon's tried list).
383
+ carriedAuthors = []) {
299
384
  // R3 (OBS-186): participation is keyed on PATHS. The compiler's assignment comes from the DECLARED
300
385
  // files[]; the operator's floor may RAISE it to full and can never lower it. `complexityThreshold` is
301
386
  // retired — the branch that returned a green skip on a complexity comparison is gone, and with it the
@@ -370,7 +455,7 @@ priorReviewers = []) {
370
455
  const { floor: reviewerFloor, cause: reviewerFloorCause } = gateReviewerFloor(task, cfg, author, channels, priorReviewers);
371
456
  const floorMeta = { reviewerFloor, reviewerFloorCause };
372
457
  let rotationSeat;
373
- const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor, reviewHistory, reviewHistory ? (seat) => { rotationSeat = seat; } : undefined, demotedReviewers);
458
+ const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor, reviewHistory, reviewHistory ? (seat) => { rotationSeat = seat; } : undefined, demotedReviewers, carriedAuthorVendors(channels, carriedAuthors));
374
459
  if (!reviewer) {
375
460
  // meta.noEligibleReviewer lets run-gates' review-retry keep the ORIGINAL unparseable result when
376
461
  // the retry finds no second seat — a truthful cause beats a synthetic no-reviewer failure.
@@ -390,6 +475,7 @@ priorReviewers = []) {
390
475
  if (capFail)
391
476
  return capFail;
392
477
  const nonce = generateVerdictNonce();
478
+ const repoRoot = daemonRepoRoot(worktree, artifactDir);
393
479
  const prompt = `TICKMARKR-REVIEW
394
480
  You are a skeptical cross-vendor code reviewer. Another agent (vendor: ${author.adapter}) authored this diff.
395
481
  Look for correctness bugs, security issues, and acceptance-criteria gaps. Approve only if you would merge it.
@@ -397,13 +483,13 @@ Look for correctness bugs, security issues, and acceptance-criteria gaps. Approv
397
483
  ${COMPLETION_FAKING_CHECKLIST}
398
484
 
399
485
  ## Task ${task.id}: ${task.title} (complexity ${task.complexity})
486
+ ${renderGoalSection(task.goal, repoRoot)}
400
487
  ## Acceptance criteria
401
488
  ${task.acceptance.map((a) => `- ${renderAcceptanceItem(a)}`).join("\n")}
402
489
 
403
490
  ${renderDeclaredWriteScope(task.files)}
404
491
 
405
- ${priorMaterials.length ? `## Prior materials this attempt must close
406
- ${priorMaterials.map((finding) => `Fingerprint: ${finding.fingerprint}\n${finding.note}`).join("\n\n")}
492
+ ${priorMaterials.length ? `${renderPriorMaterials(priorMaterials)}
407
493
 
408
494
  ` : ""}## Diff
409
495
  \`\`\`diff
@@ -483,17 +569,22 @@ The top-level comments array is optional. Use it only for actionable line-anchor
483
569
  const findings = v && Array.isArray(v.findings) ? v.findings : null;
484
570
  const priorIds = new Set(priorMaterials.map((finding) => finding.fingerprint));
485
571
  const closureInvalid = isReviewClosureInvalid(v, priorIds);
572
+ const closureMismatch = closureInvalid && isReviewClosureMismatch(v, priorIds);
486
573
  // findings decides the verdict on its own; the legacy path still needs approve + issues to parse.
487
574
  if (!v || closureInvalid || (findings === null && (typeof v.approve !== "boolean" || !Array.isArray(v.issues)))) {
488
575
  // OBS-196: name the cause and persist the raw bytes — a ruled-on "unparseable" without its
489
576
  // evidence cannot be audited, and a cutoff must never be indistinguishable from a parse defect.
490
577
  const bytes = llm.seatAuthoredBytes ?? Buffer.byteLength(raw.trim(), "utf8");
491
- const cause = closureInvalid ? "malformed-verdict" : llm.launchNeverStarted ? "launch-never-started"
492
- : llm.timedOut ? (bytes > 0 ? "truncated" : "silent")
493
- : classifyVerdictCause(raw, nonce, "approve", llm);
578
+ const cause = closureMismatch ? "closure-mismatch" : closureInvalid ? "malformed-verdict"
579
+ : llm.launchNeverStarted ? "launch-never-started"
580
+ : llm.silentAtBeat ? "silent"
581
+ : llm.timedOut ? (bytes > 0 ? "truncated" : "silent")
582
+ : classifyVerdictCause(raw, nonce, "approve", llm);
494
583
  const failure = cause === "malformed-verdict"
495
584
  ? "review output unparseable"
496
- : "review dispatch failed — no structurally valid nonce-bound response";
585
+ : cause === "closure-mismatch"
586
+ ? "review verdict closes no carried fingerprint — closure ids match none of the carried materials"
587
+ : "review dispatch failed — no structurally valid nonce-bound response";
497
588
  return {
498
589
  gate: "review",
499
590
  pass: false,
@@ -508,6 +599,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
508
599
  provider,
509
600
  ...(cause === "malformed-verdict" ? { unparseable: true } : { noVerdict: true, classification: "infra", infra: true }),
510
601
  cause,
602
+ ...(closureMismatch ? { resolved: v?.resolved, reraised: v?.reraised, carriedFingerprints: [...priorIds] } : {}),
511
603
  bytes, seatAuthoredBytes: bytes,
512
604
  ...(saved ? { rawPath: saved } : {}),
513
605
  ...(savedBrief ? { briefPath: savedBrief } : {}),
@@ -5,7 +5,9 @@ import { type Baseline } from "./baseline.js";
5
5
  import { type GateVia } from "./llm.js";
6
6
  import { type PriorReviewer } from "./review.js";
7
7
  import type { GateResult } from "./types.js";
8
+ import { type VerificationRetryCause } from "../run/recovery.js";
8
9
  import { type StructuredFinding } from "../run/journal.js";
10
+ import { type VerificationScope } from "./cache.js";
9
11
  export type LoadProvider = () => number;
10
12
  /** Test seam — inject deterministic load samples; production always reads os.loadavg. */
11
13
  export declare function setLoadProviderForTests(provider: LoadProvider): void;
@@ -43,6 +45,8 @@ export type GateEvent = {
43
45
  result: GateResult;
44
46
  };
45
47
  export interface GateContext {
48
+ authorizeInfraRetry?: (subject: string, cause?: VerificationRetryCause) => boolean;
49
+ verificationScope?: VerificationScope;
46
50
  worktree: string;
47
51
  baseRef: string;
48
52
  result: WorkerResult;
@@ -57,13 +61,17 @@ export interface GateContext {
57
61
  carriedFindings?: readonly StructuredFinding[];
58
62
  excludeReviewers?: string[];
59
63
  demotedReviewers?: Set<string>;
64
+ reviewNoVerdicts?: Map<string, string[]>;
65
+ carriedAuthors?: readonly string[];
60
66
  reviewHistory?: string[];
61
67
  priorReviewers?: PriorReviewer[];
62
68
  artifactDir?: string;
63
- pipeline?: "v185" | "legacy";
64
69
  selectTests?: boolean;
70
+ requiredRepairTests?: readonly string[];
71
+ selectionReason?: string;
65
72
  collateral?: ReadonlyArray<string>;
66
73
  onGate?: (e: GateEvent) => void | Promise<void>;
74
+ stateDir?: string;
67
75
  }
68
76
  /**
69
77
  * The configured test command narrowed to these files. Mirrors testFiltered's `--` rule (acceptance.ts:104):