tickmarkr 1.87.0 → 1.90.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (93) hide show
  1. package/dist/adapters/catalog.d.ts +18 -1
  2. package/dist/adapters/catalog.js +44 -1
  3. package/dist/adapters/fake.d.ts +2 -1
  4. package/dist/adapters/fake.js +7 -0
  5. package/dist/adapters/grok.js +11 -0
  6. package/dist/adapters/kimi.d.ts +2 -1
  7. package/dist/adapters/kimi.js +36 -0
  8. package/dist/adapters/opencode.js +17 -0
  9. package/dist/adapters/pi.js +11 -0
  10. package/dist/adapters/prompt.js +8 -1
  11. package/dist/adapters/registry.js +76 -57
  12. package/dist/adapters/types.d.ts +34 -3
  13. package/dist/adapters/types.js +99 -1
  14. package/dist/cli/commands/approve.d.ts +2 -0
  15. package/dist/cli/commands/approve.js +104 -84
  16. package/dist/cli/commands/compile.d.ts +1 -1
  17. package/dist/cli/commands/compile.js +29 -12
  18. package/dist/cli/commands/doctor.d.ts +16 -0
  19. package/dist/cli/commands/doctor.js +52 -0
  20. package/dist/cli/commands/init.js +2 -1
  21. package/dist/cli/commands/plan.d.ts +1 -1
  22. package/dist/cli/commands/plan.js +10 -1
  23. package/dist/cli/commands/report.js +49 -0
  24. package/dist/cli/commands/status.js +298 -96
  25. package/dist/cli/commands/verify.d.ts +9 -0
  26. package/dist/cli/commands/verify.js +177 -0
  27. package/dist/cli/harness.d.ts +13 -0
  28. package/dist/cli/harness.js +50 -0
  29. package/dist/cli/index.d.ts +1 -1
  30. package/dist/cli/index.js +3 -1
  31. package/dist/compile/collateral.js +11 -11
  32. package/dist/compile/common.js +2 -2
  33. package/dist/compile/index.d.ts +14 -3
  34. package/dist/compile/index.js +36 -10
  35. package/dist/compile/native.d.ts +15 -1
  36. package/dist/compile/native.js +310 -28
  37. package/dist/config/config.js +2 -2
  38. package/dist/drivers/subprocess.d.ts +6 -1
  39. package/dist/drivers/subprocess.js +9 -4
  40. package/dist/gates/acceptance.d.ts +21 -1
  41. package/dist/gates/acceptance.js +67 -22
  42. package/dist/gates/artifact-manifest.d.ts +119 -0
  43. package/dist/gates/artifact-manifest.js +357 -0
  44. package/dist/gates/baseline.d.ts +45 -5
  45. package/dist/gates/baseline.js +119 -15
  46. package/dist/gates/llm.js +37 -26
  47. package/dist/gates/review.d.ts +16 -11
  48. package/dist/gates/review.js +44 -150
  49. package/dist/gates/run-gates.js +124 -7
  50. package/dist/gates/scope.js +3 -3
  51. package/dist/graph/files-glob.d.ts +18 -0
  52. package/dist/graph/files-glob.js +22 -0
  53. package/dist/graph/schema.d.ts +3 -1
  54. package/dist/graph/schema.js +4 -1
  55. package/dist/run/daemon.d.ts +44 -0
  56. package/dist/run/daemon.js +2334 -1973
  57. package/dist/run/git.d.ts +53 -0
  58. package/dist/run/git.js +119 -5
  59. package/dist/run/interactive-seed.d.ts +6 -2
  60. package/dist/run/interactive-seed.js +72 -5
  61. package/dist/run/journal.d.ts +9 -1
  62. package/dist/run/journal.js +99 -9
  63. package/dist/run/lock.d.ts +11 -0
  64. package/dist/run/lock.js +97 -6
  65. package/dist/run/merge.d.ts +4 -1
  66. package/dist/run/merge.js +26 -7
  67. package/dist/run/outcome.d.ts +50 -0
  68. package/dist/run/outcome.js +152 -0
  69. package/dist/run/protocol.d.ts +460 -0
  70. package/dist/run/protocol.js +433 -0
  71. package/dist/run/supervision.d.ts +29 -0
  72. package/dist/run/supervision.js +189 -0
  73. package/fixtures/authoring-lints/01-awk-range-self-pass.spec.md +12 -0
  74. package/fixtures/authoring-lints/02-judge-text-key-miss.spec.md +7 -0
  75. package/fixtures/authoring-lints/03-c1-t41-rendered-observable.spec.md +8 -0
  76. package/fixtures/authoring-lints/04-c1-t24-named-file.spec.md +8 -0
  77. package/fixtures/authoring-lints/05-c2-t24-t28-dep-inversion.spec.md +7 -0
  78. package/fixtures/authoring-lints/06-c2-denumbered-coupling.spec.md +7 -0
  79. package/fixtures/authoring-lints/07-c3a-t41-line-count-proxy.spec.md +7 -0
  80. package/fixtures/authoring-lints/08-c3b-t41-governance-referent.spec.md +7 -0
  81. package/fixtures/authoring-lints/09-c4-universals-without-pointer.spec.md +7 -0
  82. package/fixtures/authoring-lints/10-c5-t34-conjunct-flood.spec.md +7 -0
  83. package/fixtures/authoring-lints/11-c6-t34-q3-q9-q20-bundle.spec.md +7 -0
  84. package/fixtures/authoring-lints/12-c7-t24-prose-seam.spec.md +8 -0
  85. package/fixtures/wrapped-acceptance.native.md +29 -0
  86. package/package.json +1 -1
  87. package/schema/rungraph.schema.json +21 -2
  88. package/skills/tickmarkr-overseer/SKILL.md +262 -5
  89. package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +79 -8
  90. package/skills/tickmarkr-overseer/scripts/watch-contamination.sh +80 -0
  91. package/skills/tickmarkr-overseer/scripts/watch-context.sh +86 -0
  92. package/skills/tickmarkr-overseer/scripts/watch-parks.sh +96 -0
  93. package/skills/tickmarkr-overseer/scripts/watch-pending-input.sh +201 -0
@@ -1,11 +1,20 @@
1
1
  import { z } from "zod";
2
- import type { WorkerAdapter } from "./types.js";
2
+ import { type WorkerAdapter } from "./types.js";
3
3
  export declare const CLI_NAME_RE: RegExp;
4
4
  export declare const MODEL_ID_FIELDS: readonly ["id", "selector", "model"];
5
5
  export type ModelIdField = (typeof MODEL_ID_FIELDS)[number];
6
+ export declare const DRIVE_TRUST_DIALOG_MESSAGE = "drive contract must declare trustDialog \u2014 either the workspace-trust prompt this CLI renders, captured verbatim as {fingerprint, key}, or {kind: none, reason}";
6
7
  export declare const CliDriveSchema: z.ZodObject<{
7
8
  headless: z.ZodString;
8
9
  interactive: z.ZodNullable<z.ZodString>;
10
+ trustDialog: z.ZodUnion<readonly [z.ZodObject<{
11
+ kind: z.ZodOptional<z.ZodLiteral<"dialog">>;
12
+ fingerprint: z.ZodString;
13
+ key: z.ZodString;
14
+ }, z.core.$strict>, z.ZodObject<{
15
+ kind: z.ZodLiteral<"none">;
16
+ reason: z.ZodString;
17
+ }, z.core.$strict>]>;
9
18
  listModels: z.ZodOptional<z.ZodObject<{
10
19
  argv: z.ZodArray<z.ZodString>;
11
20
  parser: z.ZodEnum<{
@@ -35,6 +44,14 @@ export declare const CliEntrySchema: z.ZodObject<{
35
44
  drive: z.ZodOptional<z.ZodObject<{
36
45
  headless: z.ZodString;
37
46
  interactive: z.ZodNullable<z.ZodString>;
47
+ trustDialog: z.ZodUnion<readonly [z.ZodObject<{
48
+ kind: z.ZodOptional<z.ZodLiteral<"dialog">>;
49
+ fingerprint: z.ZodString;
50
+ key: z.ZodString;
51
+ }, z.core.$strict>, z.ZodObject<{
52
+ kind: z.ZodLiteral<"none">;
53
+ reason: z.ZodString;
54
+ }, z.core.$strict>]>;
38
55
  listModels: z.ZodOptional<z.ZodObject<{
39
56
  argv: z.ZodArray<z.ZodString>;
40
57
  parser: z.ZodEnum<{
@@ -10,6 +10,7 @@ import { grok } from "./grok.js";
10
10
  import { kimi } from "./kimi.js";
11
11
  import { opencode } from "./opencode.js";
12
12
  import { pi } from "./pi.js";
13
+ import { TRUST_DIALOG_VARIANTS, TrustDialogSchema } from "./types.js";
13
14
  export const CLI_NAME_RE = /^[a-z0-9-]+$/;
14
15
  // `field` names the ONE key projected out of an array of objects — a model id and nothing else.
15
16
  // `{"models":[{…}]}` is the normal shape of a modern CLI's --json output, and without a selector the
@@ -29,9 +30,19 @@ const ListModelsSchema = z.object({
29
30
  ctx.addIssue({ code: "custom", path: ["field"], message: "field is only read by the json parser" });
30
31
  }
31
32
  });
33
+ // v1.89 T1 / OBS-414: a drive contract declares its trust posture or it is not a drive contract.
34
+ // `declarativeAdapter()` builds every catalog-driven CLI, so a contract without this field would
35
+ // reintroduce the silent opt-out at the one construction site no adapter file covers. Absence gets
36
+ // its own operator sentence; a malformed declaration keeps the union's own diagnosis.
37
+ // Quote-free by construction: a zod issue reaches the operator as a JSON dump, where an embedded
38
+ // double quote is escaped and no longer reads back as the sentence that was written.
39
+ export const DRIVE_TRUST_DIALOG_MESSAGE = "drive contract must declare trustDialog — either the workspace-trust prompt this CLI renders, captured verbatim as {fingerprint, key}, or {kind: none, reason}";
32
40
  export const CliDriveSchema = z.object({
33
41
  headless: z.string().min(1),
34
42
  interactive: z.string().min(1).nullable(),
43
+ trustDialog: z.union(TRUST_DIALOG_VARIANTS, {
44
+ error: (issue) => (issue.input === undefined ? DRIVE_TRUST_DIALOG_MESSAGE : undefined),
45
+ }),
35
46
  listModels: ListModelsSchema.optional(),
36
47
  }).strict().superRefine((drive, ctx) => {
37
48
  for (const [field, template] of [["headless", drive.headless], ["interactive", drive.interactive]]) {
@@ -82,6 +93,14 @@ function validateCliEntry(entry) {
82
93
  if (drive.adapter.id !== base.id || drive.adapter.vendor !== base.vendor) {
83
94
  throw new Error(`native drive identity mismatch for ${base.id}`);
84
95
  }
96
+ // v1.89 T1 / OBS-414: native adapters carry their declaration in code, so the compiler enforces
97
+ // PRESENCE — it cannot enforce that the bytes are a real capture. The same schema the drive
98
+ // contract uses runs here, on the registry's own construction path, so a blank fingerprint or a
99
+ // prompt that is not a workspace-trust gate is unbuildable from either side of the catalog.
100
+ const trust = TrustDialogSchema.safeParse(drive.adapter.trustDialog);
101
+ if (!trust.success) {
102
+ throw new Error(`invalid trust declaration for ${base.id}: ${trust.error.issues.map((i) => i.message).join("; ")}`);
103
+ }
85
104
  return { ...base, drive };
86
105
  }
87
106
  // Package-owned, deterministic order. This is the sole shipped definition array: candidate-name
@@ -103,7 +122,31 @@ export const CLI_CATALOG = [
103
122
  native(pi, "pi"),
104
123
  native(grok, "grok"),
105
124
  native(kimi, "kimi"),
106
- "gemini", "qwen", "aider", "goose", "amp", "droid", "auggie", "crush", "omp",
125
+ "gemini", "qwen", "aider", "goose", "amp", "droid", "auggie", "crush",
126
+ {
127
+ id: "omp",
128
+ binary: "omp",
129
+ // PROBE-omp-v189.md, re-recorded by hand 2026-08-07: `omp --version` → `omp/17.2.10`.
130
+ // A prefix match, so a patch bump does not re-open the identity gate.
131
+ identity: "^omp/",
132
+ // omp is a multi-provider gateway; a selected model carries its real vendor in its own prefix.
133
+ vendor: "mixed",
134
+ drive: {
135
+ // The same probe recorded -p as the fail-closed print mode (exit 1 on an unknown model).
136
+ // interactive deliberately omits it: `interactive` is nullable and `headless` is required, so
137
+ // the invisible configuration is the easier one to write — this one is written visible.
138
+ headless: "omp -p --model {model} @{promptFile}",
139
+ interactive: "omp --model {model} @{promptFile}",
140
+ trustDialog: {
141
+ kind: "none",
142
+ reason: "omp 17.2.10 showed no workspace-trust prompt during the recorded interactive probe (PROBE-omp-v189.md, 2026-08-07)",
143
+ },
144
+ // The recorded payload is an OBJECT keyed by models: 370 entries, selector prefixed 370/370
145
+ // and the neighbouring bare `id` prefixed 0/370. Projecting `id` would return 370 plausible
146
+ // ids that route to nothing, so the selector is the only id key this contract may name.
147
+ listModels: { argv: ["models", "ls", "--json"], parser: "json", path: "models", field: "selector" },
148
+ },
149
+ },
107
150
  ].flatMap((entry) => typeof entry === "string"
108
151
  ? [{ id: entry, binary: entry, identity: ".+", vendor: null }]
109
152
  : [entry]);
@@ -1,6 +1,6 @@
1
1
  import type { TickmarkrConfig } from "../config/config.js";
2
2
  import type { Task } from "../graph/schema.js";
3
- import { type Assignment, type AuthHealth, type BillingChannel, type ContextUsage, type Invocation, type SessionRef, type TokenUsage, type WorkerAdapter, type WorkerResult } from "./types.js";
3
+ import { type Assignment, type AuthHealth, type BillingChannel, type ContextUsage, type Invocation, type SessionRef, type TokenUsage, type TrustDialog, type WorkerAdapter, type WorkerResult } from "./types.js";
4
4
  export interface FakeScript {
5
5
  tasks: Record<string, Array<{
6
6
  shell: string;
@@ -25,6 +25,7 @@ export declare class FakeAdapter implements WorkerAdapter {
25
25
  private scriptPath;
26
26
  id: string;
27
27
  vendor: string;
28
+ trustDialog: TrustDialog;
28
29
  private script;
29
30
  private attempts;
30
31
  constructor(scriptPath: string);
@@ -7,6 +7,13 @@ export class FakeAdapter {
7
7
  scriptPath;
8
8
  id = "fake";
9
9
  vendor = "fake-a";
10
+ // v1.89 T1 / OBS-414: the scripted adapter runs `bash -c` — it has no TUI, so no prompt of any
11
+ // kind can render. Daemon tests that exercise the auto-answer seam assign a captured declaration
12
+ // to this field explicitly; that assignment is the falsifier for the claim made here.
13
+ trustDialog = {
14
+ kind: "none",
15
+ reason: "the zero-token scripted adapter dispatches a bash command and renders no TUI, so no trust prompt exists to capture",
16
+ };
10
17
  script;
11
18
  attempts = new Map();
12
19
  constructor(scriptPath) {
@@ -114,6 +114,17 @@ export const grok = {
114
114
  return { command: this.headlessCommand(ctx.promptFile, a.model) };
115
115
  },
116
116
  parse: parseWorkerResult, // prompt.ts — last-valid-trailer, hard-wrap tolerant. No grok fork.
117
+ // v1.89 T1 / OBS-414: no capture exists, so none is invented. grok 0.2.93 has never been observed
118
+ // rendering a workspace/folder-trust prompt: it loads the operator's global Claude-Code-compat
119
+ // config rather than keeping a per-directory trust store, and no grok dispatch in any recorded run
120
+ // has stalled before launch (OBS-414's own sweep lists grok's gap as MCP suppression, not trust).
121
+ // FALSIFIER, and it is cheap: dispatch grok into a never-seen worktree and read the pane. One
122
+ // capture of a trust prompt replaces this with {fingerprint, key} — the capture is the only
123
+ // admissible evidence, and a guess here would press a key on whatever it happened to match.
124
+ trustDialog: {
125
+ kind: "none",
126
+ reason: "grok 0.2.93 keeps no per-directory trust store and no recorded pane shows it rendering a workspace-trust prompt; falsified by one capture from a fresh worktree",
127
+ },
117
128
  // v1.5 MODEL-01: fail OPEN to [] (advisory detection, unlike gates). Parses the model LIST only —
118
129
  // the auth banner on line 1 is ignored BY CONSTRUCTION (parseGrokModels anchors on the header).
119
130
  // Advisory/doctor-only — never in the dispatch or auth path (types.ts:56-60).
@@ -1,5 +1,5 @@
1
1
  import type { ExecutorDriver, Slot } from "../drivers/types.js";
2
- import { type Assignment, type WorkerAdapter, type WorkerResult } from "./types.js";
2
+ import { type Assignment, type TrustDialog, type WorkerAdapter, type WorkerResult } from "./types.js";
3
3
  export declare function kimiAuthed(credentialsText: string, nowMs: number): boolean;
4
4
  export declare function parseKimiModels(raw: string): string[];
5
5
  export declare function parseKimiResult(raw: string, nonce: string): WorkerResult;
@@ -32,6 +32,7 @@ export interface KimiInteractiveSeedResult {
32
32
  sessionId?: string;
33
33
  }
34
34
  export declare const KIMI_INPUT_BOX: import("./types.js").InputBox;
35
+ export declare const KIMI_TRUST_DIALOG: TrustDialog;
35
36
  export declare function runKimiInteractiveSeed(opts: {
36
37
  driver: Pick<ExecutorDriver, "run" | "waitOutput" | "read">;
37
38
  slot: Slot;
@@ -234,6 +234,34 @@ const KIMI_SEED = {
234
234
  : confirmation;
235
235
  },
236
236
  };
237
+ // v1.89 T1 / OBS-414: the dialog that cost run …215 thirty minutes. Kimi records trust PER FOLDER
238
+ // (~/.kimi-code/workspaces.json), and tickmarkr recreates a worktree on every attempt, so every kimi
239
+ // dispatch after the first meets a brand-new path and a brand-new prompt. Neither `-y/--yolo` (tool
240
+ // calls) nor `--auto` (agent questions) covers it: the gate is raised before the agent loop.
241
+ //
242
+ // VERBATIM CAPTURES, read off live panes — never composed (v1.79 law). Each is quoted from the
243
+ // record that holds the FULLEST transcription of it, named here so a reviewer reads the same bytes:
244
+ // OBS-358, pane wW:p2TB, run-20260805-121252 (.overseer/REPAIR-v186/GATE-T28-1.md — five lines;
245
+ // .planning/OBSERVATIONS.md abridges the same capture to its first three):
246
+ // Trust this folder?
247
+ // /Users/…/.tickmarkr/worktrees.noindex/tickmarkr-run-20260805-121252--T29
248
+ // ❯ Trust this folder ← highlighted ("← highlighted" is the observer's annotation)
249
+ // Enable project MCP servers. Remembered for this folder.
250
+ // Don't trust
251
+ // OBS-406, pane wW:p32A, run-20260806-121758-…214 T5:
252
+ // Kimi Code loads project-level MCP servers (.mcp.json, .kimi-code/mcp.json) only in trusted folders.
253
+ // ❯ Trust this folder / Don't trust
254
+ //
255
+ // The fingerprint carries the SELECTION CURSOR, and that is the load-bearing half: OBS-406's own
256
+ // cursor-less watcher matched bare `Trust this folder` and produced 258 false wakes in 25 minutes —
257
+ // every one of them a supervisor pane with the words on screen as prose. A live modal renders a
258
+ // cursor glyph; prose never does. Enter accepts the highlighted "Trust this folder" option, and the
259
+ // two captures above are the reason the cursor+label pair is the fingerprint rather than either
260
+ // prompt's heading: the same gate ships two different headings.
261
+ export const KIMI_TRUST_DIALOG = {
262
+ fingerprint: "❯ Trust this folder",
263
+ key: "Enter",
264
+ };
237
265
  // v1.69 T7 / v1.71 T2: thin wrapper over the daemon's generic dispatch path for kimi-only tests.
238
266
  export async function runKimiInteractiveSeed(opts) {
239
267
  const r = await runInteractiveSeed({ ...opts, adapter: kimi });
@@ -260,6 +288,14 @@ export const kimi = {
260
288
  // v1.65 T3: every flag the command builders below hardcode (-S is resumeCommand's) — verified in
261
289
  // `kimi --help` 2026-07-22.
262
290
  hardcodedFlags: { binary: "kimi", flags: ["-p", "--model", "--output-format", "-S"] },
291
+ // KNOWN GAP, stated rather than implied: this declaration is consumed by the daemon's trust loop,
292
+ // which is entered only AFTER runInteractiveSeed returns — and that seed waits for
293
+ // `readinessMatch` for the whole task timeout, which is exactly the window this prompt blocks in
294
+ // (run …215). So on a kimi startup pane the declaration is correct and still unreached: the seed
295
+ // reports `readiness pattern not seen` with zero keypresses. Answering it during the readiness
296
+ // wait is a change to src/run/interactive-seed.ts — WHEN and HOW OFTEN a declaration is answered
297
+ // is T19's half of this contract, and src/run may not appear in this task's diff.
298
+ trustDialog: KIMI_TRUST_DIALOG,
263
299
  // KIMI-02 headless. -p prompt mode + explicit model, NO permission flag: kimi 0.26.0 rejects
264
300
  // -p combined with -y/--auto at argument parse time ("Cannot combine --prompt with --yolo",
265
301
  // OBS-67 — doctor probes all failed on it), and prompt mode is already non-interactive with
@@ -28,6 +28,23 @@ export const opencode = {
28
28
  return { command: this.headlessCommand(ctx.promptFile, a.model) };
29
29
  },
30
30
  parse: parseWorkerResult,
31
+ // v1.89 T1 / OBS-414: opencode 1.17.15 renders no workspace-trust gate, and the prompt it DOES
32
+ // render must never be declared here. A prior attempt at this contract pointed opencode's
33
+ // fingerprint at its tool-permission modal — captured verbatim as the three lines
34
+ // Permission required
35
+ // Allow once
36
+ // enter confirm
37
+ // The daemon answers a match with Enter, which selects "Allow once" and silently approves the
38
+ // first arbitrary tool request in every slot: stall protection converted into auto-approval
39
+ // (.planning/RULING-v189-T1-reauthor.md:11). Tool approval and workspace trust are different
40
+ // gates, and no line of that capture — including the keybar hint that slipped past the first
41
+ // refusal list — can be declared now: the schema requires a fingerprint that names a trust gate.
42
+ // FALSIFIER: a captured opencode pane showing a folder/workspace-trust prompt — not a permission
43
+ // prompt — replaces this with {fingerprint, key}.
44
+ trustDialog: {
45
+ kind: "none",
46
+ reason: "opencode 1.17.15 renders no workspace-trust prompt; its only modal is the tool-permission prompt (Permission required / Allow once), a different gate that must never be auto-answered; falsified by a captured opencode pane showing a folder-trust prompt",
47
+ },
31
48
  // v1.5 MODEL-01: fail OPEN to [] (advisory detection, unlike gates). Plain `models` (NOT --refresh):
32
49
  // opencode reads its own cache offline; --refresh adds an avoidable network dependency (RESEARCH
33
50
  // anti-pattern). Live-verified 2026-07-10, opencode 1.17.15.
@@ -62,6 +62,17 @@ export const pi = {
62
62
  return { command: this.headlessCommand(ctx.promptFile, a.model) };
63
63
  },
64
64
  parse: parseWorkerResult,
65
+ // v1.89 T1 / OBS-414: pi HAS a per-directory trust prompt, and both command builders above already
66
+ // suppress it with --approve — a global option legal in both modes (pi --help 0.80.3), chosen for
67
+ // exactly this reason. Live-checked 2026-07-10 in a fresh worktree: no trust prompt reached the
68
+ // pane, so there is nothing captured, and a fingerprint here would be a guess about a dialog this
69
+ // adapter never lets render.
70
+ // FALSIFIER: drop --approve from interactiveCommand, dispatch into a fresh worktree, and capture
71
+ // the prompt that appears — then this becomes {fingerprint, key} rather than a suppression claim.
72
+ trustDialog: {
73
+ kind: "none",
74
+ reason: "pi 0.80.3's per-directory trust prompt is suppressed before it renders by --approve on both command builders (live-checked 2026-07-10, fresh worktree, no prompt); falsified by removing --approve and capturing the pane",
75
+ },
65
76
  // v1.5 MODEL-01: fail OPEN to [] (detection is advisory — unlike gates' fail-closed posture).
66
77
  // spawnSync mirrors probeVersion; 15s timeout. Live-verified 2026-07-10, pi 0.80.3.
67
78
  listModels: async () => {
@@ -2,6 +2,13 @@ import { mkdirSync, writeFileSync } from "node:fs";
2
2
  import { join } from "node:path";
3
3
  import { renderAcceptanceItem } from "../graph/schema.js";
4
4
  import { classifyVerdictCause } from "../gates/verdict-cause.js";
5
+ // v1.87 T5: the scope rule states only what the runtime enforces, and no more. What it enforces:
6
+ // `scope.allowDeviations` comes from the operator's config, is snapshotted and frozen when a gate
7
+ // round starts (run-gates.ts `allowDeviations`), and is never written back — so a worker has no
8
+ // reach into it at all, and the round that judges its work judges against one fixed list. The
9
+ // retired "unless the operator's config allowlists that path" advertised an exemption the worker
10
+ // could neither invoke nor influence. Nothing here names a capability a worker can invoke mid-run,
11
+ // and the allowlist stays exactly as immutable during a run as it was before this text changed.
5
12
  export function buildTaskPrompt(task, feedback = "", nonce = "") {
6
13
  const list = (xs) => xs.map((x) => `- ${x}`).join("\n");
7
14
  return `You are an autonomous coding worker dispatched by tickmarkr into an isolated git worktree.
@@ -15,7 +22,7 @@ ${task.files.length ? `\n## File scope — touch ONLY paths matching:\n${list(ta
15
22
  ## Rules
16
23
  - Work only inside the current directory (your isolated worktree). Never push. Never switch branches.
17
24
  - Make small atomic git commits as you go (git add + git commit, conventional messages).
18
- - Touch ONLY paths matching the file scope. Out-of-scope edits FAIL the scope gate unless the operator's config allowlists that path — declaring a deviation never passes the gate. If you cannot complete the task without an out-of-scope edit, stop and report ok:false explaining why in "summary". List any out-of-scope paths you did touch, each with a reason, in "deviations" (journaled for the operator's audit).
25
+ - Touch ONLY paths matching the file scope. Out-of-scope edits FAIL the scope gate. The operator's allowlist is fixed when the run starts and nothing you do can change it while your work is judged; declaring a deviation never passes the gate either. If you cannot complete the task without an out-of-scope edit, stop and report ok:false explaining why in "summary". List any out-of-scope paths you did touch, each with a reason, in "deviations" (journaled for the operator's audit).
19
26
  - Do not ask questions; you are unattended. Make the smallest correct change.
20
27
  ${feedback ? `\n## Previous attempt failed gates — fix these specifically\n${feedback}\n` : ""}
21
28
  When finished, end your final message with exactly one line (no code fence):
@@ -1,5 +1,5 @@
1
1
  import { spawnSync } from "node:child_process";
2
- import { existsSync, mkdtempSync, readFileSync, realpathSync, renameSync, statSync, writeFileSync } from "node:fs";
2
+ import { existsSync, mkdtempSync, readFileSync, realpathSync, renameSync, rmSync, statSync, writeFileSync } from "node:fs";
3
3
  import { tmpdir } from "node:os";
4
4
  import { join } from "node:path";
5
5
  import { modelLints, suggestOverlay } from "./model-lints.js";
@@ -281,6 +281,9 @@ function declarativeAdapter(entry) {
281
281
  return health;
282
282
  },
283
283
  channels: (cfg) => channelsFromConfig(entry.id, cfg),
284
+ // v1.89 T1 / OBS-414: straight from the validated drive contract — no default and no fallback,
285
+ // which is the only way this factory cannot reintroduce the optionality the type just removed.
286
+ trustDialog: drive.trustDialog,
284
287
  headlessCommand: (promptFile, model) => renderDriveTemplate(drive.headless, promptFile, model),
285
288
  interactiveCommand: (promptFile, model) => drive.interactive === null
286
289
  ? null
@@ -395,67 +398,83 @@ export async function probeModels(cfg, repoRoot, adapters, health, onProgress) {
395
398
  // Tests inject FakeAdapter; never let an incidental default adapter spend a real token in the suite.
396
399
  if (process.env.VITEST && SHIPPED_NATIVE_ADAPTERS.has(a))
397
400
  return;
398
- const verdicts = {};
399
401
  const priorModelAuth = priorHealth?.[a.id]?.modelAuth;
400
- const probeRoot = a.probeCwd === "neutral" ? mkdtempSync(join(tmpdir(), "tickmarkr-probe-")) : repoRoot;
401
- const store = (model, verdict) => {
402
- verdicts[model] = verdict;
403
- onProgress?.(a.id, model, probeModelStatus(verdict), verdict.durationMs);
404
- };
405
- // One bounded probe call. verdict:null = a first-pass failure that earns the one serial retry
406
- // (OBS-72: a failure inside the concurrent batch is indistinguishable from adapter self-contention
407
- // until re-probed alone); a retry attempt always returns a final verdict.
408
- const attempt = async (model, retry) => {
409
- const t0 = Date.now();
410
- const probedAt = new Date().toISOString();
411
- const v = (authed, reason) => ({ authed, ...(reason !== undefined ? { reason } : {}), probedAt, durationMs: Date.now() - t0 });
412
- try {
413
- if (typeof a.headlessCommand !== "function")
414
- return { verdict: v(false, "headless probe unavailable"), timedOut: false };
415
- const promptFile = join(mkdtempSync(join(tmpdir(), "tickmarkr-auth-")), "probe.md");
416
- writeFileSync(promptFile, MODEL_PROBE_PROMPT);
417
- const r = await sh(a.headlessCommand(promptFile, model), probeRoot, MODEL_PROBE_TIMEOUT_MS);
418
- // T2 rule unchanged: a prior doctor.json timeout skips the retry — a persistently dead
419
- // model (e.g. opencode glm-5.2) costs one attempt instead of two every run.
420
- if (r.timedOut && !retry && priorModelAuth?.[model]?.reason?.includes("timed out") === true) {
421
- return { verdict: v(false, `probe timed out (repeat — retry skipped) (${MODEL_PROBE_TIMEOUT_MS}ms)`), timedOut: true };
402
+ const runProbes = async (probeRoot) => {
403
+ const verdicts = {};
404
+ const store = (model, verdict) => {
405
+ verdicts[model] = verdict;
406
+ onProgress?.(a.id, model, probeModelStatus(verdict), verdict.durationMs);
407
+ };
408
+ // One bounded probe call. verdict:null = a first-pass failure that earns the one serial retry
409
+ // (OBS-72: a failure inside the concurrent batch is indistinguishable from adapter self-contention
410
+ // until re-probed alone); a retry attempt always returns a final verdict.
411
+ const attempt = async (model, retry) => {
412
+ const t0 = Date.now();
413
+ const probedAt = new Date().toISOString();
414
+ const v = (authed, reason) => ({ authed, ...(reason !== undefined ? { reason } : {}), probedAt, durationMs: Date.now() - t0 });
415
+ try {
416
+ if (typeof a.headlessCommand !== "function")
417
+ return { verdict: v(false, "headless probe unavailable"), timedOut: false };
418
+ const promptDir = mkdtempSync(join(tmpdir(), "tickmarkr-auth-"));
419
+ try {
420
+ const promptFile = join(promptDir, "probe.md");
421
+ writeFileSync(promptFile, MODEL_PROBE_PROMPT);
422
+ const r = await sh(a.headlessCommand(promptFile, model), probeRoot, MODEL_PROBE_TIMEOUT_MS);
423
+ // T2 rule unchanged: a prior doctor.json timeout skips the retry — a persistently dead
424
+ // model (e.g. opencode glm-5.2) costs one attempt instead of two every run.
425
+ if (r.timedOut && !retry && priorModelAuth?.[model]?.reason?.includes("timed out") === true) {
426
+ return { verdict: v(false, `probe timed out (repeat — retry skipped) (${MODEL_PROBE_TIMEOUT_MS}ms)`), timedOut: true };
427
+ }
428
+ const reason = probeFailure(r.code, r.stdout, r.stderr, r.timedOut, MODEL_PROBE_TIMEOUT_MS);
429
+ if (!reason)
430
+ return { verdict: v(true), timedOut: false };
431
+ if (!retry)
432
+ return { verdict: null, timedOut: r.timedOut === true };
433
+ return {
434
+ verdict: r.timedOut && retry.firstTimedOut ? v(false, `probe timed out twice (${MODEL_PROBE_TIMEOUT_MS}ms)`) : v(false, reason),
435
+ timedOut: r.timedOut === true,
436
+ };
437
+ }
438
+ finally {
439
+ rmSync(promptDir, { recursive: true, force: true });
440
+ }
422
441
  }
423
- const reason = probeFailure(r.code, r.stdout, r.stderr, r.timedOut, MODEL_PROBE_TIMEOUT_MS);
424
- if (!reason)
425
- return { verdict: v(true), timedOut: false };
426
- if (!retry)
427
- return { verdict: null, timedOut: r.timedOut === true };
428
- return {
429
- verdict: r.timedOut && retry.firstTimedOut ? v(false, `probe timed out twice (${MODEL_PROBE_TIMEOUT_MS}ms)`) : v(false, reason),
430
- timedOut: r.timedOut === true,
431
- };
432
- }
433
- catch (e) {
434
- return { verdict: retry ? v(false, String(e)) : null, timedOut: false };
442
+ catch (e) {
443
+ return { verdict: retry ? v(false, String(e)) : null, timedOut: false };
444
+ }
445
+ };
446
+ // models probe concurrently too (v1.33.5) — sequential chains made init wall time Σ(models×60s)
447
+ // on the slowest adapter (measured 96s); each probe is one tiny prompt, safe to overlap.
448
+ // Capped at MODEL_PROBE_CONCURRENCY per adapter unless the adapter declares its own cap
449
+ // (codex: 1 — OBS-72, its concurrent probes self-contend in one repo).
450
+ const retries = [];
451
+ await mapLimit(Object.keys(cfg.tiers[a.id]?.models ?? {}), a.probeConcurrency ?? MODEL_PROBE_CONCURRENCY, async (model) => {
452
+ const first = await attempt(model);
453
+ if (first.verdict)
454
+ store(model, first.verdict);
455
+ else
456
+ retries.push({ model, firstTimedOut: first.timedOut });
457
+ });
458
+ // OBS-72: re-probe each first-pass failure once with ONE probe in flight, only after the
459
+ // concurrent batch drained — an in-slot retry still races its concurrency partner. Success here
460
+ // was contention; a second failure is the real verdict. Successful first-pass probes stored
461
+ // above and never wait; cost is one extra call per genuinely dead model only.
462
+ for (const { model, firstTimedOut } of retries) {
463
+ const second = await attempt(model, { firstTimedOut });
464
+ if (second.verdict)
465
+ store(model, second.verdict);
435
466
  }
467
+ h.modelAuth = verdicts;
436
468
  };
437
- // models probe concurrently too (v1.33.5) — sequential chains made init wall time Σ(models×60s)
438
- // on the slowest adapter (measured 96s); each probe is one tiny prompt, safe to overlap.
439
- // Capped at MODEL_PROBE_CONCURRENCY per adapter unless the adapter declares its own cap
440
- // (codex: 1 — OBS-72, its concurrent probes self-contend in one repo).
441
- const retries = [];
442
- await mapLimit(Object.keys(cfg.tiers[a.id]?.models ?? {}), a.probeConcurrency ?? MODEL_PROBE_CONCURRENCY, async (model) => {
443
- const first = await attempt(model);
444
- if (first.verdict)
445
- store(model, first.verdict);
446
- else
447
- retries.push({ model, firstTimedOut: first.timedOut });
448
- });
449
- // OBS-72: re-probe each first-pass failure once with ONE probe in flight, only after the
450
- // concurrent batch drained — an in-slot retry still races its concurrency partner. Success here
451
- // was contention; a second failure is the real verdict. Successful first-pass probes stored
452
- // above and never wait; cost is one extra call per genuinely dead model only.
453
- for (const { model, firstTimedOut } of retries) {
454
- const second = await attempt(model, { firstTimedOut });
455
- if (second.verdict)
456
- store(model, second.verdict);
469
+ if (a.probeCwd !== "neutral")
470
+ return runProbes(repoRoot);
471
+ const probeRoot = mkdtempSync(join(tmpdir(), "tickmarkr-probe-"));
472
+ try {
473
+ await runProbes(probeRoot);
474
+ }
475
+ finally {
476
+ rmSync(probeRoot, { recursive: true, force: true });
457
477
  }
458
- h.modelAuth = verdicts;
459
478
  }));
460
479
  }
461
480
  // T2: caps concurrent probes per adapter at 2 — v1.33.5 regression, 4 concurrent codex exec in one
@@ -77,11 +77,42 @@ export type TrustVerdict = {
77
77
  status: "action-required";
78
78
  command: string;
79
79
  };
80
- export interface TrustDialog {
80
+ export interface CapturedTrustDialog {
81
+ kind?: "dialog";
81
82
  fingerprint: string;
82
83
  key: string;
83
84
  }
84
- export declare function matchesTrustDialog(paneText: string, dialog: TrustDialog): boolean;
85
+ export interface NoTrustDialog {
86
+ kind: "none";
87
+ reason: string;
88
+ }
89
+ export type TrustDialog = CapturedTrustDialog | NoTrustDialog;
90
+ export declare const TRUST_DIALOG_BLANK_MESSAGE = "trust-dialog fingerprint must be verbatim bytes captured from a workspace-trust prompt \u2014 a blank or whitespace-only fingerprint matches every pane";
91
+ export declare const CLAUDE_TRUST_PANE: string;
92
+ export declare const CODEX_TRUST_PANE: string;
93
+ export declare const CURSOR_TRUST_PANE = "Workspace Trust Required\nTrust this folder?";
94
+ export declare const KIMI_TRUST_PANE: string;
95
+ export declare const KIMI_MCP_TRUST_PANE: string;
96
+ export declare const RECORDED_TRUST_PANES: readonly string[];
97
+ export declare const APPROVED_TRUST_FINGERPRINTS: readonly string[];
98
+ export declare const TRUST_DIALOG_UNRECORDED_MESSAGE = "trust-dialog fingerprint is not one of the approved workspace-trust captures \u2014 it must be exactly an entry of APPROVED_TRUST_FINGERPRINTS in src/adapters/types.ts (each the distinctive bytes of a recorded pane), not a prefix of one, a substring of one, or a sentence describing the prompt; declare {kind: none, reason} if the CLI renders none";
99
+ export declare const TRUST_DIALOG_VARIANTS: readonly [z.ZodObject<{
100
+ kind: z.ZodOptional<z.ZodLiteral<"dialog">>;
101
+ fingerprint: z.ZodString;
102
+ key: z.ZodString;
103
+ }, z.core.$strict>, z.ZodObject<{
104
+ kind: z.ZodLiteral<"none">;
105
+ reason: z.ZodString;
106
+ }, z.core.$strict>];
107
+ export declare const TrustDialogSchema: z.ZodUnion<readonly [z.ZodObject<{
108
+ kind: z.ZodOptional<z.ZodLiteral<"dialog">>;
109
+ fingerprint: z.ZodString;
110
+ key: z.ZodString;
111
+ }, z.core.$strict>, z.ZodObject<{
112
+ kind: z.ZodLiteral<"none">;
113
+ reason: z.ZodString;
114
+ }, z.core.$strict>]>;
115
+ export declare function matchesTrustDialog(paneText: string, dialog: TrustDialog): dialog is CapturedTrustDialog;
85
116
  export interface InputBox {
86
117
  fingerprint: string;
87
118
  match?(paneText: string): boolean;
@@ -118,7 +149,7 @@ export interface WorkerAdapter {
118
149
  collectUsage?(cwd: string, sinceMs: number): TokenUsage | undefined;
119
150
  contextUsage?(session: SessionRef): ContextUsage | null;
120
151
  trust?(repoRoot: string): TrustVerdict;
121
- trustDialog?: TrustDialog;
152
+ trustDialog: TrustDialog;
122
153
  inputBox?: InputBox;
123
154
  hardcodedFlags?: {
124
155
  binary: string;
@@ -27,8 +27,106 @@ export function modelAuthed(health, model, allowUnverifiedModels = false) {
27
27
  const authed = health?.modelAuth?.[model]?.authed;
28
28
  return authed === true || (authed === undefined && allowUnverifiedModels);
29
29
  }
30
+ export const TRUST_DIALOG_BLANK_MESSAGE = "trust-dialog fingerprint must be verbatim bytes captured from a workspace-trust prompt — a blank or whitespace-only fingerprint matches every pane";
31
+ // v1.89 T1 / OBS-414 round 3 — THE EVIDENCE. The word "trust" is not evidence of a capture: codex's
32
+ // real recorded gate is "Do you trust the contents of this directory?" and the handwritten prose
33
+ // "Do you trust this command?" is one word away, so no regex separates them — only the record does.
34
+ // Round 2 accepted the prose, and the daemon presses Enter on a match, which is auto-approval of an
35
+ // arbitrary tool call. So a fingerprint must be an ENUMERATED capture (APPROVED_TRUST_FINGERPRINTS
36
+ // below), and every pane here is VERBATIM, quoted from the record named beside it — the panes are the
37
+ // evidence each approved entry is checked against, never the acceptance rule themselves.
38
+ //
39
+ // Cost of the posture, stated plainly: an operator adding a CLI through the YAML drive contract
40
+ // cannot declare a trust dialog whose capture is not enumerated here — they declare {kind:"none",
41
+ // reason} and page a human, or contribute the capture. That is deliberate. The alternative is a stall
42
+ // protection that any plausible sentence can turn into a permission auto-approver.
43
+ // v1.75 T2 / OBS-137, claude-code 2.1.218 startup gate.
44
+ export const CLAUDE_TRUST_PANE = [
45
+ "Accessing workspace:",
46
+ "/tmp/untrusted-project",
47
+ "Quick safety check: Is this a project you created or one you trust? (Like your own code, a well-known open source project, or work from your team).",
48
+ "Yes, I trust this folder",
49
+ ].join("\n");
50
+ // v1.75 T2 / OBS-137, codex 0.144.6 startup gate.
51
+ export const CODEX_TRUST_PANE = [
52
+ "Do you trust the contents of this directory?",
53
+ "Working with untrusted contents comes with higher risk of prompt injection.",
54
+ "Trusting the directory allows project-local config, hooks, and exec policies to load.",
55
+ "› 1. Yes, continue",
56
+ "Press enter to continue",
57
+ ].join("\n");
58
+ // v1.22 T5 / OBS-19, cursor-agent's per-worktree dialog.
59
+ export const CURSOR_TRUST_PANE = "Workspace Trust Required\nTrust this folder?";
60
+ // OBS-358, live pane wW:p2TB of run-20260805-121252 — the dialog that cost 30 minutes — quoted from
61
+ // the record holding all five lines, .overseer/REPAIR-v186/GATE-T28-1.md (OBSERVATIONS.md abridges
62
+ // it to the first three). "← highlighted" is the observer's annotation of the selected row, kept
63
+ // exactly as recorded rather than tidied away.
64
+ export const KIMI_TRUST_PANE = [
65
+ "Trust this folder?",
66
+ " /Users/…/.tickmarkr/worktrees.noindex/tickmarkr-run-20260805-121252--T29",
67
+ "❯ Trust this folder ← highlighted",
68
+ " Enable project MCP servers. Remembered for this folder.",
69
+ " Don't trust",
70
+ ].join("\n");
71
+ // OBS-406, live pane wW:p32A of run-20260806-121758-…214 T5 — the same gate in its MCP-trust
72
+ // wording, which is why kimi's fingerprint is the cursor+option row both headings share.
73
+ export const KIMI_MCP_TRUST_PANE = [
74
+ "Kimi Code loads project-level MCP servers (.mcp.json, .kimi-code/mcp.json) only in trusted folders.",
75
+ " ❯ Trust this folder / Don't trust",
76
+ ].join("\n");
77
+ export const RECORDED_TRUST_PANES = [
78
+ CLAUDE_TRUST_PANE, CODEX_TRUST_PANE, CURSOR_TRUST_PANE, KIMI_TRUST_PANE, KIMI_MCP_TRUST_PANE,
79
+ ];
80
+ // v1.89 T1 / OBS-414 round 4 — EXACT, not "leading bytes of a recorded line". Round 3 accepted any
81
+ // prefix of a captured line, and every prefix of a trust prompt is also a substring of unrelated
82
+ // panes: `{fingerprint: "Trust"}` passed, then matched a tool-permission pane reading
83
+ // "Trust this command?" and handed the daemon an Enter to press on it — the auto-approval defect,
84
+ // reopened by the check meant to close it. `"Trust this folder"` (kimi's row minus its cursor glyph)
85
+ // passed the same way, and that exact string is the one OBS-406 measured producing 258 false wakes
86
+ // in 25 minutes on supervisor panes with the words on screen as prose.
87
+ //
88
+ // So the approved fingerprints are ENUMERATED, one per shipped capture, each the distinctive bytes
89
+ // of its own pane and nothing shorter. Kimi's carries the selection cursor because a live modal
90
+ // renders one and prose never does. Membership is exact; the corpus check below it keeps a list
91
+ // entry from drifting away from the pane it claims to quote.
92
+ export const APPROVED_TRUST_FINGERPRINTS = [
93
+ "Quick safety check: Is this a project you created or one you trust?", // claude-code 2.1.218, OBS-137
94
+ "Do you trust the contents of this directory?", // codex 0.144.6, OBS-137
95
+ "Workspace Trust Required", // cursor-agent, OBS-19
96
+ "❯ Trust this folder", // kimi 0.29.0, OBS-358 + OBS-406 — the cursor glyph is load-bearing
97
+ ];
98
+ const isApprovedFingerprint = (fingerprint) => APPROVED_TRUST_FINGERPRINTS.includes(fingerprint)
99
+ && RECORDED_TRUST_PANES.some((pane) => pane.includes(fingerprint));
100
+ export const TRUST_DIALOG_UNRECORDED_MESSAGE = "trust-dialog fingerprint is not one of the approved workspace-trust captures — it must be exactly an entry of APPROVED_TRUST_FINGERPRINTS in src/adapters/types.ts (each the distinctive bytes of a recorded pane), not a prefix of one, a substring of one, or a sentence describing the prompt; declare {kind: none, reason} if the CLI renders none";
101
+ // The two variants, exported as one tuple so every schema that embeds a trust declaration (the
102
+ // drive contract in catalog.ts) is built from these bytes rather than restating them.
103
+ export const TRUST_DIALOG_VARIANTS = [
104
+ z.object({
105
+ kind: z.literal("dialog").optional(),
106
+ // superRefine, not chained refine(): one failure yields ONE diagnosis, so the operator reads the
107
+ // reason their declaration was refused instead of every rule it happened to trip.
108
+ fingerprint: z.string().superRefine((fingerprint, ctx) => {
109
+ const message = fingerprint.trim().length === 0 ? TRUST_DIALOG_BLANK_MESSAGE
110
+ : isApprovedFingerprint(fingerprint) ? undefined
111
+ : TRUST_DIALOG_UNRECORDED_MESSAGE;
112
+ if (message)
113
+ ctx.addIssue({ code: "custom", message });
114
+ }),
115
+ key: z.string().min(1),
116
+ }).strict(),
117
+ z.object({
118
+ kind: z.literal("none"),
119
+ reason: z.string().refine((r) => r.trim().length > 0, "a no-dialog declaration must state a falsifiable reason"),
120
+ }).strict(),
121
+ ];
122
+ export const TrustDialogSchema = z.union(TRUST_DIALOG_VARIANTS);
123
+ // The discrimination lives HERE, at the keypress boundary every caller crosses: a no-dialog
124
+ // declaration never matches, so sendKey is unreachable for it without editing this function. The
125
+ // blank guard is the same fail-closed posture one layer below the schema.
30
126
  export function matchesTrustDialog(paneText, dialog) {
31
- return paneText.includes(dialog.fingerprint);
127
+ if (dialog.kind === "none")
128
+ return false;
129
+ return dialog.fingerprint.trim().length > 0 && paneText.includes(dialog.fingerprint);
32
130
  }
33
131
  const inputBoxes = new Map();
34
132
  export function declareInputBox(adapterId, inputBox) {