tickmarkr 2.3.0 → 2.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/dist/adapters/catalog-remote.d.ts +1 -4
  2. package/dist/adapters/catalog-remote.js +52 -42
  3. package/dist/adapters/catalog.js +5 -3
  4. package/dist/adapters/claude-code.d.ts +1 -1
  5. package/dist/adapters/claude-code.js +8 -5
  6. package/dist/adapters/model-lints.d.ts +9 -5
  7. package/dist/adapters/model-lints.js +56 -15
  8. package/dist/adapters/model-windows.js +11 -0
  9. package/dist/adapters/prompt.js +1 -0
  10. package/dist/adapters/qwen.d.ts +5 -0
  11. package/dist/adapters/qwen.js +153 -0
  12. package/dist/adapters/types.d.ts +1 -0
  13. package/dist/adapters/types.js +1 -0
  14. package/dist/cli/commands/compile.js +32 -6
  15. package/dist/cli/commands/doctor.d.ts +4 -3
  16. package/dist/cli/commands/doctor.js +20 -6
  17. package/dist/cli/commands/fleet.d.ts +4 -0
  18. package/dist/cli/commands/fleet.js +53 -14
  19. package/dist/cli/commands/plan.js +29 -5
  20. package/dist/cli/commands/status.d.ts +1 -0
  21. package/dist/cli/commands/status.js +45 -1
  22. package/dist/cli/commands/verify.d.ts +1 -0
  23. package/dist/cli/commands/verify.js +5 -0
  24. package/dist/cli/index.d.ts +1 -1
  25. package/dist/cli/index.js +2 -2
  26. package/dist/compile/collateral.d.ts +2 -9
  27. package/dist/compile/collateral.js +2 -9
  28. package/dist/compile/index.d.ts +4 -1
  29. package/dist/compile/index.js +41 -7
  30. package/dist/compile/native.d.ts +4 -2
  31. package/dist/compile/native.js +55 -6
  32. package/dist/compile/ownership.js +34 -9
  33. package/dist/config/config.d.ts +1 -0
  34. package/dist/config/config.js +51 -5
  35. package/dist/drivers/herdr.d.ts +1 -0
  36. package/dist/drivers/herdr.js +11 -1
  37. package/dist/drivers/orca.d.ts +14 -1
  38. package/dist/drivers/orca.js +114 -15
  39. package/dist/drivers/types.d.ts +10 -0
  40. package/dist/gates/baseline.js +26 -7
  41. package/dist/gates/review.d.ts +4 -2
  42. package/dist/gates/review.js +24 -13
  43. package/dist/gates/run-gates.d.ts +5 -2
  44. package/dist/gates/run-gates.js +34 -17
  45. package/dist/route/preference.d.ts +4 -0
  46. package/dist/route/preference.js +40 -0
  47. package/dist/route/router.js +15 -2
  48. package/dist/run/consult.d.ts +1 -0
  49. package/dist/run/consult.js +34 -7
  50. package/dist/run/daemon.d.ts +4 -0
  51. package/dist/run/daemon.js +128 -50
  52. package/dist/run/git.d.ts +2 -0
  53. package/dist/run/git.js +36 -5
  54. package/dist/run/journal.js +4 -1
  55. package/dist/tui/ink/fleet-app.d.ts +4 -0
  56. package/dist/tui/ink/fleet-app.js +45 -16
  57. package/package.json +59 -1
@@ -1,5 +1,6 @@
1
1
  import { existsSync, readFileSync, statSync } from "node:fs";
2
2
  import { join } from "node:path";
3
+ import { parse } from "yaml";
3
4
  import { validateGraph } from "../graph/schema.js";
4
5
  import { taskUnitContractErrors } from "./collateral.js";
5
6
  import { blocksCompile, ownershipFindings, renderOwnershipFinding } from "./ownership.js";
@@ -33,8 +34,40 @@ function detect(src) {
33
34
  // tasks are too large to converge. A violation is a compile error, never a warning — the failures it
34
35
  // prevents (silently dropped commits, a 28-dispatch task) are invisible until they have already cost
35
36
  // hours, which is exactly the class of thing that has to fail at authoring time.
36
- function enforceTaskUnitContract(g, src) {
37
- const errors = taskUnitContractErrors(g.tasks);
37
+ function repoOverlayMode(repoRoot) {
38
+ if (!repoRoot)
39
+ return undefined;
40
+ try {
41
+ const cfg = parse(readFileSync(join(repoRoot, ".tickmarkr", "config.yaml"), "utf8"));
42
+ return typeof cfg?.routing?.mode === "string" ? cfg.routing.mode : undefined;
43
+ }
44
+ catch {
45
+ return undefined;
46
+ }
47
+ }
48
+ function hasModeOverride(src) {
49
+ try {
50
+ for (const line of readFileSync(src, "utf8").split("\n")) {
51
+ if (/^##\s+T\d+:/i.test(line))
52
+ return false;
53
+ if (/^mode-override:\s*true\s*(?:#.*)?$/.test(line))
54
+ return true;
55
+ }
56
+ }
57
+ catch { /* unreadable specs fail elsewhere */ }
58
+ return false;
59
+ }
60
+ function enforceModeOverlay(graph, src, repoRoot) {
61
+ if (graph.spec.source !== "native" || graph.mode === undefined)
62
+ return;
63
+ const overlayMode = repoOverlayMode(repoRoot);
64
+ if (overlayMode === undefined || overlayMode === graph.mode || hasModeOverride(src))
65
+ return;
66
+ throw new CompileError(`${src} front-matter mode ${graph.mode} disagrees with repository routing.mode ${overlayMode}; `
67
+ + `write mode-override: true beside the mode line to make the override explicit.`);
68
+ }
69
+ function enforceTaskUnitContract(g, src, repoRoot) {
70
+ const errors = taskUnitContractErrors(g.tasks, repoRoot);
38
71
  if (errors.length > 0) {
39
72
  throw new CompileError(`${src} violates the task unit contract (${errors.length} error${errors.length > 1 ? "s" : ""}):\n`
40
73
  + errors.map((e) => ` - ${e}`).join("\n"));
@@ -52,7 +85,8 @@ export function finalizePlan(plan, src, repoRoot) {
52
85
  ...(plan.base !== undefined ? { base: plan.base } : {}),
53
86
  },
54
87
  tasks: plan.tasks,
55
- }), src);
88
+ }), src, repoRoot);
89
+ enforceModeOverlay(graph, src, repoRoot);
56
90
  // overseer-217 removal condition, now paid: on this milestone's authored graph the conventional
57
91
  // name map emitted 21 raw unowned-test findings; review found 1 real and 20 false, while intersecting
58
92
  // with a direct import or command-entry spawn retained the real one and left 0 false positives. That
@@ -73,11 +107,11 @@ export function finalizePlan(plan, src, repoRoot) {
73
107
  }
74
108
  return graph;
75
109
  }
76
- function compilePlan(src, type, root) {
110
+ function compilePlan(src, type, root, options = {}) {
77
111
  const kind = type ?? detect(src);
78
112
  const graph = kind === "speckit" ? compileSpecKit(src)
79
113
  : kind === "gsd" ? compileGsd(src, root)
80
- : kind === "native" ? compileNative(src)
114
+ : kind === "native" ? compileNative(src, { strict: options.strict })
81
115
  : kind === "prd" ? compilePrd(src)
82
116
  : null;
83
117
  if (!graph) {
@@ -93,6 +127,6 @@ function compilePlan(src, type, root) {
93
127
  tasks: graph.tasks,
94
128
  };
95
129
  }
96
- export function compileSource(src, type, root, beforeFinalize = (plan) => plan) {
97
- return finalizePlan(beforeFinalize(compilePlan(src, type, root)), src, root);
130
+ export function compileSource(src, type, root, beforeFinalize = (plan) => plan, options = {}) {
131
+ return finalizePlan(beforeFinalize(compilePlan(src, type, root, options)), src, root);
98
132
  }
@@ -17,7 +17,7 @@ export declare const COLLECTABLE_TESTS = "tests/**/*.test.ts";
17
17
  export declare const LEGACY_PREFIX: string;
18
18
  export declare const TICKMARKR_NATIVE_MARKER: RegExp;
19
19
  export declare const NATIVE_MARKER: RegExp;
20
- export declare const AUTHORING_LINT_CODES: readonly ["criterion-scope", "dependency-closure", "dependency-coupling", "proxy-metric", "external-referent", "closed-enumeration", "one-behavior", "concern-bundle", "seam-exists"];
20
+ export declare const AUTHORING_LINT_CODES: readonly ["criterion-scope", "dependency-closure", "dependency-coupling", "proxy-metric", "external-referent", "closed-enumeration", "one-behavior", "concern-bundle", "seam-exists", "fence-symbol-absent"];
21
21
  export type AuthoringLintCode = (typeof AUTHORING_LINT_CODES)[number];
22
22
  export interface AuthoringLintFinding {
23
23
  code: AuthoringLintCode;
@@ -31,5 +31,7 @@ export interface AuthoringLintFinding {
31
31
  * compile boundary; the remaining checks are review findings, emitted by compileNative below.
32
32
  */
33
33
  export declare function authoringLintFindings(tasks: readonly Task[], file: string): AuthoringLintFinding[];
34
- export declare function compileNative(file: string): RunGraph;
34
+ export declare function compileNative(file: string, options?: {
35
+ strict?: boolean;
36
+ }): RunGraph;
35
37
  export declare function specTemplate(): string;
@@ -83,6 +83,7 @@ export const AUTHORING_LINT_CODES = [
83
83
  "one-behavior",
84
84
  "concern-bundle",
85
85
  "seam-exists",
86
+ "fence-symbol-absent",
86
87
  ];
87
88
  // The authoring frame's observable oracle is the branch-tip test corpus, never the mutable index or
88
89
  // an author's untracked checkout. Load it once per repository: one bounded git read replaces a grep
@@ -217,6 +218,31 @@ function criterionScopeFinding(task, text, criterion, id, tests) {
217
218
  detail: `criterion names asserting file${missing.length === 1 ? "" : "s"} outside expanded files[] and context[]: ${missing.join(", ")} — ${remedy}`,
218
219
  };
219
220
  }
221
+ // OBS-898: the fence lint's corpus is listed ONCE per compile by git (tracked plus untracked-not-ignored,
222
+ // the same set a checkout walk sees minus the ignored trees) and filtered per task. The walk it replaces
223
+ // descended every dot-directory (.tickmarkr/runs alone held 3 799 files here) once PER TASK, so a
224
+ // 20-task spec compiled in seconds and the committed-spec corpus test timed out. Fail-open like
225
+ // testsAtHead: no git answer means no corpus, and the lint stays quiet rather than wrong.
226
+ function repoFileList(root) {
227
+ const ls = spawnSync("git", ["-C", root, "ls-files", "-z", "--cached", "--others", "--exclude-standard"], { encoding: "utf8", maxBuffer: 1 << 25 });
228
+ if (ls.status !== 0 || typeof ls.stdout !== "string")
229
+ return [];
230
+ return ls.stdout.split("\0").filter(Boolean).sort();
231
+ }
232
+ function repoFiles(files, entries) {
233
+ const scoped = filesGlob(entries.map((entry) => entry.replace(/^\.\//, "")));
234
+ return files.filter(scoped);
235
+ }
236
+ const PRESERVATION_RE = /\b(?:keeps?|preserves?|retains?|does\s+not\s+weaken|do\s+not\s+weaken|not\s+weaken)\b/i;
237
+ const IDENTIFIER_SHAPED_RE = /^[A-Za-z_$][A-Za-z0-9_$]*$/;
238
+ function fencedIdentifiers(text) {
239
+ if (!PRESERVATION_RE.test(text))
240
+ return [];
241
+ return [...new Set([...text.matchAll(/`([^`\n]{1,120})`/g)]
242
+ .map((match) => match[1].trim())
243
+ .filter((token) => IDENTIFIER_SHAPED_RE.test(token) && /[A-Z0-9_$]/.test(token)))]
244
+ .sort();
245
+ }
220
246
  function exportedIdentifier(root, files, identifier) {
221
247
  const escaped = identifier.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
222
248
  const declaration = new RegExp(`\\bexport\\s+(?:(?:declare|default|async)\\s+)*(?:function|class|const|let|var|interface|type)\\s+${escaped}\\b|\\bexport\\s*\\{[^}]*\\b${escaped}\\b`);
@@ -240,14 +266,33 @@ export function authoringLintFindings(tasks, file) {
240
266
  const root = repositoryRoot(file);
241
267
  let indexedTests;
242
268
  const testIndex = () => indexedTests ??= root ? testsAtHead(root) : [];
269
+ let indexedFiles;
270
+ const fileIndex = () => indexedFiles ??= root ? repoFileList(root) : [];
243
271
  const findings = [];
244
272
  for (const task of tasks) {
245
273
  const criteria = criterionTexts(task);
274
+ const taskFiles = root && task.files.length ? repoFiles(fileIndex(), task.files) : [];
275
+ const taskTexts = new Map(taskFiles.map((path) => {
276
+ try {
277
+ return [path, readFileSync(join(root, path), "utf8")];
278
+ }
279
+ catch {
280
+ return [path, ""];
281
+ }
282
+ }));
246
283
  for (const [index, text] of criteria.entries()) {
247
284
  const needsTestIndex = /`[^`\n]{2,120}`|\b\d+\s*(?:\/|of)\s*\d+\b|\b[A-Za-z0-9_.-]+\.test\.ts\b/.test(text);
248
285
  const scope = criterionScopeFinding(task, text, index + 1, id, needsTestIndex ? testIndex() : []);
249
286
  if (scope)
250
287
  findings.push(scope);
288
+ for (const symbol of fencedIdentifiers(text)) {
289
+ if (!taskTexts.size || [...taskTexts.values()].some((body) => body.includes(symbol)))
290
+ continue;
291
+ findings.push({
292
+ code: "fence-symbol-absent", fixtureId: id, taskId: task.id, criterion: index + 1,
293
+ detail: `preservation fence cites \`${symbol}\` but that symbol has zero hits in this task's files[] (${taskFiles.join(", ")}) — OBS-604`,
294
+ });
295
+ }
251
296
  if (/line[- ]count|physical line|not greater than (?:the|\d)|no larger than|at most \d+ (?:lines|bytes)/i.test(text)) {
252
297
  findings.push({ code: "proxy-metric", fixtureId: id, taskId: task.id, criterion: index + 1, detail: "criterion uses a proxy size metric; state the structural intent instead" });
253
298
  }
@@ -308,7 +353,7 @@ function renderAuthoringFinding(finding) {
308
353
  const criterion = finding.criterion === undefined ? "" : ` criterion ${finding.criterion}`;
309
354
  return `tickmarkr: authoring-lint[${finding.code}] fixture ${finding.fixtureId} task ${finding.taskId}${criterion}: ${finding.detail}`;
310
355
  }
311
- export function compileNative(file) {
356
+ export function compileNative(file, options = {}) {
312
357
  if (!existsSync(file))
313
358
  throw new CompileError(`no such native spec file: ${file}`);
314
359
  const content = readFileSync(file, "utf8");
@@ -638,13 +683,17 @@ export function compileNative(file) {
638
683
  tasks,
639
684
  });
640
685
  const authoringFindings = authoringLintFindings(result.tasks, file);
641
- for (const finding of authoringFindings.filter(({ code }) => code !== "criterion-scope")) {
686
+ const blockingCodes = new Set(["criterion-scope", "fence-symbol-absent"]);
687
+ const blockingFindings = authoringFindings.filter((finding) => options.strict || blockingCodes.has(finding.code));
688
+ for (const finding of authoringFindings.filter((finding) => !blockingFindings.includes(finding))) {
642
689
  console.warn(renderAuthoringFinding(finding));
643
690
  }
644
- const scopeErrors = authoringFindings.filter(({ code }) => code === "criterion-scope");
645
- if (scopeErrors.length > 0) {
646
- throw new CompileError(`${file} violates the criterion-scope authoring lint (${scopeErrors.length} error${scopeErrors.length === 1 ? "" : "s"}):\n`
647
- + scopeErrors.map((finding) => ` - ${renderAuthoringFinding(finding)}`).join("\n"));
691
+ if (blockingFindings.length > 0) {
692
+ const strict = options.strict ? " under --strict" : "";
693
+ const onlyCriterionScope = blockingFindings.every(({ code }) => code === "criterion-scope");
694
+ const label = onlyCriterionScope && !options.strict ? "the criterion-scope authoring lint" : `authoring lints${strict}`;
695
+ throw new CompileError(`${file} violates ${label} (${blockingFindings.length} error${blockingFindings.length === 1 ? "" : "s"}):\n`
696
+ + blockingFindings.map((finding) => ` - ${renderAuthoringFinding(finding)}`).join("\n"));
648
697
  }
649
698
  // v1.19 read-old/write-new: a plain-string acceptance item compiles as a judge oracle. This is the
650
699
  // one-time nudge toward typed oracles (command/test/judge); PRD/Spec Kit/GSD stay silent (untouched).
@@ -22,14 +22,31 @@ function testSources(repoRoot) {
22
22
  return [];
23
23
  }
24
24
  }
25
- function namedSources(test, tasks) {
25
+ function sourceFiles(repoRoot) {
26
+ const root = join(repoRoot, "src");
27
+ try {
28
+ return readdirSync(root, { recursive: true, encoding: "utf8" })
29
+ .filter((path) => /\.(?:[cm]?[jt]sx?)$/.test(path))
30
+ .map((path) => `src/${normalize(path)}`)
31
+ .sort();
32
+ }
33
+ catch {
34
+ return [];
35
+ }
36
+ }
37
+ // OBS-898: `allSources` is indexed once per ownership pass by the caller, never per test source.
38
+ function namedSources(test, tasks, allSources) {
26
39
  const stem = basename(test).replace(/\.test\.ts$/, "");
27
40
  const matches = new Map();
28
41
  for (const task of tasks) {
29
- for (const entry of task.files.map(normalize).filter((path) => path.startsWith("src/") && !/[*?{[]/.test(path))) {
30
- const source = basename(entry, extname(entry));
31
- if (stem === source || stem.startsWith(`${source}-`)) {
32
- matches.set(`${task.id}:${entry}`, { taskId: task.id, source: entry });
42
+ for (const entry of task.files.map(normalize).filter((path) => path.startsWith("src/"))) {
43
+ const fromGlob = /[*?{[]/.test(entry);
44
+ const entries = fromGlob ? allSources.filter(filesGlob(entry)) : [entry];
45
+ for (const sourcePath of entries) {
46
+ const source = basename(sourcePath, extname(sourcePath));
47
+ if (stem === source || stem.startsWith(`${source}-`)) {
48
+ matches.set(`${task.id}:${sourcePath}`, { taskId: task.id, source: sourcePath, fromGlob });
49
+ }
33
50
  }
34
51
  }
35
52
  }
@@ -148,19 +165,27 @@ export function ownershipFindings(tasks, repoRoot) {
148
165
  predictedBy.set(hit, ids);
149
166
  }
150
167
  }
168
+ const globOwnedTests = new Map();
169
+ const allSources = sourceFiles(repoRoot);
151
170
  for (const source of sources) {
152
171
  const ids = predictedBy.get(source.path) ?? new Set();
153
- for (const { taskId } of namedSources(source.path, tasks))
154
- ids.add(taskId);
172
+ for (const named of namedSources(source.path, tasks, allSources)) {
173
+ ids.add(named.taskId);
174
+ if (named.fromGlob) {
175
+ const owners = globOwnedTests.get(source.path) ?? new Set();
176
+ owners.add(named.taskId);
177
+ globOwnedTests.set(source.path, owners);
178
+ }
179
+ }
155
180
  if (ids.size > 0)
156
181
  predictedBy.set(source.path, ids);
157
182
  }
158
183
  const findings = [];
159
184
  for (const [test, taskIds] of predictedBy) {
160
- if (owners(test).length === 0) {
185
+ if (owners(test).length === 0 && !(globOwnedTests.get(test)?.size)) {
161
186
  const ids = [...taskIds].sort();
162
187
  const source = sourceByPath.get(test);
163
- const evidence = corroboration(source, namedSources(test, tasks));
188
+ const evidence = corroboration(source, namedSources(test, tasks, allSources));
164
189
  findings.push({
165
190
  code: "unowned-test",
166
191
  test,
@@ -266,6 +266,7 @@ export declare const TickmarkrConfigSchema: z.ZodObject<{
266
266
  }, z.core.$strip>;
267
267
  review: z.ZodObject<{
268
268
  complexityThreshold: z.ZodNumber;
269
+ timeoutMs: z.ZodNumber;
269
270
  required: z.ZodBoolean;
270
271
  prefer: z.ZodOptional<z.ZodArray<z.ZodString>>;
271
272
  policy: z.ZodOptional<z.ZodEnum<{
@@ -355,6 +355,7 @@ export const TickmarkrConfigSchema = z.object({
355
355
  // because routing hints and the setup cockpit still read it as a complexity landmark; NOTHING in
356
356
  // src/gates/review.ts reads it any more, and a value here can no longer disable a review.
357
357
  complexityThreshold: z.number(),
358
+ timeoutMs: z.number().int().positive(),
358
359
  required: z.boolean(),
359
360
  prefer: z.array(z.string()).optional(),
360
361
  // R3: the operator's participation floor. "full" forces a cross-vendor review on every task;
@@ -484,8 +485,15 @@ export const DEFAULT_CONFIG = {
484
485
  // 2026-07-16 (cursor-agent 2026.07.09); gives cursor a cheap-tier channel so low-complexity shapes
485
486
  // stop burning its mid. Operator-approved 2026-07-16.
486
487
  "composer-2.5-fast": "cheap",
488
+ // OBS-871: LiveBench 2026_06_25, fetched 2026-09-03; released 2026-09-01.
489
+ "claude-fable-5-1": "frontier",
490
+ // OBS-871: LiveBench 2026_06_25, fetched 2026-09-03; released 2026-09-02.
491
+ "gemini-3.8-flash": "mid",
492
+ },
493
+ windows: {
494
+ "composer-2.5": 200_000, "composer-2.5-fast": 200_000,
495
+ "claude-fable-5-1": 1_000_000, "gemini-3.8-flash": 1_000_000,
487
496
  },
488
- windows: { "composer-2.5": 200_000, "composer-2.5-fast": 200_000 },
489
497
  },
490
498
  // GLM-5.2 → mid per benchmark policy (2026-07): SWE-bench Pro 62.1 (> GPT-5.5 58.6), FrontierSWE 74.4 ≈ Opus 4.8;
491
499
  // no independent Terminal-Bench score → conservative mid, overlays may raise.
@@ -507,8 +515,46 @@ export const DEFAULT_CONFIG = {
507
515
  // cross-harness) is pre-existing, FLEET-05 owns it.
508
516
  pi: {
509
517
  vendor: "zhipu", channel: "sub",
510
- models: { "zai/glm-5.2": "mid" },
511
- windows: { "zai/glm-5.2": 1_000_000 },
518
+ models: {
519
+ "zai/glm-5.2": "mid",
520
+ // OBS-871: LiveBench 2026_06_25, fetched 2026-09-03; GLM-5.3 released 2026-08-18.
521
+ "zai/glm-5.3": "frontier",
522
+ // OBS-871: same LiveBench source and 2026-09-03 fetch; speed variant remains mid.
523
+ "zai/glm-5.3-flash": "mid",
524
+ },
525
+ windows: { "zai/glm-5.2": 1_000_000, "zai/glm-5.3": 1_000_000, "zai/glm-5.3-flash": 200_000 },
526
+ },
527
+ // OBS-871: gateway ids and bands from LiveBench 2026_06_25, fetched 2026-09-03.
528
+ omp: {
529
+ vendor: "mixed", channel: "sub",
530
+ models: {
531
+ "google/gemini-3.8-flash": "mid",
532
+ "zai/glm-5.3": "frontier",
533
+ "alibaba/qwen3.8-max": "frontier",
534
+ },
535
+ modelOverrides: {
536
+ "google/gemini-3.8-flash": { vendor: "google" },
537
+ "zai/glm-5.3": { vendor: "zhipu" },
538
+ "alibaba/qwen3.8-max": { vendor: "alibaba" },
539
+ },
540
+ windows: {
541
+ "google/gemini-3.8-flash": 1_000_000,
542
+ "zai/glm-5.3": 1_000_000,
543
+ "alibaba/qwen3.8-max": 1_000_000,
544
+ },
545
+ },
546
+ // OBS-871: Qwen 3.8 Max scored 64.6 in LiveBench 2026_06_25, fetched 2026-09-03.
547
+ qwen: {
548
+ vendor: "alibaba", channel: "sub",
549
+ models: { "qwen3.8-max": "frontier" },
550
+ windows: { "qwen3.8-max": 1_000_000 },
551
+ },
552
+ // OBS-871: prime-agent listed this prime-inference route on 2026-09-03; GLM-5.2 remains mid.
553
+ "prime-agent": {
554
+ vendor: "mixed", channel: "sub",
555
+ models: { "prime-inference/z-ai/glm-5.2": "mid" },
556
+ modelOverrides: { "prime-inference/z-ai/glm-5.2": { vendor: "zhipu" } },
557
+ windows: { "prime-inference/z-ai/glm-5.2": 1_000_000 },
512
558
  },
513
559
  // Native grok CLI (Phase 40). grok-4.5 → mid: same benchmark provenance as the retired cursor-agent
514
560
  // grok-4.5-xhigh seed above (AA 54, TB2.1 83.3%, SWE-b Pro 64.7% — researched 2026-07-10).
@@ -543,7 +589,7 @@ export const DEFAULT_CONFIG = {
543
589
  judge: { adapter: "claude-code", model: "fable" },
544
590
  // R3: no `policy` floor — the neutral floor leaves the compiler's per-task assignment standing, so
545
591
  // the path-keyed rule is reachable out of the box rather than raised to full by construction.
546
- review: { complexityThreshold: 7, required: true, criticalPaths: [...DEFAULT_REVIEW_CRITICAL_PATHS] },
592
+ review: { complexityThreshold: 7, timeoutMs: 900_000, required: true, criticalPaths: [...DEFAULT_REVIEW_CRITICAL_PATHS] },
547
593
  consult: { adapter: "claude-code", model: "fable", stallMinutes: 15 },
548
594
  // v1.4: gate LLM calls (judge/review/consult) run headless by default; pane opts back into visible agents.
549
595
  // v1.2: workers are the real agent TUI in the pane; "print" restores the -p-rendered-in-pane path.
@@ -765,7 +811,7 @@ export function configTemplate(overlay) {
765
811
  # test: npm test
766
812
  # byShape:
767
813
  # docs: { acceptance: false, review: false } # baseline, evidence, and scope are mandatory
768
- # review: { complexityThreshold: 7, required: true, prefer: [codex:gpt-5.6-sol, kimi] }
814
+ # review: { complexityThreshold: 7, timeoutMs: 900000, required: true, prefer: [codex:gpt-5.6-sol, kimi] }
769
815
  # # prefer: ordered reviewer seat preference (adapter | adapter:model); ranks
770
816
  # # diversity-eligible channels only — never admits a same-vendor/same-model reviewer
771
817
  # consult: { adapter: claude-code, model: fable, stallMinutes: 15, prefer: [codex:gpt-5.6-sol, kimi:kimi-code/k3] }
@@ -56,6 +56,7 @@ export declare class HerdrDriver implements ExecutorDriver {
56
56
  private journal?;
57
57
  id: string;
58
58
  interactive: boolean;
59
+ readSource: string;
59
60
  private groups;
60
61
  private groupSerial;
61
62
  private deliverySerial;
@@ -142,6 +142,7 @@ export class HerdrDriver {
142
142
  journal;
143
143
  id = "herdr";
144
144
  interactive = true;
145
+ readSource = "recent-unwrapped";
145
146
  groups = new Map();
146
147
  // grouped slot()/close() mutate shared group state across awaits — serialize them so two
147
148
  // concurrent first members can never both create the group tab (mergeSerial idiom, daemon.ts)
@@ -1339,7 +1340,16 @@ export class HerdrDriver {
1339
1340
  if (alive.has(tab))
1340
1341
  continue;
1341
1342
  const closed = await this.herdr(`tab close ${shq(tab)}`);
1342
- if (closed.code !== 0) {
1343
+ if (closed.code !== 0 && /tab_not_found/i.test(`${closed.stderr}\n${closed.stdout}`)) {
1344
+ this.journalReconcile(journalHandle, "tab-reconcile-close-skipped", undefined, {
1345
+ tabId: tab,
1346
+ runId,
1347
+ sweeperRunId: runId,
1348
+ exitCode: 0,
1349
+ reason: "tab_not_found",
1350
+ });
1351
+ }
1352
+ else if (closed.code !== 0) {
1343
1353
  this.journalReconcile(journalHandle, "tab-reconcile-close-failed", undefined, {
1344
1354
  tabId: tab,
1345
1355
  runId,
@@ -2,7 +2,7 @@ import { type ShResult } from "../run/git.js";
2
2
  import { type JournalEvent } from "../run/journal.js";
3
3
  import { type ExecutorDriver, type NotifyOpts, type Slot, type SlotOpts } from "./types.js";
4
4
  /** The response families the ONE shared envelope parser serves. There is no second JSON seam. */
5
- export declare const ORCA_RESPONSE_FAMILIES: readonly ["status", "create", "list", "read", "send", "wait", "show", "close", "worktree-current", "hooks-status"];
5
+ export declare const ORCA_RESPONSE_FAMILIES: readonly ["status", "create", "list", "read", "send", "wait", "show", "close", "worktree-current", "worktree-set", "hooks-status"];
6
6
  export type OrcaFamily = (typeof ORCA_RESPONSE_FAMILIES)[number];
7
7
  export declare const ORCA_FIXTURE_VERSION = "1.4.195";
8
8
  export declare const ORCA_CLI_COMMAND_ENV = "ORCA_CLI_COMMAND";
@@ -118,6 +118,8 @@ export declare class OrcaDriver implements ExecutorDriver {
118
118
  private journalRoots;
119
119
  private narrate?;
120
120
  private hookCoverage?;
121
+ private taskWorktrees;
122
+ private pendingProjects;
121
123
  constructor(opts?: OrcaDriverOpts);
122
124
  private call;
123
125
  /** The live runtime's identity, or an explicit failure. A missing or unreachable runtime is a
@@ -131,7 +133,13 @@ export declare class OrcaDriver implements ExecutorDriver {
131
133
  private state;
132
134
  private latched;
133
135
  private assertAvailable;
136
+ describe(slot: Slot): {
137
+ surface?: string;
138
+ hostPlatform?: string;
139
+ };
134
140
  run(slot: Slot, cmd: string): Promise<void>;
141
+ private sendText;
142
+ private sendReceipt;
135
143
  private create;
136
144
  /**
137
145
  * A freshly-created git checkout does not become a valid Orca selector atomically. Ask
@@ -188,6 +196,11 @@ export declare class OrcaDriver implements ExecutorDriver {
188
196
  * only after this method validates all four fields. Any malformed/refused wait remains explicit. */
189
197
  private waitCondition;
190
198
  waitAgentStatus(slot: Slot, status: string, timeoutMs: number): Promise<boolean>;
199
+ sendKey(slot: Slot, key: string): Promise<void>;
200
+ nudge(slot: Slot, message: string): Promise<boolean>;
201
+ narrator(cwd: string, command: string, runId?: string): Promise<Slot>;
202
+ project(taskId: string, state: "in-progress" | "in-review" | "completed"): Promise<void>;
203
+ private setWorkspaceStatus;
191
204
  notify(msg: string, opts?: NotifyOpts): Promise<void>;
192
205
  narrateWith(narrate: (event: JournalEvent) => void): void;
193
206
  close(slot: Slot): Promise<void>;
@@ -34,7 +34,7 @@ import { formatOwnedName, panesToClose, parseOwnedName } from "./types.js";
34
34
  /** The response families the ONE shared envelope parser serves. There is no second JSON seam. */
35
35
  export const ORCA_RESPONSE_FAMILIES = [
36
36
  "status", "create", "list", "read", "send", "wait", "show", "close",
37
- "worktree-current", "hooks-status",
37
+ "worktree-current", "worktree-set", "hooks-status",
38
38
  ];
39
39
  export const ORCA_FIXTURE_VERSION = "1.4.195";
40
40
  export const ORCA_CLI_COMMAND_ENV = "ORCA_CLI_COMMAND";
@@ -57,6 +57,7 @@ const POLL_MS = 200;
57
57
  export const WORKTREE_ADOPTION_TIMEOUT_MS = 60_000;
58
58
  const WORKTREE_ADOPTION_POLL_MS = 1_000;
59
59
  const WORKTREE_ADOPTION_JOURNAL_MS = 2_000;
60
+ const NUDGE_ECHO_TIMEOUT_MS = 2_000;
60
61
  const SYSTEM_TIME = {
61
62
  now: () => Date.now(),
62
63
  sleep: (ms) => new Promise((resolve) => setTimeout(resolve, ms)),
@@ -262,6 +263,8 @@ export class OrcaDriver {
262
263
  journalRoots = new Map();
263
264
  narrate;
264
265
  hookCoverage;
266
+ taskWorktrees = new Map();
267
+ pendingProjects = new Map();
265
268
  constructor(opts = {}) {
266
269
  this.bin = opts.bin ?? resolveOrcaCliBinary(process.cwd(), { env: opts.env, platform: opts.platform }) ?? "orca";
267
270
  // Config values flow into a shell here: every argv element is quoted, always.
@@ -358,6 +361,15 @@ export class OrcaDriver {
358
361
  // receipt, relist, reconcile — is against THIS value.
359
362
  const worktree = canonicalWorktreePath(cwd);
360
363
  this.slots.set(id, { title, cwd: worktree, agent: opts?.agent, buf: "", recoveries: 0, recovering: false });
364
+ const owned = parseOwnedName(title);
365
+ if (owned?.role === "worker") {
366
+ this.taskWorktrees.set(owned.taskId, worktree);
367
+ const pending = this.pendingProjects.get(owned.taskId);
368
+ if (pending) {
369
+ await this.setWorkspaceStatus(worktree, pending);
370
+ this.pendingProjects.delete(owned.taskId);
371
+ }
372
+ }
361
373
  return { id, name: title, cwd: worktree, group: opts?.group };
362
374
  }
363
375
  /** Where to invoke the CLI for this slot's calls (see OrcaSlotState.dir). */
@@ -378,6 +390,17 @@ export class OrcaDriver {
378
390
  if (st.unavailable)
379
391
  throw new OrcaUnavailableError(family, st.unavailable, "");
380
392
  }
393
+ describe(slot) {
394
+ const st = this.state(slot);
395
+ // The daemon asks only after run(), but callers probing a lazy slot must see absence rather
396
+ // than an invented placement. The interface's object spread accepts this runtime absence.
397
+ if (!st.handle)
398
+ return undefined;
399
+ return {
400
+ ...(st.surface === undefined ? {} : { surface: st.surface }),
401
+ ...(st.hostPlatform === undefined ? {} : { hostPlatform: st.hostPlatform }),
402
+ };
403
+ }
381
404
  async run(slot, cmd) {
382
405
  const st = this.state(slot);
383
406
  if (!st.handle) {
@@ -387,22 +410,28 @@ export class OrcaDriver {
387
410
  this.assertAvailable("send", st);
388
411
  // Later deliveries go into the terminal this slot already owns — never a second create.
389
412
  // terminalOp proves the runtime binding before the handle goes on the wire.
390
- const env = await this.terminalOp("send", st, (h) => this.call("send", ["terminal", "send", "--terminal", h, "--text", cmd, "--enter"], this.cliCwd(st)), { mutating: true });
391
- // Recorded 1.4.186 send receipt: result.send = {handle, accepted, bytesWritten} — there is no
392
- // `delivered`. `ok:true` alone is not a receipt: only an accepted receipt naming THIS handle
393
- // proves the text was submitted, so anything else is a failed send, never a silent no-op.
413
+ await this.sendText(st, cmd);
414
+ }
415
+ async sendText(st, text) {
416
+ const env = await this.terminalOp("send", st, (h) => this.call("send", ["terminal", "send", "--terminal", h, "--text", text, "--enter"], this.cliCwd(st)), { mutating: true });
417
+ // Recorded 1.4.195 send receipt: result.send = {handle, accepted, bytesWritten}. `ok:true`
418
+ // alone is not delivery: the accepted receipt must name this handle and account for the bytes.
419
+ const receipt = this.sendReceipt(env, st);
420
+ const expectedBytes = Buffer.byteLength(text, "utf8") + 1;
421
+ if (receipt.accepted !== true || typeof receipt.bytesWritten !== "number" || receipt.bytesWritten !== expectedBytes) {
422
+ throw new OrcaError("send", `send receipt does not report expected byte delivery (accepted: ${JSON.stringify(receipt.accepted)}, bytesWritten: ${receipt.bytesWritten}, expected: ${expectedBytes})`, env.raw);
423
+ }
424
+ }
425
+ sendReceipt(env, st) {
394
426
  const receipt = env.result.send;
395
427
  if (typeof receipt !== "object" || receipt === null || Array.isArray(receipt)) {
396
428
  throw new OrcaError("send", "send response carries no send receipt", env.raw);
397
429
  }
398
- const s = receipt;
399
- if (str(s.handle) !== st.handle) {
400
- throw new OrcaError("send", `send receipt names terminal ${str(s.handle) ?? "none"}, not the addressed ${st.handle}`, env.raw);
401
- }
402
- const expectedBytes = Buffer.byteLength(cmd, "utf8") + 1; // 1 for the --enter newline
403
- if (s.accepted !== true || typeof s.bytesWritten !== "number" || s.bytesWritten !== expectedBytes) {
404
- throw new OrcaError("send", `send receipt does not report expected byte delivery (accepted: ${JSON.stringify(s.accepted)}, bytesWritten: ${s.bytesWritten}, expected: ${expectedBytes})`, env.raw);
430
+ const parsed = receipt;
431
+ if (str(parsed.handle) !== st.handle) {
432
+ throw new OrcaError("send", `send receipt names terminal ${str(parsed.handle) ?? "none"}, not the addressed ${st.handle}`, env.raw);
405
433
  }
434
+ return parsed;
406
435
  }
407
436
  async create(st, cmd) {
408
437
  await this.probeRuntime(this.cliCwd(st));
@@ -435,9 +464,10 @@ export class OrcaDriver {
435
464
  st.handle = handle;
436
465
  // The handle is bound to the runtime identity that ANSWERED its create.
437
466
  st.runtimeId = env.runtimeId;
438
- const surface = str(term.surface);
439
- if (surface !== undefined && surface !== "visible") {
440
- await this.notify(`tickmarkr orca terminal created on ${surface} surface`, { tier: "attention" });
467
+ st.surface = str(term.surface);
468
+ st.hostPlatform = str(term.hostPlatform);
469
+ if (st.surface !== undefined && st.surface !== "visible") {
470
+ await this.notify(`tickmarkr orca terminal created on ${st.surface} surface`, { tier: "attention" });
441
471
  }
442
472
  }
443
473
  /**
@@ -920,6 +950,75 @@ export class OrcaDriver {
920
950
  await this.time.sleep(Math.min(this.pollMs, left2));
921
951
  }
922
952
  }
953
+ async sendKey(slot, key) {
954
+ const st = this.state(slot);
955
+ this.assertAvailable("send", st);
956
+ if (key === "enter") {
957
+ await this.sendText(st, "");
958
+ return;
959
+ }
960
+ if (key === "ctrl+c") {
961
+ const env = await this.terminalOp("send", st, (h) => this.call("send", ["terminal", "send", "--terminal", h, "--interrupt"], this.cliCwd(st)), { mutating: true });
962
+ // UNRECORDED SHAPE: Orca 1.4.195 was not captured for `send --interrupt`. Until a live
963
+ // receipt exists, validate only the shared envelope and its handle binding; do not invent
964
+ // accepted/bytesWritten semantics for an interrupt.
965
+ this.sendReceipt(env, st);
966
+ return;
967
+ }
968
+ throw new OrcaError("send", `Orca has no terminal key verb for ${JSON.stringify(key)}`, "");
969
+ }
970
+ async nudge(slot, message) {
971
+ if (!message)
972
+ return false;
973
+ try {
974
+ const st = this.state(slot);
975
+ if (!await this.waitCondition(st, "tui-idle", 1))
976
+ return false;
977
+ // Drain to the current stream cursor before sending. A message already present makes the
978
+ // proof ambiguous, so decline rather than reporting delivery from old scrollback.
979
+ const before = await this.sweep(st);
980
+ if (before.includes(message) || joinWrapped(before).includes(message))
981
+ return false;
982
+ await this.sendText(st, message);
983
+ const deadline = this.time.now() + NUDGE_ECHO_TIMEOUT_MS;
984
+ for (;;) {
985
+ const after = await this.sweep(st);
986
+ if (after.includes(message) || joinWrapped(after).includes(message))
987
+ return true;
988
+ const left = deadline - this.time.now();
989
+ if (left <= 0)
990
+ return false;
991
+ await this.time.sleep(Math.min(this.pollMs, left));
992
+ }
993
+ }
994
+ catch {
995
+ return false;
996
+ }
997
+ }
998
+ async narrator(cwd, command, runId) {
999
+ if (!runId)
1000
+ throw new OrcaError("create", "Orca narrator requires a run identity", "");
1001
+ const slot = await this.slot(cwd, formatOwnedName({ role: "watch", taskId: "run", attempt: 0, runId }), { owned: { role: "watch", taskId: "run", attempt: 0, runId } });
1002
+ await this.run(slot, command);
1003
+ return slot;
1004
+ }
1005
+ async project(taskId, state) {
1006
+ const worktree = this.taskWorktrees.get(taskId);
1007
+ if (!worktree) {
1008
+ // The daemon projects in-progress immediately before it creates the task checkout/slot.
1009
+ // Hold only that latest state; slot() applies it once the task's own path is known.
1010
+ this.pendingProjects.set(taskId, state);
1011
+ return;
1012
+ }
1013
+ await this.setWorkspaceStatus(worktree, state);
1014
+ }
1015
+ async setWorkspaceStatus(worktree, state) {
1016
+ // UNRECORDED SHAPE: no Orca 1.4.195 `worktree set` receipt was captured. The shared envelope
1017
+ // parser is the complete success proof here; no result payload is assumed or fabricated.
1018
+ await this.call("worktree-set", [
1019
+ "worktree", "set", "--worktree", `path:${worktree}`, "--workspace-status", state,
1020
+ ], worktree);
1021
+ }
923
1022
  async notify(msg, opts) {
924
1023
  if (opts?.tier === "routine")
925
1024
  return;