tickmarkr 2.3.0 → 2.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. package/dist/adapters/catalog-remote.d.ts +12 -4
  2. package/dist/adapters/catalog-remote.js +97 -45
  3. package/dist/adapters/catalog.js +5 -3
  4. package/dist/adapters/claude-code.d.ts +1 -1
  5. package/dist/adapters/claude-code.js +8 -5
  6. package/dist/adapters/codex.js +6 -7
  7. package/dist/adapters/model-lints.d.ts +9 -5
  8. package/dist/adapters/model-lints.js +56 -15
  9. package/dist/adapters/model-windows.js +11 -0
  10. package/dist/adapters/prompt.js +1 -0
  11. package/dist/adapters/qwen.d.ts +5 -0
  12. package/dist/adapters/qwen.js +153 -0
  13. package/dist/adapters/registry.js +13 -1
  14. package/dist/adapters/types.d.ts +1 -0
  15. package/dist/adapters/types.js +1 -0
  16. package/dist/cli/commands/compile.d.ts +3 -0
  17. package/dist/cli/commands/compile.js +91 -34
  18. package/dist/cli/commands/doctor.d.ts +4 -3
  19. package/dist/cli/commands/doctor.js +28 -9
  20. package/dist/cli/commands/fleet.d.ts +4 -0
  21. package/dist/cli/commands/fleet.js +60 -15
  22. package/dist/cli/commands/init.js +12 -13
  23. package/dist/cli/commands/plan.js +75 -11
  24. package/dist/cli/commands/run.js +20 -1
  25. package/dist/cli/commands/status.d.ts +1 -0
  26. package/dist/cli/commands/status.js +59 -16
  27. package/dist/cli/commands/verify.d.ts +1 -0
  28. package/dist/cli/commands/verify.js +5 -0
  29. package/dist/cli/commands/version.js +2 -2
  30. package/dist/cli/index.d.ts +1 -1
  31. package/dist/cli/index.js +2 -2
  32. package/dist/compile/collateral.d.ts +14 -12
  33. package/dist/compile/collateral.js +32 -33
  34. package/dist/compile/index.d.ts +4 -1
  35. package/dist/compile/index.js +53 -8
  36. package/dist/compile/native.d.ts +4 -2
  37. package/dist/compile/native.js +63 -6
  38. package/dist/compile/ownership.js +41 -10
  39. package/dist/config/config.d.ts +1 -0
  40. package/dist/config/config.js +51 -5
  41. package/dist/drivers/herdr.d.ts +2 -0
  42. package/dist/drivers/herdr.js +43 -4
  43. package/dist/drivers/orca.d.ts +18 -1
  44. package/dist/drivers/orca.js +163 -15
  45. package/dist/drivers/types.d.ts +10 -0
  46. package/dist/gates/baseline.d.ts +26 -2
  47. package/dist/gates/baseline.js +115 -13
  48. package/dist/gates/review.d.ts +6 -4
  49. package/dist/gates/review.js +26 -31
  50. package/dist/gates/run-gates.d.ts +5 -2
  51. package/dist/gates/run-gates.js +34 -17
  52. package/dist/graph/graph.d.ts +20 -0
  53. package/dist/graph/graph.js +66 -1
  54. package/dist/route/preference.d.ts +6 -0
  55. package/dist/route/preference.js +40 -0
  56. package/dist/route/router.js +15 -2
  57. package/dist/run/consult.d.ts +1 -0
  58. package/dist/run/consult.js +35 -7
  59. package/dist/run/daemon.d.ts +9 -0
  60. package/dist/run/daemon.js +266 -59
  61. package/dist/run/git.d.ts +4 -0
  62. package/dist/run/git.js +51 -6
  63. package/dist/run/journal.d.ts +1 -1
  64. package/dist/run/journal.js +5 -2
  65. package/dist/run/lock.d.ts +6 -0
  66. package/dist/run/lock.js +41 -1
  67. package/dist/tui/ink/fleet-app.d.ts +4 -0
  68. package/dist/tui/ink/fleet-app.js +45 -16
  69. package/package.json +59 -1
  70. package/skills/tickmarkr-overseer/SKILL.md +39 -4
  71. package/skills/tickmarkr-overseer/scripts/grade-ci.sh +79 -0
  72. package/skills/tickmarkr-overseer/scripts/watch-context.sh +90 -4
@@ -460,23 +460,18 @@ export function newDirectoryLints(tasks, repoRoot) {
460
460
  }
461
461
  return lines;
462
462
  }
463
- /**
464
- * OBS-76 class: sweep src/ for out-of-scope source files that reference a symbol the acceptance
465
- * criteria name — the v1.52 router.ts omission, named at plan time instead of one judge round in.
466
- * Advisory plan output only, same contract as collateralLints.
467
- */
468
- export function sourceScopeLints(tasks, repoRoot) {
469
- const newDirLints = newDirectoryLints(tasks, repoRoot);
463
+ /** Structured OBS-76 findings shared by human rendering and native compile enforcement. */
464
+ export function sourceScopeFindings(tasks, repoRoot) {
470
465
  const perTask = tasks
471
466
  .map((t) => ({ t, needles: criteriaSymbols(t.acceptance) }))
472
467
  .filter((x) => x.needles.length);
473
468
  if (!perTask.length)
474
- return newDirLints;
469
+ return [];
475
470
  const srcFiles = walkCode(repoRoot, "src");
476
471
  if (!srcFiles.length)
477
- return newDirLints;
472
+ return [];
478
473
  const read = makeReader(repoRoot);
479
- const lines = [];
474
+ const findings = [];
480
475
  for (const { t, needles } of perTask) {
481
476
  // OBS-22: scopeGate accepts picomatch globs; advisory warnings must agree.
482
477
  const scoped = filesGlob(t.files.map((f) => f.replace(/^\.\//, "")));
@@ -495,11 +490,23 @@ export function sourceScopeLints(tasks, repoRoot) {
495
490
  if (!hits.length)
496
491
  continue;
497
492
  hits.sort();
498
- const listed = hits.slice(0, MAX_HITS_PER_TASK).join(", ");
499
- const tail = hits.length > MAX_HITS_PER_TASK ? " (capped)" : "";
500
- lines.push(`${t.id}: criteria implicate out-of-scope source not in files[]: ${listed}${tail}`);
493
+ findings.push({ taskId: t.id, paths: hits });
501
494
  }
502
- return [...newDirLints, ...lines];
495
+ return findings;
496
+ }
497
+ /** Human display is derived from the structured finding; no enforcement path parses this text. */
498
+ export function renderSourceScopeFinding(finding) {
499
+ const listed = finding.paths.slice(0, MAX_HITS_PER_TASK).join(", ");
500
+ const tail = finding.paths.length > MAX_HITS_PER_TASK ? " (capped)" : "";
501
+ return `${finding.taskId}: criteria implicate out-of-scope source not in files[]: ${listed}${tail}`;
502
+ }
503
+ /**
504
+ * OBS-76 class: sweep src/ for out-of-scope source files that reference a symbol the acceptance
505
+ * criteria name — the v1.52 router.ts omission, named at plan time instead of one judge round in.
506
+ * Advisory plan output only, same contract as collateralLints.
507
+ */
508
+ export function sourceScopeLints(tasks, repoRoot, findings = sourceScopeFindings(tasks, repoRoot)) {
509
+ return [...newDirectoryLints(tasks, repoRoot), ...findings.map(renderSourceScopeFinding)];
503
510
  }
504
511
  // ── Task Unit Contract (OBS-212 / OBS-214) ────────────────────────────────────────────────────
505
512
  // These are ERRORS, not advisories. Everything above this line is a plan-time warning the author
@@ -835,9 +842,9 @@ export function symbolOwnershipErrors(tasks, repoRoot) {
835
842
  * cannot be a warning, because nothing downstream would stop it — the gate would honour the policy.
836
843
  */
837
844
  export function reviewParticipationErrors(tasks, review) {
838
- // The shipped critical paths are a FLOOR, not a default that a config replaces: this lint resolves
839
- // its config from a repo root the compile seam cannot always name (see taskUnitContractErrors), and
840
- // the one direction that must never happen is a wrong root lowering enforcement below what tickmarkr
845
+ // The shipped critical paths are a FLOOR, not a default that a config replaces: a programmatic
846
+ // compile may rely on taskUnitContractErrors' process.cwd() default, and the one direction that
847
+ // must never happen is a mismatched invocation root lowering enforcement below what tickmarkr
841
848
  // ships. Union is monotone — a repo-local list can only add.
842
849
  const critical = [...new Set([...DEFAULT_REVIEW_CRITICAL_PATHS, ...(review.criticalPaths ?? [])])];
843
850
  const errors = [];
@@ -871,11 +878,10 @@ export function reviewParticipationErrors(tasks, review) {
871
878
  return errors;
872
879
  }
873
880
  /**
874
- * The ACTIVE participation config for a compile rooted at `repoRoot`. Same default-argument contract
875
- * as `repoRoot` itself: the CLI and daemon compile from inside the target repo, so the repo's own
876
- * config (plus the global overlay) is the config the run will gate under. A config the loader cannot
877
- * read degrades to the shipped defaults rather than crashing the compile — every command that reaches
878
- * this seam loads the same config itself and reports a malformed one loudly.
881
+ * The ACTIVE participation config for a compile rooted at `repoRoot`. CLI compilation supplies its
882
+ * target root; a programmatic caller that omits it uses taskUnitContractErrors' process.cwd() default,
883
+ * so its active overlay is the invocation repository. Native CLI compiles independently read the
884
+ * overlay fail-closed at the finalization seam, so malformed YAML is loud.
879
885
  */
880
886
  function activeReviewParticipation(repoRoot) {
881
887
  try {
@@ -888,23 +894,16 @@ function activeReviewParticipation(repoRoot) {
888
894
  }
889
895
  /**
890
896
  * Every Task Unit Contract violation in one pass, ready to throw. `repoRoot` backs the
891
- * symbol-ownership lint and the participation config, and defaults to the invocation directory —
892
- * correct for the CLI/daemon, which compile from inside the target repo.
897
+ * symbol-ownership lint and participation overlay, and defaults to the invocation directory for
898
+ * programmatic compiles. The CLI passes its target repository explicitly.
893
899
  *
894
900
  * The read declaration (`context:`) rides on this parameter, on the task objects themselves — the
895
901
  * aggregator never infers it from the repo, the config, or a sibling task. It is OPTIONAL: a caller
896
902
  * whose task objects carry no `context` gets byte-identical verdicts to the ones it got before the
897
903
  * field existed, because an absent declaration grants no authority at all.
898
904
  *
899
- * KNOWN GAP (R2 review, T6/T7 scope): the compile seam in src/compile/index.ts
900
- * (`enforceTaskUnitContract`) does not thread `compileSource`'s `root` argument into this call, so a
901
- * PROGRAMMATIC compile whose target repo differs from process.cwd() resolves the wrong root and can
902
- * miss that repo's own `review.criticalPaths`. Threading it is one line in src/compile/index.ts,
903
- * outside this task's file scope. Two things bound the exposure meanwhile, both live: the critical
904
- * set here is the UNION with the shipped defaults, so a wrong root can never lower enforcement below
905
- * the shipped floor; and the review GATE — which is handed the run's real config — refuses to skip a
906
- * critical path itself (src/gates/review.ts), so a lint this seam misses costs a late verdict rather
907
- * than an unreviewed one. The compile lint is the early warning; the gate is the fail-closed backstop.
905
+ * The compile seam in src/compile/index.ts threads `compileSource`'s repo root here when supplied;
906
+ * otherwise this function deliberately checks process.cwd().
908
907
  */
909
908
  export function taskUnitContractErrors(tasks, repoRoot = process.cwd(), review = activeReviewParticipation(repoRoot)) {
910
909
  return [
@@ -10,6 +10,9 @@ export type PlanIR = {
10
10
  tasks: RunGraph["tasks"];
11
11
  };
12
12
  export type PlanFinalizationHook = (plan: PlanIR) => PlanIR;
13
+ export type CompileOptions = {
14
+ strict?: boolean;
15
+ };
13
16
  export declare function finalizePlan(plan: PlanIR, src: string, repoRoot?: string): RunGraph;
14
- export declare function compileSource(src: string, type?: SourceType, root?: string, beforeFinalize?: PlanFinalizationHook): RunGraph;
17
+ export declare function compileSource(src: string, type?: SourceType, root?: string, beforeFinalize?: PlanFinalizationHook, options?: CompileOptions): RunGraph;
15
18
  export {};
@@ -1,5 +1,6 @@
1
1
  import { existsSync, readFileSync, statSync } from "node:fs";
2
2
  import { join } from "node:path";
3
+ import { parse } from "yaml";
3
4
  import { validateGraph } from "../graph/schema.js";
4
5
  import { taskUnitContractErrors } from "./collateral.js";
5
6
  import { blocksCompile, ownershipFindings, renderOwnershipFinding } from "./ownership.js";
@@ -33,8 +34,48 @@ function detect(src) {
33
34
  // tasks are too large to converge. A violation is a compile error, never a warning — the failures it
34
35
  // prevents (silently dropped commits, a 28-dispatch task) are invisible until they have already cost
35
36
  // hours, which is exactly the class of thing that has to fail at authoring time.
36
- function enforceTaskUnitContract(g, src) {
37
- const errors = taskUnitContractErrors(g.tasks);
37
+ function repoOverlayMode(repoRoot) {
38
+ if (!repoRoot)
39
+ return undefined;
40
+ const path = join(repoRoot, ".tickmarkr", "config.yaml");
41
+ if (!existsSync(path))
42
+ return undefined;
43
+ try {
44
+ const cfg = parse(readFileSync(path, "utf8"));
45
+ return typeof cfg?.routing?.mode === "string" ? cfg.routing.mode : undefined;
46
+ }
47
+ catch (error) {
48
+ throw new CompileError(`${path} does not parse, so compile cannot prove the repository routing overlay: `
49
+ + `${error instanceof Error ? error.message : String(error)}`);
50
+ }
51
+ }
52
+ function hasModeOverride(src) {
53
+ try {
54
+ for (const line of readFileSync(src, "utf8").split("\n")) {
55
+ if (/^##\s+T\d+:/i.test(line))
56
+ return false;
57
+ if (/^mode-override:\s*true\s*(?:#.*)?$/.test(line))
58
+ return true;
59
+ }
60
+ }
61
+ catch { /* unreadable specs fail elsewhere */ }
62
+ return false;
63
+ }
64
+ function enforceModeOverlay(graph, src, repoRoot) {
65
+ if (graph.spec.source !== "native")
66
+ return;
67
+ // Read every native compile's overlay before checking whether either side declares a mode. A
68
+ // corrupt file cannot masquerade as an absent mode and silently skip the disagreement gate.
69
+ const overlayMode = repoOverlayMode(repoRoot);
70
+ if (graph.mode === undefined)
71
+ return;
72
+ if (overlayMode === undefined || overlayMode === graph.mode || hasModeOverride(src))
73
+ return;
74
+ throw new CompileError(`${src} front-matter mode ${graph.mode} disagrees with repository routing.mode ${overlayMode}; `
75
+ + `write mode-override: true beside the mode line to make the override explicit.`);
76
+ }
77
+ function enforceTaskUnitContract(g, src, repoRoot) {
78
+ const errors = taskUnitContractErrors(g.tasks, repoRoot);
38
79
  if (errors.length > 0) {
39
80
  throw new CompileError(`${src} violates the task unit contract (${errors.length} error${errors.length > 1 ? "s" : ""}):\n`
40
81
  + errors.map((e) => ` - ${e}`).join("\n"));
@@ -42,7 +83,7 @@ function enforceTaskUnitContract(g, src) {
42
83
  return g;
43
84
  }
44
85
  export function finalizePlan(plan, src, repoRoot) {
45
- const graph = enforceTaskUnitContract(validateGraph({
86
+ const graph = validateGraph({
46
87
  version: plan.version,
47
88
  ...(plan.mode !== undefined ? { mode: plan.mode } : {}),
48
89
  spec: {
@@ -52,7 +93,11 @@ export function finalizePlan(plan, src, repoRoot) {
52
93
  ...(plan.base !== undefined ? { base: plan.base } : {}),
53
94
  },
54
95
  tasks: plan.tasks,
55
- }), src);
96
+ });
97
+ // Overlay readability is compile truth, not a mode-disagreement optimization. Check it before
98
+ // other repo-dependent lints so malformed native config cannot be hidden by an earlier finding.
99
+ enforceModeOverlay(graph, src, repoRoot);
100
+ enforceTaskUnitContract(graph, src, repoRoot);
56
101
  // overseer-217 removal condition, now paid: on this milestone's authored graph the conventional
57
102
  // name map emitted 21 raw unowned-test findings; review found 1 real and 20 false, while intersecting
58
103
  // with a direct import or command-entry spawn retained the real one and left 0 false positives. That
@@ -73,11 +118,11 @@ export function finalizePlan(plan, src, repoRoot) {
73
118
  }
74
119
  return graph;
75
120
  }
76
- function compilePlan(src, type, root) {
121
+ function compilePlan(src, type, root, options = {}) {
77
122
  const kind = type ?? detect(src);
78
123
  const graph = kind === "speckit" ? compileSpecKit(src)
79
124
  : kind === "gsd" ? compileGsd(src, root)
80
- : kind === "native" ? compileNative(src)
125
+ : kind === "native" ? compileNative(src, { strict: options.strict })
81
126
  : kind === "prd" ? compilePrd(src)
82
127
  : null;
83
128
  if (!graph) {
@@ -93,6 +138,6 @@ function compilePlan(src, type, root) {
93
138
  tasks: graph.tasks,
94
139
  };
95
140
  }
96
- export function compileSource(src, type, root, beforeFinalize = (plan) => plan) {
97
- return finalizePlan(beforeFinalize(compilePlan(src, type, root)), src, root);
141
+ export function compileSource(src, type, root, beforeFinalize = (plan) => plan, options = {}) {
142
+ return finalizePlan(beforeFinalize(compilePlan(src, type, root, options)), src, root);
98
143
  }
@@ -17,7 +17,7 @@ export declare const COLLECTABLE_TESTS = "tests/**/*.test.ts";
17
17
  export declare const LEGACY_PREFIX: string;
18
18
  export declare const TICKMARKR_NATIVE_MARKER: RegExp;
19
19
  export declare const NATIVE_MARKER: RegExp;
20
- export declare const AUTHORING_LINT_CODES: readonly ["criterion-scope", "dependency-closure", "dependency-coupling", "proxy-metric", "external-referent", "closed-enumeration", "one-behavior", "concern-bundle", "seam-exists"];
20
+ export declare const AUTHORING_LINT_CODES: readonly ["criterion-scope", "dependency-closure", "dependency-coupling", "proxy-metric", "external-referent", "closed-enumeration", "one-behavior", "concern-bundle", "seam-exists", "fence-symbol-absent"];
21
21
  export type AuthoringLintCode = (typeof AUTHORING_LINT_CODES)[number];
22
22
  export interface AuthoringLintFinding {
23
23
  code: AuthoringLintCode;
@@ -31,5 +31,7 @@ export interface AuthoringLintFinding {
31
31
  * compile boundary; the remaining checks are review findings, emitted by compileNative below.
32
32
  */
33
33
  export declare function authoringLintFindings(tasks: readonly Task[], file: string): AuthoringLintFinding[];
34
- export declare function compileNative(file: string): RunGraph;
34
+ export declare function compileNative(file: string, options?: {
35
+ strict?: boolean;
36
+ }): RunGraph;
35
37
  export declare function specTemplate(): string;
@@ -83,6 +83,7 @@ export const AUTHORING_LINT_CODES = [
83
83
  "one-behavior",
84
84
  "concern-bundle",
85
85
  "seam-exists",
86
+ "fence-symbol-absent",
86
87
  ];
87
88
  // The authoring frame's observable oracle is the branch-tip test corpus, never the mutable index or
88
89
  // an author's untracked checkout. Load it once per repository: one bounded git read replaces a grep
@@ -217,6 +218,31 @@ function criterionScopeFinding(task, text, criterion, id, tests) {
217
218
  detail: `criterion names asserting file${missing.length === 1 ? "" : "s"} outside expanded files[] and context[]: ${missing.join(", ")} — ${remedy}`,
218
219
  };
219
220
  }
221
+ // OBS-898: the fence lint's corpus is listed ONCE per compile by git (tracked plus untracked-not-ignored,
222
+ // the same set a checkout walk sees minus the ignored trees) and filtered per task. The walk it replaces
223
+ // descended every dot-directory (.tickmarkr/runs alone held 3 799 files here) once PER TASK, so a
224
+ // 20-task spec compiled in seconds and the committed-spec corpus test timed out. Fail-open like
225
+ // testsAtHead: no git answer means no corpus, and the lint stays quiet rather than wrong.
226
+ function repoFileList(root) {
227
+ const ls = spawnSync("git", ["-C", root, "ls-files", "-z", "--cached", "--others", "--exclude-standard"], { encoding: "utf8", maxBuffer: 1 << 25 });
228
+ if (ls.status !== 0 || typeof ls.stdout !== "string")
229
+ return [];
230
+ return ls.stdout.split("\0").filter(Boolean).sort();
231
+ }
232
+ function repoFiles(files, entries) {
233
+ const scoped = filesGlob(entries.map((entry) => entry.replace(/^\.\//, "")));
234
+ return files.filter(scoped);
235
+ }
236
+ const PRESERVATION_RE = /\b(?:keeps?|preserves?|retains?|does\s+not\s+weaken|do\s+not\s+weaken|not\s+weaken)\b/i;
237
+ const IDENTIFIER_SHAPED_RE = /^[A-Za-z_$][A-Za-z0-9_$]*$/;
238
+ function fencedIdentifiers(text) {
239
+ if (!PRESERVATION_RE.test(text))
240
+ return [];
241
+ return [...new Set([...text.matchAll(/`([^`\n]{1,120})`/g)]
242
+ .map((match) => match[1].trim())
243
+ .filter((token) => IDENTIFIER_SHAPED_RE.test(token) && /[A-Z0-9_$]/.test(token)))]
244
+ .sort();
245
+ }
220
246
  function exportedIdentifier(root, files, identifier) {
221
247
  const escaped = identifier.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
222
248
  const declaration = new RegExp(`\\bexport\\s+(?:(?:declare|default|async)\\s+)*(?:function|class|const|let|var|interface|type)\\s+${escaped}\\b|\\bexport\\s*\\{[^}]*\\b${escaped}\\b`);
@@ -240,14 +266,41 @@ export function authoringLintFindings(tasks, file) {
240
266
  const root = repositoryRoot(file);
241
267
  let indexedTests;
242
268
  const testIndex = () => indexedTests ??= root ? testsAtHead(root) : [];
269
+ let indexedFiles;
270
+ const fileIndex = () => indexedFiles ??= root ? repoFileList(root) : [];
243
271
  const findings = [];
244
272
  for (const task of tasks) {
245
273
  const criteria = criterionTexts(task);
274
+ const taskFiles = root && task.files.length ? repoFiles(fileIndex(), task.files) : [];
275
+ const taskTexts = new Map(taskFiles.map((path) => {
276
+ try {
277
+ return [path, readFileSync(join(root, path), "utf8")];
278
+ }
279
+ catch {
280
+ return [path, ""];
281
+ }
282
+ }));
246
283
  for (const [index, text] of criteria.entries()) {
247
284
  const needsTestIndex = /`[^`\n]{2,120}`|\b\d+\s*(?:\/|of)\s*\d+\b|\b[A-Za-z0-9_.-]+\.test\.ts\b/.test(text);
248
285
  const scope = criterionScopeFinding(task, text, index + 1, id, needsTestIndex ? testIndex() : []);
249
286
  if (scope)
250
287
  findings.push(scope);
288
+ for (const symbol of fencedIdentifiers(text)) {
289
+ // Empty files[] is the deliberately unrestricted scope and a non-repository programmatic
290
+ // compile has no disk corpus to prove against. But a declared files[] that expands to zero
291
+ // repository files is evidence, not absence: its preservation fence cannot be satisfied.
292
+ if (!root || task.files.length === 0)
293
+ continue;
294
+ if ([...taskTexts.values()].some((body) => body.includes(symbol)))
295
+ continue;
296
+ const matchDetail = taskFiles.length === 0
297
+ ? `this task's files[] matched zero files on disk (${task.files.join(", ")})`
298
+ : `that symbol has zero hits in this task's files[] (${taskFiles.join(", ")})`;
299
+ findings.push({
300
+ code: "fence-symbol-absent", fixtureId: id, taskId: task.id, criterion: index + 1,
301
+ detail: `preservation fence cites \`${symbol}\` but ${matchDetail} — OBS-604`,
302
+ });
303
+ }
251
304
  if (/line[- ]count|physical line|not greater than (?:the|\d)|no larger than|at most \d+ (?:lines|bytes)/i.test(text)) {
252
305
  findings.push({ code: "proxy-metric", fixtureId: id, taskId: task.id, criterion: index + 1, detail: "criterion uses a proxy size metric; state the structural intent instead" });
253
306
  }
@@ -308,7 +361,7 @@ function renderAuthoringFinding(finding) {
308
361
  const criterion = finding.criterion === undefined ? "" : ` criterion ${finding.criterion}`;
309
362
  return `tickmarkr: authoring-lint[${finding.code}] fixture ${finding.fixtureId} task ${finding.taskId}${criterion}: ${finding.detail}`;
310
363
  }
311
- export function compileNative(file) {
364
+ export function compileNative(file, options = {}) {
312
365
  if (!existsSync(file))
313
366
  throw new CompileError(`no such native spec file: ${file}`);
314
367
  const content = readFileSync(file, "utf8");
@@ -638,13 +691,17 @@ export function compileNative(file) {
638
691
  tasks,
639
692
  });
640
693
  const authoringFindings = authoringLintFindings(result.tasks, file);
641
- for (const finding of authoringFindings.filter(({ code }) => code !== "criterion-scope")) {
694
+ const blockingCodes = new Set(["criterion-scope", "fence-symbol-absent"]);
695
+ const blockingFindings = authoringFindings.filter((finding) => options.strict || blockingCodes.has(finding.code));
696
+ for (const finding of authoringFindings.filter((finding) => !blockingFindings.includes(finding))) {
642
697
  console.warn(renderAuthoringFinding(finding));
643
698
  }
644
- const scopeErrors = authoringFindings.filter(({ code }) => code === "criterion-scope");
645
- if (scopeErrors.length > 0) {
646
- throw new CompileError(`${file} violates the criterion-scope authoring lint (${scopeErrors.length} error${scopeErrors.length === 1 ? "" : "s"}):\n`
647
- + scopeErrors.map((finding) => ` - ${renderAuthoringFinding(finding)}`).join("\n"));
699
+ if (blockingFindings.length > 0) {
700
+ const strict = options.strict ? " under --strict" : "";
701
+ const onlyCriterionScope = blockingFindings.every(({ code }) => code === "criterion-scope");
702
+ const label = onlyCriterionScope && !options.strict ? "the criterion-scope authoring lint" : `authoring lints${strict}`;
703
+ throw new CompileError(`${file} violates ${label} (${blockingFindings.length} error${blockingFindings.length === 1 ? "" : "s"}):\n`
704
+ + blockingFindings.map((finding) => ` - ${renderAuthoringFinding(finding)}`).join("\n"));
648
705
  }
649
706
  // v1.19 read-old/write-new: a plain-string acceptance item compiles as a judge oracle. This is the
650
707
  // one-time nudge toward typed oracles (command/test/judge); PRD/Spec Kit/GSD stay silent (untouched).
@@ -22,19 +22,41 @@ function testSources(repoRoot) {
22
22
  return [];
23
23
  }
24
24
  }
25
- function namedSources(test, tasks) {
26
- const stem = basename(test).replace(/\.test\.ts$/, "");
25
+ function sourceFiles(repoRoot) {
26
+ const root = join(repoRoot, "src");
27
+ try {
28
+ return readdirSync(root, { recursive: true, encoding: "utf8" })
29
+ .filter((path) => /\.(?:[cm]?[jt]sx?)$/.test(path))
30
+ .map((path) => `src/${normalize(path)}`)
31
+ .sort();
32
+ }
33
+ catch {
34
+ return [];
35
+ }
36
+ }
37
+ // OBS-898: expand the task's source declarations against ONE source-tree listing, once per compile.
38
+ // Tests then query this in-memory index; neither a test file nor a second corroboration pass can
39
+ // trigger another source walk or glob expansion.
40
+ function namedSourceIndex(tasks, allSources) {
27
41
  const matches = new Map();
28
42
  for (const task of tasks) {
29
- for (const entry of task.files.map(normalize).filter((path) => path.startsWith("src/") && !/[*?{[]/.test(path))) {
30
- const source = basename(entry, extname(entry));
31
- if (stem === source || stem.startsWith(`${source}-`)) {
32
- matches.set(`${task.id}:${entry}`, { taskId: task.id, source: entry });
43
+ for (const entry of task.files.map(normalize).filter((path) => path.startsWith("src/"))) {
44
+ const fromGlob = /[*?{[]/.test(entry);
45
+ const entries = fromGlob ? allSources.filter(filesGlob(entry)) : [entry];
46
+ for (const sourcePath of entries) {
47
+ matches.set(`${task.id}:${sourcePath}`, { taskId: task.id, source: sourcePath, fromGlob });
33
48
  }
34
49
  }
35
50
  }
36
51
  return [...matches.values()];
37
52
  }
53
+ function namedSources(test, index) {
54
+ const stem = basename(test).replace(/\.test\.ts$/, "");
55
+ return index.filter(({ source }) => {
56
+ const sourceStem = basename(source, extname(source));
57
+ return stem === sourceStem || stem.startsWith(`${sourceStem}-`);
58
+ });
59
+ }
38
60
  const moduleKey = (path) => normalize(path).replace(/\.(?:[cm]?[jt]sx?)$/, "");
39
61
  function directImportSpecifiers(text) {
40
62
  // Comments cannot create an edge. Keep strings intact because they are the import target.
@@ -148,19 +170,28 @@ export function ownershipFindings(tasks, repoRoot) {
148
170
  predictedBy.set(hit, ids);
149
171
  }
150
172
  }
173
+ const globOwnedTests = new Map();
174
+ const allSources = sourceFiles(repoRoot);
175
+ const namedIndex = namedSourceIndex(tasks, allSources);
151
176
  for (const source of sources) {
152
177
  const ids = predictedBy.get(source.path) ?? new Set();
153
- for (const { taskId } of namedSources(source.path, tasks))
154
- ids.add(taskId);
178
+ for (const named of namedSources(source.path, namedIndex)) {
179
+ ids.add(named.taskId);
180
+ if (named.fromGlob) {
181
+ const owners = globOwnedTests.get(source.path) ?? new Set();
182
+ owners.add(named.taskId);
183
+ globOwnedTests.set(source.path, owners);
184
+ }
185
+ }
155
186
  if (ids.size > 0)
156
187
  predictedBy.set(source.path, ids);
157
188
  }
158
189
  const findings = [];
159
190
  for (const [test, taskIds] of predictedBy) {
160
- if (owners(test).length === 0) {
191
+ if (owners(test).length === 0 && !(globOwnedTests.get(test)?.size)) {
161
192
  const ids = [...taskIds].sort();
162
193
  const source = sourceByPath.get(test);
163
- const evidence = corroboration(source, namedSources(test, tasks));
194
+ const evidence = corroboration(source, namedSources(test, namedIndex));
164
195
  findings.push({
165
196
  code: "unowned-test",
166
197
  test,
@@ -266,6 +266,7 @@ export declare const TickmarkrConfigSchema: z.ZodObject<{
266
266
  }, z.core.$strip>;
267
267
  review: z.ZodObject<{
268
268
  complexityThreshold: z.ZodNumber;
269
+ timeoutMs: z.ZodNumber;
269
270
  required: z.ZodBoolean;
270
271
  prefer: z.ZodOptional<z.ZodArray<z.ZodString>>;
271
272
  policy: z.ZodOptional<z.ZodEnum<{
@@ -355,6 +355,7 @@ export const TickmarkrConfigSchema = z.object({
355
355
  // because routing hints and the setup cockpit still read it as a complexity landmark; NOTHING in
356
356
  // src/gates/review.ts reads it any more, and a value here can no longer disable a review.
357
357
  complexityThreshold: z.number(),
358
+ timeoutMs: z.number().int().positive(),
358
359
  required: z.boolean(),
359
360
  prefer: z.array(z.string()).optional(),
360
361
  // R3: the operator's participation floor. "full" forces a cross-vendor review on every task;
@@ -484,8 +485,15 @@ export const DEFAULT_CONFIG = {
484
485
  // 2026-07-16 (cursor-agent 2026.07.09); gives cursor a cheap-tier channel so low-complexity shapes
485
486
  // stop burning its mid. Operator-approved 2026-07-16.
486
487
  "composer-2.5-fast": "cheap",
488
+ // OBS-871: LiveBench 2026_06_25, fetched 2026-09-03; released 2026-09-01.
489
+ "claude-fable-5-1": "frontier",
490
+ // OBS-871: LiveBench 2026_06_25, fetched 2026-09-03; released 2026-09-02.
491
+ "gemini-3.8-flash": "mid",
492
+ },
493
+ windows: {
494
+ "composer-2.5": 200_000, "composer-2.5-fast": 200_000,
495
+ "claude-fable-5-1": 1_000_000, "gemini-3.8-flash": 1_000_000,
487
496
  },
488
- windows: { "composer-2.5": 200_000, "composer-2.5-fast": 200_000 },
489
497
  },
490
498
  // GLM-5.2 → mid per benchmark policy (2026-07): SWE-bench Pro 62.1 (> GPT-5.5 58.6), FrontierSWE 74.4 ≈ Opus 4.8;
491
499
  // no independent Terminal-Bench score → conservative mid, overlays may raise.
@@ -507,8 +515,46 @@ export const DEFAULT_CONFIG = {
507
515
  // cross-harness) is pre-existing, FLEET-05 owns it.
508
516
  pi: {
509
517
  vendor: "zhipu", channel: "sub",
510
- models: { "zai/glm-5.2": "mid" },
511
- windows: { "zai/glm-5.2": 1_000_000 },
518
+ models: {
519
+ "zai/glm-5.2": "mid",
520
+ // OBS-871: LiveBench 2026_06_25, fetched 2026-09-03; GLM-5.3 released 2026-08-18.
521
+ "zai/glm-5.3": "frontier",
522
+ // OBS-871: same LiveBench source and 2026-09-03 fetch; speed variant remains mid.
523
+ "zai/glm-5.3-flash": "mid",
524
+ },
525
+ windows: { "zai/glm-5.2": 1_000_000, "zai/glm-5.3": 1_000_000, "zai/glm-5.3-flash": 200_000 },
526
+ },
527
+ // OBS-871: gateway ids and bands from LiveBench 2026_06_25, fetched 2026-09-03.
528
+ omp: {
529
+ vendor: "mixed", channel: "sub",
530
+ models: {
531
+ "google/gemini-3.8-flash": "mid",
532
+ "zai/glm-5.3": "frontier",
533
+ "alibaba/qwen3.8-max": "frontier",
534
+ },
535
+ modelOverrides: {
536
+ "google/gemini-3.8-flash": { vendor: "google" },
537
+ "zai/glm-5.3": { vendor: "zhipu" },
538
+ "alibaba/qwen3.8-max": { vendor: "alibaba" },
539
+ },
540
+ windows: {
541
+ "google/gemini-3.8-flash": 1_000_000,
542
+ "zai/glm-5.3": 1_000_000,
543
+ "alibaba/qwen3.8-max": 1_000_000,
544
+ },
545
+ },
546
+ // OBS-871: Qwen 3.8 Max scored 64.6 in LiveBench 2026_06_25, fetched 2026-09-03.
547
+ qwen: {
548
+ vendor: "alibaba", channel: "sub",
549
+ models: { "qwen3.8-max": "frontier" },
550
+ windows: { "qwen3.8-max": 1_000_000 },
551
+ },
552
+ // OBS-871: prime-agent listed this prime-inference route on 2026-09-03; GLM-5.2 remains mid.
553
+ "prime-agent": {
554
+ vendor: "mixed", channel: "sub",
555
+ models: { "prime-inference/z-ai/glm-5.2": "mid" },
556
+ modelOverrides: { "prime-inference/z-ai/glm-5.2": { vendor: "zhipu" } },
557
+ windows: { "prime-inference/z-ai/glm-5.2": 1_000_000 },
512
558
  },
513
559
  // Native grok CLI (Phase 40). grok-4.5 → mid: same benchmark provenance as the retired cursor-agent
514
560
  // grok-4.5-xhigh seed above (AA 54, TB2.1 83.3%, SWE-b Pro 64.7% — researched 2026-07-10).
@@ -543,7 +589,7 @@ export const DEFAULT_CONFIG = {
543
589
  judge: { adapter: "claude-code", model: "fable" },
544
590
  // R3: no `policy` floor — the neutral floor leaves the compiler's per-task assignment standing, so
545
591
  // the path-keyed rule is reachable out of the box rather than raised to full by construction.
546
- review: { complexityThreshold: 7, required: true, criticalPaths: [...DEFAULT_REVIEW_CRITICAL_PATHS] },
592
+ review: { complexityThreshold: 7, timeoutMs: 900_000, required: true, criticalPaths: [...DEFAULT_REVIEW_CRITICAL_PATHS] },
547
593
  consult: { adapter: "claude-code", model: "fable", stallMinutes: 15 },
548
594
  // v1.4: gate LLM calls (judge/review/consult) run headless by default; pane opts back into visible agents.
549
595
  // v1.2: workers are the real agent TUI in the pane; "print" restores the -p-rendered-in-pane path.
@@ -765,7 +811,7 @@ export function configTemplate(overlay) {
765
811
  # test: npm test
766
812
  # byShape:
767
813
  # docs: { acceptance: false, review: false } # baseline, evidence, and scope are mandatory
768
- # review: { complexityThreshold: 7, required: true, prefer: [codex:gpt-5.6-sol, kimi] }
814
+ # review: { complexityThreshold: 7, timeoutMs: 900000, required: true, prefer: [codex:gpt-5.6-sol, kimi] }
769
815
  # # prefer: ordered reviewer seat preference (adapter | adapter:model); ranks
770
816
  # # diversity-eligible channels only — never admits a same-vendor/same-model reviewer
771
817
  # consult: { adapter: claude-code, model: fable, stallMinutes: 15, prefer: [codex:gpt-5.6-sol, kimi:kimi-code/k3] }
@@ -56,6 +56,7 @@ export declare class HerdrDriver implements ExecutorDriver {
56
56
  private journal?;
57
57
  id: string;
58
58
  interactive: boolean;
59
+ readSource: string;
59
60
  private groups;
60
61
  private groupSerial;
61
62
  private deliverySerial;
@@ -70,6 +71,7 @@ export declare class HerdrDriver implements ExecutorDriver {
70
71
  private watches;
71
72
  constructor(bin?: string, workersPerTab?: number, time?: HerdrTimeSource, journal?: DriverJournal | undefined);
72
73
  private appendDispatchRetry;
74
+ private appendPaneClose;
73
75
  private openRunJournal;
74
76
  private liveSupervisionSeats;
75
77
  private journalReconcile;
@@ -142,6 +142,7 @@ export class HerdrDriver {
142
142
  journal;
143
143
  id = "herdr";
144
144
  interactive = true;
145
+ readSource = "recent-unwrapped";
145
146
  groups = new Map();
146
147
  // grouped slot()/close() mutate shared group state across awaits — serialize them so two
147
148
  // concurrent first members can never both create the group tab (mergeSerial idiom, daemon.ts)
@@ -195,6 +196,29 @@ export class HerdrDriver {
195
196
  // recovery to the file and the pipe while the operator's rail stays silent about it.
196
197
  Journal.open(repoRoot, owned.runId, this.narrate).append("dispatch-retry", owned.taskId, data);
197
198
  }
199
+ // OBS-906: close() is the harvest path, not a reconcile path, so the sweep's close row cannot
200
+ // describe it. Persist the worker slot identity beside the exact pane/tab address this path used.
201
+ // A driver used outside a daemon has no bound run journal; the injectable sink remains the unit
202
+ // seam there, while every daemon-created worktree is bound by worktree() below.
203
+ appendPaneClose(slot, placement) {
204
+ const owned = parseOwnedName(slot.name);
205
+ if (owned?.role !== "worker")
206
+ return;
207
+ try {
208
+ const data = { slot: slot.name, ...placement };
209
+ if (this.journal) {
210
+ this.journal("pane-close", slot.name, data);
211
+ return;
212
+ }
213
+ const repoRoot = this.journalRoots.get(slot.cwd);
214
+ if (!repoRoot)
215
+ return;
216
+ Journal.open(repoRoot, owned.runId, this.narrate).append("pane-close", owned.taskId, data);
217
+ }
218
+ catch {
219
+ /* close is best-effort; a journal failure must not make a successful pane reap fatal */
220
+ }
221
+ }
198
222
  openRunJournal(runId) {
199
223
  const roots = new Set(this.journalRoots.values());
200
224
  roots.add(process.cwd());
@@ -1100,11 +1124,15 @@ export class HerdrDriver {
1100
1124
  return this.serial(() => this.closeGrouped(slot));
1101
1125
  }
1102
1126
  if (slot.tabId) {
1103
- await this.herdr(`tab close ${shq(slot.tabId)}`); // reaps the slot's whole tab, best-effort
1127
+ const closed = await this.herdr(`tab close ${shq(slot.tabId)}`); // reaps the slot's whole tab, best-effort
1128
+ if (closed.code === 0)
1129
+ this.appendPaneClose(slot, { tabId: slot.tabId });
1104
1130
  return;
1105
1131
  }
1106
1132
  const pane = await this.paneId(slot);
1107
- await this.herdr(`pane close ${shq(pane)}`); // best-effort
1133
+ const closed = await this.herdr(`pane close ${shq(pane)}`); // best-effort
1134
+ if (closed.code === 0)
1135
+ this.appendPaneClose(slot, { paneId: pane });
1108
1136
  }
1109
1137
  // D-08 ref-counted teardown, PER GENERATION (VIS-09 item 2): pane close per member; a generation's
1110
1138
  // tab closes only when ITS OWN last member leaves; the group entry dies when all generations are gone.
@@ -1118,7 +1146,9 @@ export class HerdrDriver {
1118
1146
  if (!gen)
1119
1147
  return; // generation already torn down — its tab is gone
1120
1148
  const pane = await this.paneId(slot);
1121
- await this.herdr(`pane close ${shq(pane)}`); // best-effort
1149
+ const closed = await this.herdr(`pane close ${shq(pane)}`); // best-effort
1150
+ if (closed.code === 0)
1151
+ this.appendPaneClose(slot, { paneId: pane, tabId: gen.tabId });
1122
1152
  gen.members = gen.members.filter((m) => m.name !== slot.name);
1123
1153
  await this.renameGroupTab(gen);
1124
1154
  if (gen.members.length === 0) {
@@ -1339,7 +1369,16 @@ export class HerdrDriver {
1339
1369
  if (alive.has(tab))
1340
1370
  continue;
1341
1371
  const closed = await this.herdr(`tab close ${shq(tab)}`);
1342
- if (closed.code !== 0) {
1372
+ if (closed.code !== 0 && /tab_not_found/i.test(`${closed.stderr}\n${closed.stdout}`)) {
1373
+ this.journalReconcile(journalHandle, "tab-reconcile-close-skipped", undefined, {
1374
+ tabId: tab,
1375
+ runId,
1376
+ sweeperRunId: runId,
1377
+ exitCode: 0,
1378
+ reason: "tab_not_found",
1379
+ });
1380
+ }
1381
+ else if (closed.code !== 0) {
1343
1382
  this.journalReconcile(journalHandle, "tab-reconcile-close-failed", undefined, {
1344
1383
  tabId: tab,
1345
1384
  runId,