tickmarkr 1.87.0 → 1.90.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/catalog.d.ts +18 -1
- package/dist/adapters/catalog.js +44 -1
- package/dist/adapters/fake.d.ts +2 -1
- package/dist/adapters/fake.js +7 -0
- package/dist/adapters/grok.js +11 -0
- package/dist/adapters/kimi.d.ts +2 -1
- package/dist/adapters/kimi.js +36 -0
- package/dist/adapters/opencode.js +17 -0
- package/dist/adapters/pi.js +11 -0
- package/dist/adapters/prompt.js +8 -1
- package/dist/adapters/registry.js +76 -57
- package/dist/adapters/types.d.ts +34 -3
- package/dist/adapters/types.js +99 -1
- package/dist/cli/commands/approve.d.ts +2 -0
- package/dist/cli/commands/approve.js +104 -84
- package/dist/cli/commands/compile.d.ts +1 -1
- package/dist/cli/commands/compile.js +29 -12
- package/dist/cli/commands/doctor.d.ts +16 -0
- package/dist/cli/commands/doctor.js +52 -0
- package/dist/cli/commands/init.js +2 -1
- package/dist/cli/commands/plan.d.ts +1 -1
- package/dist/cli/commands/plan.js +10 -1
- package/dist/cli/commands/report.js +49 -0
- package/dist/cli/commands/status.js +298 -96
- package/dist/cli/commands/verify.d.ts +9 -0
- package/dist/cli/commands/verify.js +177 -0
- package/dist/cli/harness.d.ts +13 -0
- package/dist/cli/harness.js +50 -0
- package/dist/cli/index.d.ts +1 -1
- package/dist/cli/index.js +3 -1
- package/dist/compile/collateral.js +11 -11
- package/dist/compile/common.js +2 -2
- package/dist/compile/index.d.ts +14 -3
- package/dist/compile/index.js +36 -10
- package/dist/compile/native.d.ts +15 -1
- package/dist/compile/native.js +310 -28
- package/dist/config/config.js +2 -2
- package/dist/drivers/subprocess.d.ts +6 -1
- package/dist/drivers/subprocess.js +9 -4
- package/dist/gates/acceptance.d.ts +21 -1
- package/dist/gates/acceptance.js +67 -22
- package/dist/gates/artifact-manifest.d.ts +119 -0
- package/dist/gates/artifact-manifest.js +357 -0
- package/dist/gates/baseline.d.ts +45 -5
- package/dist/gates/baseline.js +119 -15
- package/dist/gates/llm.js +37 -26
- package/dist/gates/review.d.ts +16 -11
- package/dist/gates/review.js +44 -150
- package/dist/gates/run-gates.js +124 -7
- package/dist/gates/scope.js +3 -3
- package/dist/graph/files-glob.d.ts +18 -0
- package/dist/graph/files-glob.js +22 -0
- package/dist/graph/schema.d.ts +3 -1
- package/dist/graph/schema.js +4 -1
- package/dist/run/daemon.d.ts +44 -0
- package/dist/run/daemon.js +2334 -1973
- package/dist/run/git.d.ts +53 -0
- package/dist/run/git.js +119 -5
- package/dist/run/interactive-seed.d.ts +6 -2
- package/dist/run/interactive-seed.js +72 -5
- package/dist/run/journal.d.ts +9 -1
- package/dist/run/journal.js +99 -9
- package/dist/run/lock.d.ts +11 -0
- package/dist/run/lock.js +97 -6
- package/dist/run/merge.d.ts +4 -1
- package/dist/run/merge.js +26 -7
- package/dist/run/outcome.d.ts +50 -0
- package/dist/run/outcome.js +152 -0
- package/dist/run/protocol.d.ts +460 -0
- package/dist/run/protocol.js +433 -0
- package/dist/run/supervision.d.ts +29 -0
- package/dist/run/supervision.js +189 -0
- package/fixtures/authoring-lints/01-awk-range-self-pass.spec.md +12 -0
- package/fixtures/authoring-lints/02-judge-text-key-miss.spec.md +7 -0
- package/fixtures/authoring-lints/03-c1-t41-rendered-observable.spec.md +8 -0
- package/fixtures/authoring-lints/04-c1-t24-named-file.spec.md +8 -0
- package/fixtures/authoring-lints/05-c2-t24-t28-dep-inversion.spec.md +7 -0
- package/fixtures/authoring-lints/06-c2-denumbered-coupling.spec.md +7 -0
- package/fixtures/authoring-lints/07-c3a-t41-line-count-proxy.spec.md +7 -0
- package/fixtures/authoring-lints/08-c3b-t41-governance-referent.spec.md +7 -0
- package/fixtures/authoring-lints/09-c4-universals-without-pointer.spec.md +7 -0
- package/fixtures/authoring-lints/10-c5-t34-conjunct-flood.spec.md +7 -0
- package/fixtures/authoring-lints/11-c6-t34-q3-q9-q20-bundle.spec.md +7 -0
- package/fixtures/authoring-lints/12-c7-t24-prose-seam.spec.md +8 -0
- package/fixtures/wrapped-acceptance.native.md +29 -0
- package/package.json +1 -1
- package/schema/rungraph.schema.json +21 -2
- package/skills/tickmarkr-overseer/SKILL.md +262 -5
- package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +79 -8
- package/skills/tickmarkr-overseer/scripts/watch-contamination.sh +80 -0
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +86 -0
- package/skills/tickmarkr-overseer/scripts/watch-parks.sh +96 -0
- package/skills/tickmarkr-overseer/scripts/watch-pending-input.sh +201 -0
package/dist/gates/run-gates.js
CHANGED
|
@@ -143,12 +143,76 @@ export async function runGates(task, ctx) {
|
|
|
143
143
|
heldTest = undefined;
|
|
144
144
|
await ctx.onGate?.({ phase: "end", gate: "test", result: held });
|
|
145
145
|
}
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
146
|
+
const sorted = [...results].sort((a, b) => GATE_NAMES.indexOf(a.gate) - GATE_NAMES.indexOf(b.gate));
|
|
147
|
+
// v1.87 T5: no round returns a MERGEABLE GREEN on a dirty tree. The battery is not the only gate
|
|
148
|
+
// that executes shell in this worktree — the acceptance gate runs command and named-test oracles
|
|
149
|
+
// (acceptance.ts:264,275) and both verdict gates dispatch a vendor CLI here — so the last word on
|
|
150
|
+
// cleanliness has to be the round's last act rather than the battery's. `results` is what the
|
|
151
|
+
// daemon merges on (daemon.ts `results.every(gateSatisfied)`), so the withdrawal lands there; the
|
|
152
|
+
// journal keeps both the green and its retraction, the honest record of a verdict that did not
|
|
153
|
+
// survive its own round.
|
|
154
|
+
const last = sorted[sorted.length - 1];
|
|
155
|
+
if (last && sorted.every((r) => r.pass || r.meta?.skipped === true)) {
|
|
156
|
+
const dirt = await dirtyWorktree();
|
|
157
|
+
if (dirt) {
|
|
158
|
+
const refusal = dirtyRoundRefusal(last.gate, dirt);
|
|
159
|
+
results[results.indexOf(last)] = refusal;
|
|
160
|
+
sorted[sorted.length - 1] = refusal;
|
|
161
|
+
await ctx.onGate?.({ phase: "end", gate: refusal.gate, result: refusal });
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
return { results: sorted, commits };
|
|
150
165
|
};
|
|
151
166
|
const toolGates = ["build", "test", "lint"].filter(enabled);
|
|
167
|
+
/**
|
|
168
|
+
* v1.87 T5: the shell gates run their commands against the WORKING TREE, while evidence, scope,
|
|
169
|
+
* the judged diff and the merge all read COMMITS. Uncommitted work is therefore visible to
|
|
170
|
+
* build/test/lint and invisible to everything that decides what ships — a green battery on a dirty
|
|
171
|
+
* tree certifies a tree nobody will ever merge, and the committed diff it stands for was never run.
|
|
172
|
+
*
|
|
173
|
+
* That is not gatable-with-a-caveat, so the battery refuses it rather than gating it and hoping.
|
|
174
|
+
* An unreadable `git status` is refused on the same rule: a tree that cannot be proven clean is not
|
|
175
|
+
* proven clean. Returns the dirt (porcelain lines) to name in the refusal, or undefined when clean.
|
|
176
|
+
*
|
|
177
|
+
* The one exemption is tickmarkr's OWN droppings — root-level `.tickmarkr-*` (the adapters' usage
|
|
178
|
+
* record). The harness wrote those, not the worker; they are not work anyone meant to merge, and
|
|
179
|
+
* refusing a tree for the harness's own litter would fail every metered run. Nothing else is
|
|
180
|
+
* exempt: an untracked source file is uncommitted work by every reading git offers.
|
|
181
|
+
*/
|
|
182
|
+
const dirtyWorktree = async () => {
|
|
183
|
+
const r = await shGit("GIT_OPTIONAL_LOCKS=0 git status --porcelain", ctx.worktree);
|
|
184
|
+
if (r.code !== 0)
|
|
185
|
+
return `git status failed (exit ${r.code}) — the worktree cannot be proven clean`;
|
|
186
|
+
const entries = r.stdout
|
|
187
|
+
.split("\n")
|
|
188
|
+
.map((l) => l.trimEnd())
|
|
189
|
+
.filter((l) => l.trim() && !/^.. \.tickmarkr-[^/]*$/.test(l));
|
|
190
|
+
return entries.length ? entries.join("\n") : undefined;
|
|
191
|
+
};
|
|
192
|
+
const DIRTY_WHY = `refusing to gate a dirty worktree: the shell gates run against the working tree while `
|
|
193
|
+
+ `evidence, scope and the merge read commits, so these uncommitted changes would be gated `
|
|
194
|
+
+ `and never merged (and the committed diff would never be run)`;
|
|
195
|
+
// `left` names the command that CREATED the dirt when one did; a round-entry refusal has no culprit.
|
|
196
|
+
const dirtyRefusal = (gate, dirt, left) => ({
|
|
197
|
+
gate,
|
|
198
|
+
pass: false,
|
|
199
|
+
details: DIRTY_WHY
|
|
200
|
+
+ (left ? `. The ${gate} command (${left}) left them behind, so every gate after it would judge a tree nobody will merge:\n` : `:\n`)
|
|
201
|
+
+ dirt,
|
|
202
|
+
meta: { dirtyWorktree: true, ...(left ? { dirtiedBy: gate } : {}) },
|
|
203
|
+
});
|
|
204
|
+
// The round-end withdrawal (see `done`). It blames no command: whatever dirtied the tree ran after
|
|
205
|
+
// the last cleanliness check, and naming a culprit this function cannot identify would be a worse
|
|
206
|
+
// record than naming the fact. `gate` is the verdict being withdrawn, not an accusation about who wrote.
|
|
207
|
+
const dirtyRoundRefusal = (gate, dirt) => ({
|
|
208
|
+
gate,
|
|
209
|
+
pass: false,
|
|
210
|
+
details: `${DIRTY_WHY}. Every gate of this round was satisfied and the round ended dirty — something after `
|
|
211
|
+
+ `the last cleanliness check (the acceptance gate's command/test oracles, or a verdict gate's `
|
|
212
|
+
+ `vendor CLI) wrote into the worktree — so this mergeable result is withdrawn rather than merged. `
|
|
213
|
+
+ `Uncommitted at round end:\n${dirt}`,
|
|
214
|
+
meta: { dirtyWorktree: true, dirtyAtRoundEnd: true },
|
|
215
|
+
});
|
|
152
216
|
// build/test/lint vs the shared baseline
|
|
153
217
|
const runBattery = async (commands, selected) => {
|
|
154
218
|
if (!toolGates.length)
|
|
@@ -158,9 +222,15 @@ export async function runGates(task, ctx) {
|
|
|
158
222
|
// not at true execution start. They are collectively sub-second (measured), so the debounce
|
|
159
223
|
// suppresses them anyway; split compareToBaseline only if a tool gate ever gets slow.
|
|
160
224
|
const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, toolGates);
|
|
225
|
+
// The same refusal AFTER the commands, because a green command can dirty the tree the check
|
|
226
|
+
// above just proved clean. Batched, legacy cannot say WHICH command did it, so the refusal
|
|
227
|
+
// lands on the last gate that had one — the round dies there either way. A red battery is
|
|
228
|
+
// reported as the red it is: the round already ends, and the command output is the better lead.
|
|
229
|
+
const dirt = toolResults.every((r) => r.pass) ? await dirtyWorktree() : undefined;
|
|
230
|
+
const blame = dirt ? [...toolGates].reverse().find((g) => commands[g]) : undefined;
|
|
161
231
|
for (const r of toolResults) {
|
|
162
232
|
await emitStart(r.gate);
|
|
163
|
-
await record(r);
|
|
233
|
+
await record(r.gate === blame ? dirtyRefusal(blame, dirt, commands[blame]) : r);
|
|
164
234
|
}
|
|
165
235
|
return;
|
|
166
236
|
}
|
|
@@ -169,6 +239,18 @@ export async function runGates(task, ctx) {
|
|
|
169
239
|
for (const g of toolGates) {
|
|
170
240
|
await emitStart(g);
|
|
171
241
|
const [r] = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [g]);
|
|
242
|
+
// The pre-battery check proves the tree clean ONCE; a command that exits 0 having rewritten a
|
|
243
|
+
// tracked file makes it dirty again, and every gate after it — including the next shell gate,
|
|
244
|
+
// which would then run against bytes HEAD does not hold — inherits that. So re-check after each
|
|
245
|
+
// command, the last one included, and fail the gate whose command did it. (A red command needs
|
|
246
|
+
// no check: it already ends the round, and its own output is the truer verdict.)
|
|
247
|
+
if (r.pass && commands[g]) {
|
|
248
|
+
const dirt = await dirtyWorktree();
|
|
249
|
+
if (dirt) {
|
|
250
|
+
await record(dirtyRefusal(g, dirt, commands[g]));
|
|
251
|
+
return;
|
|
252
|
+
}
|
|
253
|
+
}
|
|
172
254
|
if (g === "test" && selected) {
|
|
173
255
|
const screened = { ...r, meta: { ...r.meta, selectedTests: selected } };
|
|
174
256
|
// green: held (see heldTest) so the full suite below can supersede it with ONE verdict.
|
|
@@ -194,7 +276,23 @@ export async function runGates(task, ctx) {
|
|
|
194
276
|
commits = e.commits;
|
|
195
277
|
return { gate: e.gate, pass: e.pass, details: e.details };
|
|
196
278
|
};
|
|
197
|
-
|
|
279
|
+
/**
|
|
280
|
+
* v1.87 T5: the allowlist is read ONCE — at entry, before any gate of this round runs — copied out
|
|
281
|
+
* of `ctx.cfg` and frozen. Both halves are the enforcement: the copy means a later write to the
|
|
282
|
+
* daemon's live config object cannot reach the gate mid-round (the screen at line ~296 and the
|
|
283
|
+
* canonical scope gate below are two separate reads of it), and the freeze means nothing this file
|
|
284
|
+
* hands to `scopeGate` can be widened in flight either. Nothing anywhere writes it back.
|
|
285
|
+
*
|
|
286
|
+
* The boundary, stated rather than implied: this binds the allowlist for the lifetime of a round,
|
|
287
|
+
* which is the largest unit this function owns. It cannot speak for a `tickmarkr resume`, which is
|
|
288
|
+
* a new process whose config the daemon resolves afresh (src/run/daemon.ts) — binding an allowlist
|
|
289
|
+
* across a restart would have to live there, is outside this task's file scope, and is claimed
|
|
290
|
+
* neither here nor in the worker prompt. What the prompt does claim is what holds: a WORKER has no
|
|
291
|
+
* way to change this list, mid-round or otherwise.
|
|
292
|
+
*/
|
|
293
|
+
const allowDeviations = [...(ctx.cfg.scope?.allowDeviations ?? [])];
|
|
294
|
+
Object.freeze(allowDeviations);
|
|
295
|
+
const scopeResult = () => scopeGate(ctx.worktree, ctx.baseRef, task.files, ctx.result, allowDeviations);
|
|
198
296
|
const runGate = async (gate, compute) => {
|
|
199
297
|
await emitStart(gate);
|
|
200
298
|
await record(await compute());
|
|
@@ -337,6 +435,19 @@ export async function runGates(task, ctx) {
|
|
|
337
435
|
}
|
|
338
436
|
return rv;
|
|
339
437
|
};
|
|
438
|
+
// v1.87 T5: the refusal is the FIRST thing a round does, whatever that round is configured to run.
|
|
439
|
+
// Guarding only the configured build/test/lint commands left the hole this repairs: the battery is
|
|
440
|
+
// not the only gate that executes shell in this worktree — the acceptance gate runs command and
|
|
441
|
+
// named-test oracles — so a task with NO tool command configured skipped the check entirely and its
|
|
442
|
+
// oracles judged uncommitted state, on a commit whose diff nobody had run. One check at the top, and
|
|
443
|
+
// no shell-executing gate path is reachable on a dirty tree. It lands on the first gate of this
|
|
444
|
+
// round's sequence: the round dies there, exactly as it does on a red command.
|
|
445
|
+
const entryDirt = sequence.length ? await dirtyWorktree() : undefined;
|
|
446
|
+
if (entryDirt) {
|
|
447
|
+
await emitStart(sequence[0]);
|
|
448
|
+
await record(dirtyRefusal(sequence[0], entryDirt));
|
|
449
|
+
return done();
|
|
450
|
+
}
|
|
340
451
|
if (v185 && await screenBlocks())
|
|
341
452
|
return done();
|
|
342
453
|
// A non-final round may run only the tests covering its own diff; the merge-candidate round below
|
|
@@ -404,8 +515,14 @@ export async function runGates(task, ctx) {
|
|
|
404
515
|
// stream, and `fullSuite` says which suite spoke while `selectedTests` keeps what the screen ran.
|
|
405
516
|
if (selected) {
|
|
406
517
|
await emitStart("test");
|
|
518
|
+
// This is the last shell command a round can run — the judge's named-test oracle (acceptance.ts)
|
|
519
|
+
// may have run one before it, and every gate between the battery and here reads commits only, so
|
|
520
|
+
// a clean tree HERE is what makes "the gated commit is the tested tree" true at merge time.
|
|
407
521
|
const [full] = await compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"]);
|
|
408
|
-
const
|
|
522
|
+
const dirt = full.pass ? await dirtyWorktree() : undefined;
|
|
523
|
+
const merged = dirt
|
|
524
|
+
? dirtyRefusal("test", dirt, ctx.commands.test)
|
|
525
|
+
: { ...full, meta: { ...full.meta, fullSuite: true, selectedTests: selected } };
|
|
409
526
|
results[results.findIndex((r) => r.gate === "test")] = merged;
|
|
410
527
|
heldTest = undefined;
|
|
411
528
|
await ctx.onGate?.({ phase: "end", gate: "test", result: merged });
|
package/dist/gates/scope.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import
|
|
1
|
+
import { filesGlob } from "../graph/files-glob.js";
|
|
2
2
|
import { shGitOk } from "../run/git.js";
|
|
3
3
|
// OBS-61: the live integration tip can advance between attempts (sibling merge) while a resumed
|
|
4
4
|
// worktree stays parented on the old base — merge-base(tip, HEAD) pins the diff to the worktree's
|
|
@@ -11,7 +11,7 @@ export async function scopeDiffBase(worktree, integrationTip) {
|
|
|
11
11
|
}
|
|
12
12
|
/** Pure offender split — corpus-replay oracle exercises this exact decision logic. */
|
|
13
13
|
export function dispositionOffenders(offenders, allowDeviations) {
|
|
14
|
-
const allowedMatch = allowDeviations.length ?
|
|
14
|
+
const allowedMatch = allowDeviations.length ? filesGlob(allowDeviations) : () => false;
|
|
15
15
|
const allowed = [];
|
|
16
16
|
const hard = [];
|
|
17
17
|
for (const f of offenders) {
|
|
@@ -27,7 +27,7 @@ export async function scopeGate(worktree, integrationTip, files, result, allowDe
|
|
|
27
27
|
return { gate: "scope", pass: true, details: "no file scope declared — unrestricted" };
|
|
28
28
|
const baseRef = await scopeDiffBase(worktree, integrationTip);
|
|
29
29
|
const changed = (await shGitOk(`git diff --name-only '${baseRef}..HEAD'`, worktree)).trim().split("\n").filter(Boolean);
|
|
30
|
-
const inScope =
|
|
30
|
+
const inScope = filesGlob(files); // the ONE files[] matcher — src/graph/files-glob.ts (Q120s)
|
|
31
31
|
const offenders = changed.filter((f) => !inScope(f));
|
|
32
32
|
if (!offenders.length)
|
|
33
33
|
return { gate: "scope", pass: true, details: `all ${changed.length} changed files in scope` };
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The ONE matcher for files[] / scope.allowDeviations semantics (Q120s, TRIAL T-OBS-4).
|
|
3
|
+
*
|
|
4
|
+
* Parens in a files[] entry are LITERAL path characters. Expo Router and Next.js name
|
|
5
|
+
* group directories `(app)`, `(marketing)` — but picomatch compiles a bare `(app)` to a
|
|
6
|
+
* regex CAPTURE GROUP matching `app`, so a pattern naming such a path can never match it
|
|
7
|
+
* (SentioQ run-1: T3 burned 3 dispatches on a scope red unwinnable by construction).
|
|
8
|
+
* Brackets need no help: this picomatch already compiles `[id]` to the alternation
|
|
9
|
+
* `(?:\[id\]|[id])`, literal-or-class.
|
|
10
|
+
*
|
|
11
|
+
* The price is extglobs (`@(a|b)` etc.) become literal text in files[] entries — they
|
|
12
|
+
* have no recorded use in any spec, and a scope allowlist wants boring, predictable
|
|
13
|
+
* matching over regex power. Every consumer of files[]-shaped patterns MUST match
|
|
14
|
+
* through this module; a second picomatch call with hand-rolled options is how the
|
|
15
|
+
* gate and the compiler drift apart.
|
|
16
|
+
*/
|
|
17
|
+
export declare const literalParens: (pattern: string) => string;
|
|
18
|
+
export declare function filesGlob(patterns: string | string[]): (path: string) => boolean;
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
import picomatch from "picomatch";
|
|
2
|
+
/**
|
|
3
|
+
* The ONE matcher for files[] / scope.allowDeviations semantics (Q120s, TRIAL T-OBS-4).
|
|
4
|
+
*
|
|
5
|
+
* Parens in a files[] entry are LITERAL path characters. Expo Router and Next.js name
|
|
6
|
+
* group directories `(app)`, `(marketing)` — but picomatch compiles a bare `(app)` to a
|
|
7
|
+
* regex CAPTURE GROUP matching `app`, so a pattern naming such a path can never match it
|
|
8
|
+
* (SentioQ run-1: T3 burned 3 dispatches on a scope red unwinnable by construction).
|
|
9
|
+
* Brackets need no help: this picomatch already compiles `[id]` to the alternation
|
|
10
|
+
* `(?:\[id\]|[id])`, literal-or-class.
|
|
11
|
+
*
|
|
12
|
+
* The price is extglobs (`@(a|b)` etc.) become literal text in files[] entries — they
|
|
13
|
+
* have no recorded use in any spec, and a scope allowlist wants boring, predictable
|
|
14
|
+
* matching over regex power. Every consumer of files[]-shaped patterns MUST match
|
|
15
|
+
* through this module; a second picomatch call with hand-rolled options is how the
|
|
16
|
+
* gate and the compiler drift apart.
|
|
17
|
+
*/
|
|
18
|
+
export const literalParens = (pattern) => pattern.replace(/\\?[()]/g, (m) => (m.length === 2 ? m : `\\${m}`));
|
|
19
|
+
export function filesGlob(patterns) {
|
|
20
|
+
const list = (Array.isArray(patterns) ? patterns : [patterns]).map(literalParens);
|
|
21
|
+
return picomatch(list, { dot: true });
|
|
22
|
+
}
|
package/dist/graph/schema.d.ts
CHANGED
|
@@ -4,11 +4,13 @@ export declare const GRAPH_ROUTING_MODES: readonly ["partner-led", "risk-based",
|
|
|
4
4
|
export declare const STATUSES: readonly ["pending", "running", "gated", "failed", "done", "human"];
|
|
5
5
|
export declare const GATE_NAMES: readonly ["build", "test", "lint", "evidence", "scope", "acceptance", "review"];
|
|
6
6
|
export declare const TIERS: readonly ["cheap", "mid", "frontier"];
|
|
7
|
+
export declare const SPEC_SOURCES: readonly ["speckit", "gsd", "prd", "native"];
|
|
7
8
|
export declare const ORACLES: readonly ["command", "test", "judge"];
|
|
8
9
|
export type Shape = (typeof SHAPES)[number];
|
|
9
10
|
export type TaskStatus = (typeof STATUSES)[number];
|
|
10
11
|
export type GateName = (typeof GATE_NAMES)[number];
|
|
11
12
|
export type Oracle = (typeof ORACLES)[number];
|
|
13
|
+
export type SpecSource = (typeof SPEC_SOURCES)[number];
|
|
12
14
|
export declare const AcceptanceItemSchema: z.ZodUnion<readonly [z.ZodString, z.ZodObject<{
|
|
13
15
|
oracle: z.ZodLiteral<"command">;
|
|
14
16
|
command: z.ZodString;
|
|
@@ -110,10 +112,10 @@ export declare const RunGraphSchema: z.ZodObject<{
|
|
|
110
112
|
gsd: "gsd";
|
|
111
113
|
prd: "prd";
|
|
112
114
|
native: "native";
|
|
113
|
-
taskmaster: "taskmaster";
|
|
114
115
|
}>;
|
|
115
116
|
paths: z.ZodArray<z.ZodString>;
|
|
116
117
|
hash: z.ZodString;
|
|
118
|
+
base: z.ZodOptional<z.ZodString>;
|
|
117
119
|
}, z.core.$strip>;
|
|
118
120
|
tasks: z.ZodArray<z.ZodObject<{
|
|
119
121
|
id: z.ZodString;
|
package/dist/graph/schema.js
CHANGED
|
@@ -7,6 +7,7 @@ export const STATUSES = ["pending", "running", "gated", "failed", "done", "human
|
|
|
7
7
|
export const GATE_NAMES = ["build", "test", "lint", "evidence", "scope", "acceptance", "review"];
|
|
8
8
|
const MANDATORY_GATES = ["build", "test", "lint", "evidence", "scope"];
|
|
9
9
|
export const TIERS = ["cheap", "mid", "frontier"];
|
|
10
|
+
export const SPEC_SOURCES = ["speckit", "gsd", "prd", "native"];
|
|
10
11
|
// v1.19 acceptance oracles: command (exit code), test (named test), judge (LLM, free-text rubric).
|
|
11
12
|
// A plain string is the read-old/write-new compat form — semantically a judge oracle (spec §2).
|
|
12
13
|
export const ORACLES = ["command", "test", "judge"];
|
|
@@ -98,9 +99,11 @@ export const RunGraphSchema = z
|
|
|
98
99
|
// precedence: run flag > this > repo config > global config > default (risk-based).
|
|
99
100
|
mode: z.enum(GRAPH_ROUTING_MODES).optional(),
|
|
100
101
|
spec: z.object({
|
|
101
|
-
source: z.enum(
|
|
102
|
+
source: z.enum(SPEC_SOURCES),
|
|
102
103
|
paths: z.array(z.string()),
|
|
103
104
|
hash: z.string(),
|
|
105
|
+
// Q11 compile half: an author-declared ref only. Runtime Git resolution/enforcement is later.
|
|
106
|
+
base: z.string().min(1).optional(),
|
|
104
107
|
}),
|
|
105
108
|
tasks: z.array(TaskSchema).min(1),
|
|
106
109
|
})
|
package/dist/run/daemon.d.ts
CHANGED
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
import { type WorkerAdapter } from "../adapters/types.js";
|
|
2
2
|
import { type ModeResolution, type RoutingMode, type TickmarkrConfig } from "../config/config.js";
|
|
3
3
|
import { type ExecutorDriver } from "../drivers/types.js";
|
|
4
|
+
import { type Baseline } from "../gates/baseline.js";
|
|
5
|
+
import type { GateResult } from "../gates/types.js";
|
|
4
6
|
import { Journal, type JournalEvent } from "./journal.js";
|
|
5
7
|
export interface RunOptions {
|
|
6
8
|
runId?: string;
|
|
@@ -15,6 +17,7 @@ export interface RunOptions {
|
|
|
15
17
|
mode?: RoutingMode;
|
|
16
18
|
narrate?: (event: JournalEvent) => void;
|
|
17
19
|
exit?: (code: number) => void;
|
|
20
|
+
supervise?: boolean;
|
|
18
21
|
}
|
|
19
22
|
export type ModeSource = "run flag" | "spec" | "repo config" | "global config" | "default";
|
|
20
23
|
export interface ResolvedRunMode {
|
|
@@ -40,8 +43,45 @@ export interface RunSummary {
|
|
|
40
43
|
blocked: string[];
|
|
41
44
|
tipVerify?: "passed" | "failed";
|
|
42
45
|
lastMergedTask?: string;
|
|
46
|
+
/** T14: did every approval this run accepted actually get enacted, or did the run end over one? */
|
|
47
|
+
approvalDisposition?: "complete" | "outstanding";
|
|
48
|
+
/** the accepted approvals that never reached a dispatch — named, never left to the park buckets */
|
|
49
|
+
outstandingApprovals?: string[];
|
|
43
50
|
}
|
|
51
|
+
/**
|
|
52
|
+
* T14: approvals the run accepted and never acted on. `approved` above is built ONCE at startup —
|
|
53
|
+
* deliberately, replay determinism depends on it — so an approval written while the daemon is live is
|
|
54
|
+
* inert for that run. Without this the run-end record stated only buckets and tipVerify, both
|
|
55
|
+
* accurate, over a milestone that was silently incomplete: run …230 ended tipVerify "passed" with two
|
|
56
|
+
* upheld approvals and zero subsequent dispatches. Scored per task on its NEWEST approval: a later
|
|
57
|
+
* approval is the live decision, and the events that answer it are the ones after it.
|
|
58
|
+
*/
|
|
59
|
+
export declare function outstandingApprovals(events: JournalEvent[]): string[];
|
|
44
60
|
export declare function formatSummary(s: RunSummary): string;
|
|
61
|
+
/**
|
|
62
|
+
* R3 (OBS-186): a gate that DECLINED to run is not a gate that failed. The review gate's skip branch
|
|
63
|
+
* no longer forges `pass: true` to buy passage, so the merge decision has to read the same predicate
|
|
64
|
+
* the run surfaces already read (src/run/activity.ts): pass, or an honest declared skip. Without this
|
|
65
|
+
* the honesty change would silently park every judge-only task at merge — an unrun gate blocking work
|
|
66
|
+
* it was never asked to review. `skipped` is set only by a gate that says so about ITSELF; a red
|
|
67
|
+
* verdict from a review that actually ran still fails here, exactly as before.
|
|
68
|
+
*
|
|
69
|
+
* ONE pair of predicates, every fold. `!g.pass` was correct only while the sole `pass:false` producer
|
|
70
|
+
* was a gate that actually failed; the moment a decline can be recorded red, every `!g.pass` in this
|
|
71
|
+
* file — the retry feedback brief, the review-fix eligibility test, the failing-battery list the
|
|
72
|
+
* ladder and the fingerprint cap are scored on, the structured findings attached to a blocking
|
|
73
|
+
* verdict — reads an unrun gate as a defect. `gateFailed` is the seam they now share, and the journal
|
|
74
|
+
* write below is the seam every OUT-of-file fold shares.
|
|
75
|
+
*/
|
|
76
|
+
/**
|
|
77
|
+
* T9: `meta.infra === true` overrides BOTH clauses above. A runner that died on the machine
|
|
78
|
+
* (spawn EAGAIN, OOM) without completing a suite answered nothing about the work, so the honest
|
|
79
|
+
* report of that fact must not double as authorization to merge — and it is the merge predicate,
|
|
80
|
+
* not the gate, that has to say so: classifying the failure into infra metadata while still
|
|
81
|
+
* reporting `pass: true` is exactly how a run that never verified anything gets merged. A declared
|
|
82
|
+
* skip stays satisfied; a gate that ran and passed stays satisfied.
|
|
83
|
+
*/
|
|
84
|
+
export declare const gateSatisfied: (g: GateResult) => boolean;
|
|
45
85
|
/**
|
|
46
86
|
* T4 (OBS-265): the journal with the review objections a round did NOT hinge on removed. Judge and
|
|
47
87
|
* review are now launched together, so a round can journal a failed review that the serial walk would
|
|
@@ -92,9 +132,13 @@ export declare function commandsHash(commands: Record<string, string>): string;
|
|
|
92
132
|
*/
|
|
93
133
|
export declare function verifyIntegrationTipCached(intWt: string, commands: Record<string, string>, journal: Journal, opts?: {
|
|
94
134
|
lastMergedTask?: string;
|
|
135
|
+
baseline?: Baseline;
|
|
95
136
|
}): Promise<boolean>;
|
|
96
137
|
export declare function workerTreeCpuMs(marker: string, cwd: string): Promise<{
|
|
97
138
|
ms: number;
|
|
98
139
|
resolutionMs: number;
|
|
99
140
|
} | undefined>;
|
|
141
|
+
/** Test seam — exercise the production observer's total read bound with a small real tree. */
|
|
142
|
+
export declare function setObserveBudgetBytesForTests(bytes: number): void;
|
|
143
|
+
export declare function resetObserveBudgetBytesForTests(): void;
|
|
100
144
|
export declare function runDaemon(repoRoot: string, opts?: RunOptions): Promise<RunSummary>;
|