tickmarkr 1.81.0 → 1.84.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -4
- package/dist/brand.d.ts +4 -1
- package/dist/brand.js +46 -7
- package/dist/cli/commands/approve.js +53 -8
- package/dist/cli/commands/plan.js +17 -2
- package/dist/cli/commands/report.js +15 -6
- package/dist/cli/commands/ui.js +13 -1
- package/dist/cli/index.d.ts +1 -1
- package/dist/cli/index.js +1 -1
- package/dist/compile/collateral.d.ts +19 -0
- package/dist/compile/collateral.js +90 -0
- package/dist/compile/index.js +18 -4
- package/dist/drivers/herdr.d.ts +1 -0
- package/dist/drivers/herdr.js +19 -0
- package/dist/drivers/types.d.ts +1 -0
- package/dist/gates/acceptance.js +50 -8
- package/dist/gates/llm.js +34 -19
- package/dist/gates/review.d.ts +9 -1
- package/dist/gates/review.js +173 -7
- package/dist/gates/run-gates.d.ts +1 -0
- package/dist/gates/run-gates.js +18 -1
- package/dist/report/bundle.d.ts +6 -1
- package/dist/report/bundle.js +13 -3
- package/dist/run/daemon.d.ts +5 -0
- package/dist/run/daemon.js +188 -27
- package/dist/run/journal.d.ts +5 -0
- package/dist/run/journal.js +71 -1
- package/dist/tui/cockpit/capture.d.ts +107 -1
- package/dist/tui/cockpit/capture.js +287 -10
- package/dist/tui/cockpit/components.d.ts +40 -47
- package/dist/tui/cockpit/components.js +66 -50
- package/dist/tui/cockpit/derive.d.ts +73 -1
- package/dist/tui/cockpit/derive.js +211 -15
- package/dist/tui/cockpit/keys.d.ts +114 -19
- package/dist/tui/cockpit/keys.js +413 -44
- package/dist/tui/cockpit/layout.d.ts +231 -0
- package/dist/tui/cockpit/layout.js +307 -0
- package/dist/tui/cockpit/live.d.ts +65 -3
- package/dist/tui/cockpit/live.js +418 -48
- package/dist/tui/cockpit/pointer.d.ts +261 -0
- package/dist/tui/cockpit/pointer.js +610 -0
- package/dist/tui/cockpit/run-cockpit.d.ts +93 -6
- package/dist/tui/cockpit/run-cockpit.js +805 -164
- package/dist/tui/cockpit/setup-cockpit.d.ts +146 -1
- package/dist/tui/cockpit/setup-cockpit.js +337 -5
- package/dist/tui/cockpit/views.d.ts +72 -0
- package/dist/tui/cockpit/views.js +56 -0
- package/dist/tui/cockpit/width.d.ts +105 -0
- package/dist/tui/cockpit/width.js +276 -0
- package/fixtures/speckit-sample/tasks.md +2 -2
- package/package.json +3 -1
package/README.md
CHANGED
package/dist/brand.d.ts
CHANGED
|
@@ -1,4 +1,7 @@
|
|
|
1
1
|
import type { OwnedName } from "./drivers/types.js";
|
|
2
|
+
export declare const MARK_BITMAP: readonly ["..................", "..............###.", "............####..", "..........####....", "........####......", ".###..####........", "...#####..........", "....###...........", "..................", ".................."];
|
|
3
|
+
export declare const PLAIN_MARK: string;
|
|
4
|
+
export declare const MARK: string;
|
|
2
5
|
export declare const BANNER: string;
|
|
3
6
|
/** ANSI-stripped, trailing-space-trimmed twin of BANNER — README hero and other plain surfaces. */
|
|
4
7
|
export declare const PLAIN_BANNER: string;
|
|
@@ -15,7 +18,7 @@ export declare const TICKMARKR_EXIT_TRAILER = "printf '\\nTICKMARKR_''EXIT:%s\\n
|
|
|
15
18
|
export declare function paneDispatchScript(body: string[]): string;
|
|
16
19
|
/** OBS-50: one short herdr pane-run line; bootstrap lives in the script file beside the prompt. */
|
|
17
20
|
export declare function paneDispatchCommand(scriptPath: string): string;
|
|
18
|
-
/** The settled brand green ramp (256-color), bright → deep
|
|
21
|
+
/** The settled brand green ramp (256-color), bright → deep. */
|
|
19
22
|
export declare const BRAND_RAMP: readonly [84, 78, 41, 35];
|
|
20
23
|
/** Brand green (ramp anchor 41) — the tickmark hue; also the ok/pass/authed verdict color. */
|
|
21
24
|
export declare const brand: (s: string) => string;
|
package/dist/brand.js
CHANGED
|
@@ -1,13 +1,52 @@
|
|
|
1
1
|
import { shq } from "./adapters/types.js";
|
|
2
|
-
// TTY-only pixel-tick logo (assets/mark.svg is the image twin
|
|
2
|
+
// TTY-only pixel-tick logo (assets/mark.svg is generated as the image twin of this bitmap).
|
|
3
3
|
// Never printed to pipes — the non-TTY stdout surface is byte-pinned by tests and consumed by machines.
|
|
4
4
|
const B = "\x1b[1m", R = "\x1b[0m";
|
|
5
|
-
const
|
|
5
|
+
export const MARK_BITMAP = [
|
|
6
|
+
"..................",
|
|
7
|
+
"..............###.",
|
|
8
|
+
"............####..",
|
|
9
|
+
"..........####....",
|
|
10
|
+
"........####......",
|
|
11
|
+
".###..####........",
|
|
12
|
+
"...#####..........",
|
|
13
|
+
"....###...........",
|
|
14
|
+
"..................",
|
|
15
|
+
"..................",
|
|
16
|
+
];
|
|
17
|
+
const QUADRANTS = [
|
|
18
|
+
" ", "▘", "▝", "▀", "▖", "▌", "▞", "▛",
|
|
19
|
+
"▗", "▚", "▐", "▜", "▄", "▙", "▟", "█",
|
|
20
|
+
];
|
|
21
|
+
function packMark(bitmap) {
|
|
22
|
+
const rows = [];
|
|
23
|
+
for (let y = 0; y < bitmap.length; y += 2) {
|
|
24
|
+
let row = "";
|
|
25
|
+
for (let x = 0; x < bitmap[y].length; x += 2) {
|
|
26
|
+
const mask = (bitmap[y][x] === "#" ? 1 : 0)
|
|
27
|
+
| (bitmap[y][x + 1] === "#" ? 2 : 0)
|
|
28
|
+
| (bitmap[y + 1][x] === "#" ? 4 : 0)
|
|
29
|
+
| (bitmap[y + 1][x + 1] === "#" ? 8 : 0);
|
|
30
|
+
row += QUADRANTS[mask];
|
|
31
|
+
}
|
|
32
|
+
rows.push(row);
|
|
33
|
+
}
|
|
34
|
+
return rows.join("\n");
|
|
35
|
+
}
|
|
36
|
+
// The ruled silhouette, packed from the 18x10 bitmap into a 9x5 quadrant-cell grid.
|
|
37
|
+
export const PLAIN_MARK = packMark(MARK_BITMAP);
|
|
38
|
+
// Near-black ANSI 233 knockout ink on the solid ANSI 41 brand tile.
|
|
39
|
+
export const MARK = PLAIN_MARK.split("\n")
|
|
40
|
+
.map((row) => `\x1b[38;5;233;48;5;41m${row}${R}`)
|
|
41
|
+
.join("\n");
|
|
42
|
+
// Retain the reviewed 28-column compact row so narrow-header copy wraps unchanged.
|
|
43
|
+
const BANNER_COPY_GAP = " ".repeat(10);
|
|
6
44
|
export const BANNER = [
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
45
|
+
// The fifth packed row is tile-only padding; the composed header stays four rows
|
|
46
|
+
// so setup retains every detected state at the contracted 24-row height.
|
|
47
|
+
...MARK.split("\n").slice(0, -1).map((row, index) => index === 2 ? `${row}${BANNER_COPY_GAP}${B}tickmarkr${R}`
|
|
48
|
+
: index === 3 ? `${row}${BANNER_COPY_GAP}spec in, verified work out.`
|
|
49
|
+
: row),
|
|
11
50
|
"",
|
|
12
51
|
].join("\n");
|
|
13
52
|
/** ANSI-stripped, trailing-space-trimmed twin of BANNER — README hero and other plain surfaces. */
|
|
@@ -47,7 +86,7 @@ export function paneDispatchCommand(scriptPath) {
|
|
|
47
86
|
// Every cockpit surface styles through these tokens/glyphs/helpers. Styled only
|
|
48
87
|
// on a real TTY with NO_COLOR unset; otherwise output is the plain text itself
|
|
49
88
|
// (non-TTY surfaces stay byte-pinned and machine-consumable).
|
|
50
|
-
/** The settled brand green ramp (256-color), bright → deep
|
|
89
|
+
/** The settled brand green ramp (256-color), bright → deep. */
|
|
51
90
|
export const BRAND_RAMP = [84, 78, 41, 35];
|
|
52
91
|
const visual = () => process.stdout.isTTY === true && process.env.NO_COLOR === undefined;
|
|
53
92
|
const sgr = (code) => (s) => visual() ? `\x1b[${code}m${s}${R}` : s;
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { userInfo } from "node:os";
|
|
2
2
|
import { GATE_NAMES } from "../../graph/schema.js";
|
|
3
|
-
import { ATTEMPT_CAP_RELEASE, GATE_SATISFIED_RELEASE, Journal } from "../../run/journal.js";
|
|
3
|
+
import { ATTEMPT_CAP_RELEASE, GATE_SATISFIED_RELEASE, Journal, RECHECK_RELEASE, REVIEW_UPHELD_RELEASE } from "../../run/journal.js";
|
|
4
4
|
// GATE-08 (v1.12): approve a parked human gate so the next `tickmarkr resume <runId>` dispatches it.
|
|
5
5
|
//
|
|
6
6
|
// The approval is a JOURNAL EVENT (task-approved) carrying who and when — it touches ONLY the
|
|
@@ -18,8 +18,13 @@ import { ATTEMPT_CAP_RELEASE, GATE_SATISFIED_RELEASE, Journal } from "../../run/
|
|
|
18
18
|
//
|
|
19
19
|
// Who/when is truthful, not dressed-up auth (D-03): default actor os.userInfo().username; --by overrides
|
|
20
20
|
// for delegated approval; optional --reason; the event's ts (stamped by Journal.append) is the when.
|
|
21
|
+
// OBS-189: `--uphold` is the second decision a review park offers. Plain approve accepts the diff the
|
|
22
|
+
// reviewer rejected (gate-satisfied); --uphold sides WITH the reviewer and funds ONE fixed worker
|
|
23
|
+
// attempt carrying the findings — the park costs an attempt, never the run.
|
|
21
24
|
export async function approve(argv, cwd = process.cwd()) {
|
|
22
|
-
const { runId, taskId, by, reason } = parseArgs(argv);
|
|
25
|
+
const { runId, taskId, by, reason, uphold, recheck } = parseArgs(argv);
|
|
26
|
+
if (uphold && recheck)
|
|
27
|
+
throw new Error("--uphold and --recheck are different decisions — pass one");
|
|
23
28
|
// Journal.open throws `no journal for <runId> at <dir>` on an unknown run — that IS the refusal.
|
|
24
29
|
const journal = Journal.open(cwd, runId);
|
|
25
30
|
const status = journal.replayStatuses().get(taskId);
|
|
@@ -41,8 +46,39 @@ export async function approve(argv, cwd = process.cwd()) {
|
|
|
41
46
|
}
|
|
42
47
|
}
|
|
43
48
|
const lastHuman = events[lastHumanIndex];
|
|
49
|
+
if (uphold) {
|
|
50
|
+
// Fail-closed on the DATA: uphold applies only when the newest failed gate is the review gate —
|
|
51
|
+
// any other gate has no reviewer to uphold. Never inferred from the park's prose.
|
|
52
|
+
const lastFailed = events.slice(0, lastHumanIndex).reverse().find((e) => e.event === "gate-result" && e.taskId === taskId && e.data.pass === false
|
|
53
|
+
&& typeof e.data.gate === "string" && GATE_NAMES.includes(e.data.gate))?.data.gate;
|
|
54
|
+
if (lastFailed !== "review") {
|
|
55
|
+
throw new Error(`--uphold applies to a review rejection; ${taskId}'s last failed gate is ${lastFailed ?? "none"} — refusing`);
|
|
56
|
+
}
|
|
57
|
+
journal.append("task-approved", taskId, {
|
|
58
|
+
by,
|
|
59
|
+
...(reason ? { reason } : {}),
|
|
60
|
+
via: "cli",
|
|
61
|
+
release: REVIEW_UPHELD_RELEASE,
|
|
62
|
+
gate: "review",
|
|
63
|
+
});
|
|
64
|
+
return `upheld the reviewer for ${taskId} in ${runId} — by ${by}; run \`tickmarkr resume ${runId}\` to dispatch a fixed attempt carrying the findings`;
|
|
65
|
+
}
|
|
44
66
|
const capPark = lastHuman?.data.kind === ATTEMPT_CAP_RELEASE;
|
|
45
67
|
const gateFailPark = lastHuman?.data.kind === "gate-fail";
|
|
68
|
+
if (recheck) {
|
|
69
|
+
// OBS-203: fail-closed on the PARK KIND — only a gate-fail park has a gate to re-run. Refusing
|
|
70
|
+
// elsewhere keeps --recheck from becoming a silent budget reset on a pre-dispatch human gate.
|
|
71
|
+
if (!gateFailPark) {
|
|
72
|
+
throw new Error(`--recheck applies to a gate-fail park; ${taskId}'s park kind is ${lastHuman?.data.kind ?? "none"} — refusing`);
|
|
73
|
+
}
|
|
74
|
+
journal.append("task-approved", taskId, {
|
|
75
|
+
by,
|
|
76
|
+
...(reason ? { reason } : {}),
|
|
77
|
+
via: "cli",
|
|
78
|
+
release: RECHECK_RELEASE,
|
|
79
|
+
});
|
|
80
|
+
return `re-checking ${taskId} in ${runId} — by ${by}; no gate marked satisfied, run \`tickmarkr resume ${runId}\` to re-dispatch against the full gate suite`;
|
|
81
|
+
}
|
|
46
82
|
const failedGate = gateFailPark
|
|
47
83
|
? events.slice(0, lastHumanIndex).reverse().find((e) => e.event === "gate-result" && e.taskId === taskId && e.data.pass === false
|
|
48
84
|
&& typeof e.data.gate === "string" && GATE_NAMES.includes(e.data.gate))?.data.gate
|
|
@@ -59,23 +95,32 @@ export async function approve(argv, cwd = process.cwd()) {
|
|
|
59
95
|
});
|
|
60
96
|
return `approved ${taskId} in ${runId} — by ${by}; run \`tickmarkr resume ${runId}\` to ${failedGate ? "continue past the approved gate" : "dispatch it"}`;
|
|
61
97
|
}
|
|
62
|
-
|
|
63
|
-
//
|
|
98
|
+
const USAGE = "usage: tickmarkr approve <run-id> <task-id> [--uphold|--recheck] [--by <name>] [--reason <text>]";
|
|
99
|
+
// hand-parsed argv — no CLI framework (house style). Flags --uphold, --by <name>, --reason <text>;
|
|
100
|
+
// positionals are runId then taskId. Throws usage on missing positionals (mirrors resume.ts/unlock.ts).
|
|
64
101
|
function parseArgs(argv) {
|
|
65
102
|
const positionals = [];
|
|
66
103
|
let by;
|
|
67
104
|
let reason;
|
|
105
|
+
let uphold = false;
|
|
106
|
+
let recheck = false;
|
|
68
107
|
for (let i = 0; i < argv.length; i++) {
|
|
69
108
|
const a = argv[i];
|
|
70
109
|
if (a === "--by") {
|
|
71
110
|
by = argv[++i];
|
|
72
111
|
if (!by)
|
|
73
|
-
throw new Error(
|
|
112
|
+
throw new Error(USAGE);
|
|
74
113
|
}
|
|
75
114
|
else if (a === "--reason") {
|
|
76
115
|
reason = argv[++i];
|
|
77
116
|
if (!reason)
|
|
78
|
-
throw new Error(
|
|
117
|
+
throw new Error(USAGE);
|
|
118
|
+
}
|
|
119
|
+
else if (a === "--uphold") {
|
|
120
|
+
uphold = true;
|
|
121
|
+
}
|
|
122
|
+
else if (a === "--recheck") {
|
|
123
|
+
recheck = true;
|
|
79
124
|
}
|
|
80
125
|
else {
|
|
81
126
|
positionals.push(a);
|
|
@@ -83,7 +128,7 @@ function parseArgs(argv) {
|
|
|
83
128
|
}
|
|
84
129
|
const [runId, taskId] = positionals;
|
|
85
130
|
if (!runId || !taskId) {
|
|
86
|
-
throw new Error(
|
|
131
|
+
throw new Error(USAGE);
|
|
87
132
|
}
|
|
88
|
-
return { runId, taskId, by: by ?? userInfo().username, reason };
|
|
133
|
+
return { runId, taskId, by: by ?? userInfo().username, reason, uphold, recheck };
|
|
89
134
|
}
|
|
@@ -51,9 +51,14 @@ export async function plan(argv, cwd = process.cwd(), adapters = allAdapters())
|
|
|
51
51
|
const health = cached ?? (await probeAll(adapters));
|
|
52
52
|
const channels = discoverChannels(cfg, adapters, health);
|
|
53
53
|
// VIS-04 trust ramp (VALIDATION 13-01-11): preview:true bypasses the routing.learned:off short-circuit so
|
|
54
|
-
// the static-vs-learned
|
|
54
|
+
// the static-vs-learned comparison renders even when the daemon's learned routing is off. Cold ⇒ undefined ⇒
|
|
55
55
|
// output byte-identical to today.
|
|
56
56
|
const profile = loadRoutingProfile(cwd, cfg, { preview: true });
|
|
57
|
+
// T10: the headline row names the channel the run would actually DISPATCH. The daemon builds its profile
|
|
58
|
+
// without preview, so routing.learned: off means it dispatches the static pick — the preview profile may
|
|
59
|
+
// only feed the "learned would pick" comparison line below, never the row itself. Scoped to the switch:
|
|
60
|
+
// with learned on, dispatchProfile === profile and the surface is byte-identical to before.
|
|
61
|
+
const dispatchProfile = cfg.routing.learned === "off" ? undefined : profile;
|
|
57
62
|
let deviations = 0;
|
|
58
63
|
// v1.51 T4: the mode is never invisible — the header names the resolved mode, its winning
|
|
59
64
|
// source, and the explore posture; each task row carries a floor-derivation line below.
|
|
@@ -129,7 +134,7 @@ export async function plan(argv, cwd = process.cwd(), adapters = allAdapters())
|
|
|
129
134
|
const routed = [];
|
|
130
135
|
for (const t of g.tasks) {
|
|
131
136
|
try {
|
|
132
|
-
const r = route(t, cfg, channels,
|
|
137
|
+
const r = route(t, cfg, channels, dispatchProfile);
|
|
133
138
|
routed.push({ taskId: t.id, adapter: r.assignment.adapter, model: r.assignment.model });
|
|
134
139
|
for (const l of r.lints) {
|
|
135
140
|
const suf = exclusionReason(l);
|
|
@@ -146,6 +151,16 @@ export async function plan(argv, cwd = process.cwd(), adapters = allAdapters())
|
|
|
146
151
|
const d = r.deviation;
|
|
147
152
|
lines.push(` ⇄ static would pick ${d.static} — learned picked ${d.chosen} (score ${d.score.toFixed(3)} n=${d.n} vs ${d.staticScore.toFixed(3)})`);
|
|
148
153
|
}
|
|
154
|
+
else if (dispatchProfile !== profile && profile) {
|
|
155
|
+
// learned off: the row above already names the static dispatch pick; the preview keeps its
|
|
156
|
+
// value by naming what learned routing WOULD have picked instead (same scores, same shape).
|
|
157
|
+
const pv = route(t, cfg, channels, profile);
|
|
158
|
+
if (pv.deviation) {
|
|
159
|
+
deviations++;
|
|
160
|
+
const d = pv.deviation;
|
|
161
|
+
lines.push(` ⇄ learned would pick ${d.chosen} (score ${d.score.toFixed(3)} n=${d.n} vs ${d.staticScore.toFixed(3)}) — routing.learned: off keeps the static pick`);
|
|
162
|
+
}
|
|
163
|
+
}
|
|
149
164
|
}
|
|
150
165
|
catch (e) {
|
|
151
166
|
if (!(e instanceof RoutingError))
|
|
@@ -4,7 +4,7 @@ import { ttyVisual } from "../../adapters/model-lints.js";
|
|
|
4
4
|
import { addUsage } from "../../adapters/types.js";
|
|
5
5
|
import { dim, rule, title } from "../../brand.js";
|
|
6
6
|
import { loadConfig } from "../../config/config.js";
|
|
7
|
-
import { buildProofBundle } from "../../report/bundle.js";
|
|
7
|
+
import { buildProofBundle, gateDeclined } from "../../report/bundle.js";
|
|
8
8
|
import { compareRuns } from "../../report/compare.js";
|
|
9
9
|
import { estimateCosts } from "../../report/cost.js";
|
|
10
10
|
import { cellsOf, cellSummary } from "../../route/profile.js";
|
|
@@ -97,6 +97,11 @@ const wallClock = (start, end) => {
|
|
|
97
97
|
return minutes ? `${minutes}m ${seconds % 60}s` : `${seconds}s`;
|
|
98
98
|
};
|
|
99
99
|
const detail = (value) => typeof value === "string" || typeof value === "number" ? String(value) : EM;
|
|
100
|
+
// T11: a gate that declined never ran. Drop those before the comparison metrics so the pass rate
|
|
101
|
+
// counts only gates that actually ran — a declined gate is never a pass and never pads the base.
|
|
102
|
+
// Declines are pass:true, so the gate-failure count is untouched. Reporting only: the unfiltered
|
|
103
|
+
// events still feed the record and the proof bundle, and no gate outcome changes.
|
|
104
|
+
const ranGatesOnly = (events) => events.filter((e) => e.event !== "gate-result" || !gateDeclined(e.data));
|
|
100
105
|
// v1.53 T5: supersession is derived from the run's OWN journal only — `superseded by` from the
|
|
101
106
|
// appended superseded event (last wins), `supersedes` from the run-start stamp. No cross-run scan.
|
|
102
107
|
const supersession = (events) => {
|
|
@@ -212,7 +217,8 @@ export function renderMarkdownRecord(runId, events, prices = [], rows = []) {
|
|
|
212
217
|
lines.push("- **tickmarks:**");
|
|
213
218
|
if (gates.length) {
|
|
214
219
|
for (const g of gates) {
|
|
215
|
-
|
|
220
|
+
// T11: a declined gate never ran — report it as declined, never as a pass.
|
|
221
|
+
const pass = gateDeclined(g.data) ? "declined" : g.data.pass === true ? "pass" : g.data.pass === false ? "fail" : EM;
|
|
216
222
|
const gate = typeof g.data.gate === "string" ? g.data.gate : EM;
|
|
217
223
|
lines.push(` - ${gate}: ${pass} — ${firstLine(g.data.details)}`);
|
|
218
224
|
}
|
|
@@ -279,7 +285,10 @@ function textReport(runId, events, rows, cwd) {
|
|
|
279
285
|
return ` ${shape.padEnd(10)} ${chKey.padEnd(28)} ${channel.padEnd(4)} raw=${s.nRaw} n_eff=${s.nEff} disp=${s.dispatches} q=${s.quality === undefined ? "-" : s.quality.toFixed(2)} quota=${s.quotaHits} explore-left=${s.exploreRemaining}${s.cold ? " cold (neutral)" : ""}`;
|
|
280
286
|
});
|
|
281
287
|
const gateResults = events.filter((e) => e.event === "gate-result");
|
|
282
|
-
|
|
288
|
+
// T11: gates that declined never ran — out of both the pass count and the rate base.
|
|
289
|
+
const declinedCount = gateResults.filter((e) => gateDeclined(e.data)).length;
|
|
290
|
+
const gateRan = gateResults.length - declinedCount;
|
|
291
|
+
const gatePass = gateResults.filter((e) => e.data.pass && !gateDeclined(e.data)).length;
|
|
283
292
|
const escalations = events.filter((e) => e.event === "escalation").length;
|
|
284
293
|
const consults = events.filter((e) => e.event === "consult-verdict").length;
|
|
285
294
|
const failovers = events.filter((e) => e.event === "quota-failover").length;
|
|
@@ -292,7 +301,7 @@ function textReport(runId, events, rows, cwd) {
|
|
|
292
301
|
"engagement summary — audit trail:",
|
|
293
302
|
...[...groups.entries()].map(([k, g]) => ` ${k.padEnd(30)} tasks ${g.rows.length}, attempts ${g.rows.reduce((s, r) => s + r.attempts, 0)}, done ${g.rows.filter((r) => r.outcome === "done").length}`),
|
|
294
303
|
"",
|
|
295
|
-
`tickmark rate: ${
|
|
304
|
+
`tickmark rate: ${gateRan ? Math.round((100 * gatePass) / gateRan) : 0}% (${gatePass}/${gateRan})${declinedCount ? ` · declined: ${declinedCount}` : ""}`,
|
|
296
305
|
`escalations: ${escalations} · National Office consults: ${consults} · quota failovers: ${failovers}`,
|
|
297
306
|
"",
|
|
298
307
|
"spend — tokens (measured where observed):",
|
|
@@ -349,8 +358,8 @@ export async function report(argv, cwd = process.cwd()) {
|
|
|
349
358
|
const outcome = compareRuns({
|
|
350
359
|
runId,
|
|
351
360
|
baselineRunId,
|
|
352
|
-
events,
|
|
353
|
-
baselineEvents: baseline.read(),
|
|
361
|
+
events: ranGatesOnly(events),
|
|
362
|
+
baselineEvents: ranGatesOnly(baseline.read()),
|
|
354
363
|
rows,
|
|
355
364
|
baselineRows: baseline.readTelemetry(),
|
|
356
365
|
cost: cfg.cost,
|
package/dist/cli/commands/ui.js
CHANGED
|
@@ -1,4 +1,10 @@
|
|
|
1
1
|
const NON_TTY_MSG = "tickmarkr ui: the cockpit requires a TTY — use `tickmarkr fleet --print` or `tickmarkr status --watch` for line-mode output";
|
|
2
|
+
/**
|
|
3
|
+
* The setup surface's own flag. `tickmarkr ui --setup [run-id]` opens the one
|
|
4
|
+
* surface that writes — parked human decisions, approved or upheld through the
|
|
5
|
+
* production command path behind an explicit confirm.
|
|
6
|
+
*/
|
|
7
|
+
const SETUP_FLAG = "--setup";
|
|
2
8
|
export async function ui(argv, io = {}, cwd = process.cwd()) {
|
|
3
9
|
const input = io.input ?? process.stdin;
|
|
4
10
|
const output = io.output ?? process.stdout;
|
|
@@ -15,7 +21,8 @@ export async function ui(argv, io = {}, cwd = process.cwd()) {
|
|
|
15
21
|
});
|
|
16
22
|
return "ui: closed";
|
|
17
23
|
}
|
|
18
|
-
const
|
|
24
|
+
const setup = argv.includes(SETUP_FLAG);
|
|
25
|
+
const unknownFlag = argv.find((arg) => arg.startsWith("-") && arg !== SETUP_FLAG);
|
|
19
26
|
if (unknownFlag) {
|
|
20
27
|
return { out: `tickmarkr ui: unknown flag ${unknownFlag}`, code: 1 };
|
|
21
28
|
}
|
|
@@ -47,6 +54,11 @@ export async function ui(argv, io = {}, cwd = process.cwd()) {
|
|
|
47
54
|
code: 1,
|
|
48
55
|
};
|
|
49
56
|
}
|
|
57
|
+
if (setup) {
|
|
58
|
+
const { runSetupDecisionsCockpit } = await import("../../tui/cockpit/setup-cockpit.js");
|
|
59
|
+
await runSetupDecisionsCockpit({ input, output, cwd, runId });
|
|
60
|
+
return "ui: closed";
|
|
61
|
+
}
|
|
50
62
|
await live.runLiveCockpit({ input, output, cwd, runId, binaryVersion });
|
|
51
63
|
return "ui: closed";
|
|
52
64
|
}
|
package/dist/cli/index.d.ts
CHANGED
|
@@ -5,7 +5,7 @@ export type CommandResult = string | {
|
|
|
5
5
|
};
|
|
6
6
|
export type CommandMap = Record<string, (argv: string[]) => Promise<CommandResult>>;
|
|
7
7
|
export declare const COMMANDS: CommandMap;
|
|
8
|
-
export declare const USAGE = "tickmarkr \u2014 spec-driven orchestration harness for AI coding agents\nusage: tickmarkr <command>\n init guided setup + doctor; init --agent [--force] [--docs] adds agent skills/docs\n doctor re-probe adapters, herdr, auth; print capability matrix\n fleet interactive fleet editor (fleet --print for CI drift checks)\n compile <src> spec \u2192 .tickmarkr/graph.json (fails without acceptance criteria)\n scope <intent> draft a compiled native spec beside an answered intent (--force to overwrite)\n plan dry-run routing table + cost estimate + floor lints\n eval run checked-in fixtures against every channel in isolated temp repos\n run execute the graph (--concurrency N --driver herdr|subprocess --route-strict)\n status live run state\n resume <id> continue a run from its journal\n report <id> cost/quality report (--md for committable execution record)\n profile show learned routing profile (profile reset = forget history via cursor, keeps telemetry)\n ui open the Fleet Studio TUI (full-screen tabbed cockpit)\n unlock remove a stale/garbage run lock (refuses if the holder is alive)\n approve <id> <task>
|
|
8
|
+
export declare const USAGE = "tickmarkr \u2014 spec-driven orchestration harness for AI coding agents\nusage: tickmarkr <command>\n init guided setup + doctor; init --agent [--force] [--docs] adds agent skills/docs\n doctor re-probe adapters, herdr, auth; print capability matrix\n fleet interactive fleet editor (fleet --print for CI drift checks)\n compile <src> spec \u2192 .tickmarkr/graph.json (fails without acceptance criteria)\n scope <intent> draft a compiled native spec beside an answered intent (--force to overwrite)\n plan dry-run routing table + cost estimate + floor lints\n eval run checked-in fixtures against every channel in isolated temp repos\n run execute the graph (--concurrency N --driver herdr|subprocess --route-strict)\n status live run state\n resume <id> continue a run from its journal\n report <id> cost/quality report (--md for committable execution record)\n profile show learned routing profile (profile reset = forget history via cursor, keeps telemetry)\n ui open the Fleet Studio TUI (full-screen tabbed cockpit)\n unlock remove a stale/garbage run lock (refuses if the holder is alive)\n approve <id> <task> release a park (--uphold sides with the reviewer and funds a fixed attempt; --by <name> --reason <text>); takes effect on resume";
|
|
9
9
|
export declare function dispatch(cmd: string | undefined, argv: string[], commands?: CommandMap): Promise<{
|
|
10
10
|
out: string;
|
|
11
11
|
code: number;
|
package/dist/cli/index.js
CHANGED
|
@@ -40,7 +40,7 @@ usage: tickmarkr <command>
|
|
|
40
40
|
profile show learned routing profile (profile reset = forget history via cursor, keeps telemetry)
|
|
41
41
|
ui open the Fleet Studio TUI (full-screen tabbed cockpit)
|
|
42
42
|
unlock remove a stale/garbage run lock (refuses if the holder is alive)
|
|
43
|
-
approve <id> <task>
|
|
43
|
+
approve <id> <task> release a park (--uphold sides with the reviewer and funds a fixed attempt; --by <name> --reason <text>); takes effect on resume`;
|
|
44
44
|
// pure, testable dispatcher: resolves a command, forwards argv, shapes the result — no side effects.
|
|
45
45
|
// unknown/missing cmd → USAGE (exit 1 if a cmd was typed, 0 for bare `tickmarkr`); a handler throw becomes
|
|
46
46
|
// a one-line `tickmarkr <cmd>: <message>` (never a raw stack) at exit 1.
|
|
@@ -11,3 +11,22 @@ export declare function newDirectoryLints(tasks: ReadonlyArray<Pick<Task, "id" |
|
|
|
11
11
|
* Advisory plan output only, same contract as collateralLints.
|
|
12
12
|
*/
|
|
13
13
|
export declare function sourceScopeLints(tasks: ReadonlyArray<Pick<Task, "id" | "files" | "acceptance">>, repoRoot: string): string[];
|
|
14
|
+
/** Max acceptance items per task. Consult-converged (copus/csol/cfable R2). */
|
|
15
|
+
export declare const MAX_ACCEPTANCE_ITEMS = 6;
|
|
16
|
+
/** Max files[] patterns per task. */
|
|
17
|
+
export declare const MAX_FILES_PATTERNS = 8;
|
|
18
|
+
/**
|
|
19
|
+
* OBS-212: two tasks with no dependency path between them may run concurrently and are recreated
|
|
20
|
+
* onto a moving integration tip. If they write the same file, the graph's independence claim is
|
|
21
|
+
* false — and it comes due in the carry plumbing, which resets the task branch to the advanced tip
|
|
22
|
+
* and silently drops whatever will not cherry-pick. run-20260728-110135 lost 32 verified commits in
|
|
23
|
+
* two events that way (T2 17/17, T1 15/15), each after a sibling that shared its files merged.
|
|
24
|
+
*/
|
|
25
|
+
export declare function separabilityErrors(tasks: ReadonlyArray<Pick<Task, "id" | "files" | "deps">>): string[];
|
|
26
|
+
/**
|
|
27
|
+
* OBS-214: a task too large to converge is not a task. T1 of v1.83 carried 8 acceptance items and a
|
|
28
|
+
* diff at the 130k cap; it took 28 dispatches and never passed review, while T3 (4 items) took 5.
|
|
29
|
+
*/
|
|
30
|
+
export declare function taskBudgetErrors(tasks: ReadonlyArray<Pick<Task, "id" | "files" | "acceptance">>): string[];
|
|
31
|
+
/** Every Task Unit Contract violation in one pass, ready to throw. */
|
|
32
|
+
export declare function taskUnitContractErrors(tasks: ReadonlyArray<Pick<Task, "id" | "files" | "deps" | "acceptance">>): string[];
|
|
@@ -234,3 +234,93 @@ export function sourceScopeLints(tasks, repoRoot) {
|
|
|
234
234
|
}
|
|
235
235
|
return [...newDirLints, ...lines];
|
|
236
236
|
}
|
|
237
|
+
// ── Task Unit Contract (OBS-212 / OBS-214) ────────────────────────────────────────────────────
|
|
238
|
+
// These are ERRORS, not advisories. Everything above this line is a plan-time warning the author
|
|
239
|
+
// may ignore; a graph that violates the rules below is not a graph the harness can run honestly.
|
|
240
|
+
/** Max acceptance items per task. Consult-converged (copus/csol/cfable R2). */
|
|
241
|
+
export const MAX_ACCEPTANCE_ITEMS = 6;
|
|
242
|
+
/** Max files[] patterns per task. */
|
|
243
|
+
export const MAX_FILES_PATTERNS = 8;
|
|
244
|
+
/** Tasks reachable from `id` through deps (transitive). */
|
|
245
|
+
function reachable(id, byId) {
|
|
246
|
+
const seen = new Set();
|
|
247
|
+
const stack = [...(byId.get(id) ?? [])];
|
|
248
|
+
while (stack.length) {
|
|
249
|
+
const next = stack.pop();
|
|
250
|
+
if (seen.has(next))
|
|
251
|
+
continue;
|
|
252
|
+
seen.add(next);
|
|
253
|
+
stack.push(...(byId.get(next) ?? []));
|
|
254
|
+
}
|
|
255
|
+
return seen;
|
|
256
|
+
}
|
|
257
|
+
// Conservative overlap: identical patterns, or a glob that matches the other's literal path. Two
|
|
258
|
+
// globs that merely COULD intersect are not flagged — false negatives are acceptable here, false
|
|
259
|
+
// positives would reject legitimate graphs.
|
|
260
|
+
function overlappingPatterns(a, b) {
|
|
261
|
+
const hits = new Set();
|
|
262
|
+
const isGlob = (p) => /[*?[\]{}]/.test(p);
|
|
263
|
+
for (const pa of a) {
|
|
264
|
+
for (const pb of b) {
|
|
265
|
+
if (pa === pb) {
|
|
266
|
+
hits.add(pa);
|
|
267
|
+
continue;
|
|
268
|
+
}
|
|
269
|
+
if (isGlob(pa) && !isGlob(pb) && picomatch(pa)(pb))
|
|
270
|
+
hits.add(`${pb} ⊂ ${pa}`);
|
|
271
|
+
else if (isGlob(pb) && !isGlob(pa) && picomatch(pb)(pa))
|
|
272
|
+
hits.add(`${pa} ⊂ ${pb}`);
|
|
273
|
+
}
|
|
274
|
+
}
|
|
275
|
+
return [...hits].sort();
|
|
276
|
+
}
|
|
277
|
+
/**
|
|
278
|
+
* OBS-212: two tasks with no dependency path between them may run concurrently and are recreated
|
|
279
|
+
* onto a moving integration tip. If they write the same file, the graph's independence claim is
|
|
280
|
+
* false — and it comes due in the carry plumbing, which resets the task branch to the advanced tip
|
|
281
|
+
* and silently drops whatever will not cherry-pick. run-20260728-110135 lost 32 verified commits in
|
|
282
|
+
* two events that way (T2 17/17, T1 15/15), each after a sibling that shared its files merged.
|
|
283
|
+
*/
|
|
284
|
+
export function separabilityErrors(tasks) {
|
|
285
|
+
const byId = new Map(tasks.map((t) => [t.id, t.deps ?? []]));
|
|
286
|
+
const errors = [];
|
|
287
|
+
for (let i = 0; i < tasks.length; i++) {
|
|
288
|
+
for (let j = i + 1; j < tasks.length; j++) {
|
|
289
|
+
const a = tasks[i], b = tasks[j];
|
|
290
|
+
if (reachable(a.id, byId).has(b.id) || reachable(b.id, byId).has(a.id))
|
|
291
|
+
continue; // ordered
|
|
292
|
+
const shared = overlappingPatterns(a.files ?? [], b.files ?? []);
|
|
293
|
+
if (shared.length > 0) {
|
|
294
|
+
errors.push(`${a.id} and ${b.id} both write ${shared.join(", ")} but neither depends on the other — `
|
|
295
|
+
+ `add a dependency edge to order them, or split the shared path out. Concurrent tasks that `
|
|
296
|
+
+ `write the same file are not independent, and the loser's committed work is silently `
|
|
297
|
+
+ `dropped when the integration tip advances (OBS-212).`);
|
|
298
|
+
}
|
|
299
|
+
}
|
|
300
|
+
}
|
|
301
|
+
return errors;
|
|
302
|
+
}
|
|
303
|
+
/**
|
|
304
|
+
* OBS-214: a task too large to converge is not a task. T1 of v1.83 carried 8 acceptance items and a
|
|
305
|
+
* diff at the 130k cap; it took 28 dispatches and never passed review, while T3 (4 items) took 5.
|
|
306
|
+
*/
|
|
307
|
+
export function taskBudgetErrors(tasks) {
|
|
308
|
+
const errors = [];
|
|
309
|
+
for (const t of tasks) {
|
|
310
|
+
const items = t.acceptance?.length ?? 0;
|
|
311
|
+
if (items > MAX_ACCEPTANCE_ITEMS) {
|
|
312
|
+
errors.push(`${t.id} declares ${items} acceptance items (max ${MAX_ACCEPTANCE_ITEMS}) — split it. `
|
|
313
|
+
+ `Every item is a thing one worker must satisfy at once and one reviewer must verify in one pass.`);
|
|
314
|
+
}
|
|
315
|
+
const files = t.files?.length ?? 0;
|
|
316
|
+
if (files > MAX_FILES_PATTERNS) {
|
|
317
|
+
errors.push(`${t.id} declares ${files} files[] patterns (max ${MAX_FILES_PATTERNS}) — split it. `
|
|
318
|
+
+ `A wide write surface is what makes tasks collide and diffs exceed the review cap.`);
|
|
319
|
+
}
|
|
320
|
+
}
|
|
321
|
+
return errors;
|
|
322
|
+
}
|
|
323
|
+
/** Every Task Unit Contract violation in one pass, ready to throw. */
|
|
324
|
+
export function taskUnitContractErrors(tasks) {
|
|
325
|
+
return [...separabilityErrors(tasks), ...taskBudgetErrors(tasks)];
|
|
326
|
+
}
|
package/dist/compile/index.js
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { existsSync, readFileSync, statSync } from "node:fs";
|
|
2
2
|
import { join } from "node:path";
|
|
3
|
+
import { taskUnitContractErrors } from "./collateral.js";
|
|
3
4
|
import { CompileError } from "./common.js";
|
|
4
5
|
import { compileGsd, isGsdPhaseDir } from "./gsd.js";
|
|
5
6
|
import { compileNative, TICKMARKR_NATIVE_MARKER } from "./native.js";
|
|
@@ -25,15 +26,28 @@ function detect(src) {
|
|
|
25
26
|
}
|
|
26
27
|
return null;
|
|
27
28
|
}
|
|
29
|
+
// OBS-212/214: the Task Unit Contract is enforced for EVERY front-end here, at the one seam they all
|
|
30
|
+
// pass through, so no spec dialect can author a graph whose independence claim is false or whose
|
|
31
|
+
// tasks are too large to converge. A violation is a compile error, never a warning — the failures it
|
|
32
|
+
// prevents (silently dropped commits, a 28-dispatch task) are invisible until they have already cost
|
|
33
|
+
// hours, which is exactly the class of thing that has to fail at authoring time.
|
|
34
|
+
function enforceTaskUnitContract(g, src) {
|
|
35
|
+
const errors = taskUnitContractErrors(g.tasks);
|
|
36
|
+
if (errors.length > 0) {
|
|
37
|
+
throw new CompileError(`${src} violates the task unit contract (${errors.length} error${errors.length > 1 ? "s" : ""}):\n`
|
|
38
|
+
+ errors.map((e) => ` - ${e}`).join("\n"));
|
|
39
|
+
}
|
|
40
|
+
return g;
|
|
41
|
+
}
|
|
28
42
|
export function compileSource(src, type, root) {
|
|
29
43
|
const kind = type ?? detect(src);
|
|
30
44
|
if (kind === "speckit")
|
|
31
|
-
return compileSpecKit(src);
|
|
45
|
+
return enforceTaskUnitContract(compileSpecKit(src), src);
|
|
32
46
|
if (kind === "gsd")
|
|
33
|
-
return compileGsd(src, root);
|
|
47
|
+
return enforceTaskUnitContract(compileGsd(src, root), src);
|
|
34
48
|
if (kind === "native")
|
|
35
|
-
return compileNative(src);
|
|
49
|
+
return enforceTaskUnitContract(compileNative(src), src);
|
|
36
50
|
if (kind === "prd")
|
|
37
|
-
return compilePrd(src);
|
|
51
|
+
return enforceTaskUnitContract(compilePrd(src), src);
|
|
38
52
|
throw new CompileError(`cannot detect spec type for ${src} — pass a Spec Kit feature dir (with tasks.md), a GSD phase dir (with *-PLAN.md), or a marked native/generic PRD .md file, or use --type speckit|prd|gsd|native`);
|
|
39
53
|
}
|
package/dist/drivers/herdr.d.ts
CHANGED
|
@@ -53,6 +53,7 @@ export declare class HerdrDriver implements ExecutorDriver {
|
|
|
53
53
|
private joinGroup;
|
|
54
54
|
run(slot: Slot, cmd: string): Promise<void>;
|
|
55
55
|
private deliver;
|
|
56
|
+
nudge(slot: Slot, message: string): Promise<boolean>;
|
|
56
57
|
private deliverPersistentShellCommand;
|
|
57
58
|
private submitVerifiedDelivery;
|
|
58
59
|
private settleDeliveryLine;
|
package/dist/drivers/herdr.js
CHANGED
|
@@ -524,6 +524,25 @@ export class HerdrDriver {
|
|
|
524
524
|
// The narrator launches a perpetual shell watch, not an adapter-backed input interface. Keep
|
|
525
525
|
// OBS-85's readiness + paste read-back and the historical single Enter, while worker/gate runs
|
|
526
526
|
// continue through positive post-Enter evidence in submitVerifiedDelivery.
|
|
527
|
+
// OBS-201: liveness nudge — deliver one message into the live worker TUI through the exact
|
|
528
|
+
// pincer every dispatch uses (readiness stable-frame, type-without-Enter, read-back, C-u clear
|
|
529
|
+
// on corruption, verified submit), serialized on the deliveryQueue so it can never interleave
|
|
530
|
+
// with a concurrent dispatch's paste. Targets ONLY the pinned delivered pane: the pin exists so
|
|
531
|
+
// liveness cannot drift to a label rebound onto another pane (herdr.ts paneId contract) — if no
|
|
532
|
+
// pin exists the dispatch never verifiably landed, and nudging a resolved-by-label pane would
|
|
533
|
+
// reopen that hole. Best-effort: false on any failure, never a throw.
|
|
534
|
+
async nudge(slot, message) {
|
|
535
|
+
const pinned = this.deliveredPanes.get(slot);
|
|
536
|
+
if (!pinned)
|
|
537
|
+
return false;
|
|
538
|
+
try {
|
|
539
|
+
await this.deliveryQueue(() => this.deliver(slot, message, pinned));
|
|
540
|
+
return true;
|
|
541
|
+
}
|
|
542
|
+
catch {
|
|
543
|
+
return false; // the daemon journals worker-nudge-failed and falls back to the stall window
|
|
544
|
+
}
|
|
545
|
+
}
|
|
527
546
|
async deliverPersistentShellCommand(slot, cmd) {
|
|
528
547
|
return this.deliveryQueue(async () => {
|
|
529
548
|
const pane = await this.paneId(slot);
|
package/dist/drivers/types.d.ts
CHANGED
|
@@ -72,6 +72,7 @@ export interface ExecutorDriver {
|
|
|
72
72
|
status(slot: Slot): Promise<string>;
|
|
73
73
|
read(slot: Slot, lines: number): Promise<string>;
|
|
74
74
|
sendKey?(slot: Slot, key: string): Promise<void>;
|
|
75
|
+
nudge?(slot: Slot, message: string): Promise<boolean>;
|
|
75
76
|
notify(msg: string, opts?: NotifyOpts): Promise<void>;
|
|
76
77
|
close(slot: Slot): Promise<void>;
|
|
77
78
|
worktree(repo: string, branch: string, baseRef: string): Promise<string>;
|