tickmarkr 1.87.0 → 1.90.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/catalog.d.ts +18 -1
- package/dist/adapters/catalog.js +44 -1
- package/dist/adapters/fake.d.ts +2 -1
- package/dist/adapters/fake.js +7 -0
- package/dist/adapters/grok.js +11 -0
- package/dist/adapters/kimi.d.ts +2 -1
- package/dist/adapters/kimi.js +36 -0
- package/dist/adapters/opencode.js +17 -0
- package/dist/adapters/pi.js +11 -0
- package/dist/adapters/prompt.js +8 -1
- package/dist/adapters/registry.js +76 -57
- package/dist/adapters/types.d.ts +34 -3
- package/dist/adapters/types.js +99 -1
- package/dist/cli/commands/approve.d.ts +2 -0
- package/dist/cli/commands/approve.js +104 -84
- package/dist/cli/commands/compile.d.ts +1 -1
- package/dist/cli/commands/compile.js +29 -12
- package/dist/cli/commands/doctor.d.ts +16 -0
- package/dist/cli/commands/doctor.js +52 -0
- package/dist/cli/commands/init.js +2 -1
- package/dist/cli/commands/plan.d.ts +1 -1
- package/dist/cli/commands/plan.js +10 -1
- package/dist/cli/commands/report.js +49 -0
- package/dist/cli/commands/status.js +298 -96
- package/dist/cli/commands/verify.d.ts +9 -0
- package/dist/cli/commands/verify.js +177 -0
- package/dist/cli/harness.d.ts +13 -0
- package/dist/cli/harness.js +50 -0
- package/dist/cli/index.d.ts +1 -1
- package/dist/cli/index.js +3 -1
- package/dist/compile/collateral.js +11 -11
- package/dist/compile/common.js +2 -2
- package/dist/compile/index.d.ts +14 -3
- package/dist/compile/index.js +36 -10
- package/dist/compile/native.d.ts +15 -1
- package/dist/compile/native.js +310 -28
- package/dist/config/config.js +2 -2
- package/dist/drivers/subprocess.d.ts +6 -1
- package/dist/drivers/subprocess.js +9 -4
- package/dist/gates/acceptance.d.ts +21 -1
- package/dist/gates/acceptance.js +67 -22
- package/dist/gates/artifact-manifest.d.ts +119 -0
- package/dist/gates/artifact-manifest.js +357 -0
- package/dist/gates/baseline.d.ts +45 -5
- package/dist/gates/baseline.js +119 -15
- package/dist/gates/llm.js +37 -26
- package/dist/gates/review.d.ts +16 -11
- package/dist/gates/review.js +44 -150
- package/dist/gates/run-gates.js +124 -7
- package/dist/gates/scope.js +3 -3
- package/dist/graph/files-glob.d.ts +18 -0
- package/dist/graph/files-glob.js +22 -0
- package/dist/graph/schema.d.ts +3 -1
- package/dist/graph/schema.js +4 -1
- package/dist/run/daemon.d.ts +44 -0
- package/dist/run/daemon.js +2334 -1973
- package/dist/run/git.d.ts +53 -0
- package/dist/run/git.js +119 -5
- package/dist/run/interactive-seed.d.ts +6 -2
- package/dist/run/interactive-seed.js +72 -5
- package/dist/run/journal.d.ts +9 -1
- package/dist/run/journal.js +99 -9
- package/dist/run/lock.d.ts +11 -0
- package/dist/run/lock.js +97 -6
- package/dist/run/merge.d.ts +4 -1
- package/dist/run/merge.js +26 -7
- package/dist/run/outcome.d.ts +50 -0
- package/dist/run/outcome.js +152 -0
- package/dist/run/protocol.d.ts +460 -0
- package/dist/run/protocol.js +433 -0
- package/dist/run/supervision.d.ts +29 -0
- package/dist/run/supervision.js +189 -0
- package/fixtures/authoring-lints/01-awk-range-self-pass.spec.md +12 -0
- package/fixtures/authoring-lints/02-judge-text-key-miss.spec.md +7 -0
- package/fixtures/authoring-lints/03-c1-t41-rendered-observable.spec.md +8 -0
- package/fixtures/authoring-lints/04-c1-t24-named-file.spec.md +8 -0
- package/fixtures/authoring-lints/05-c2-t24-t28-dep-inversion.spec.md +7 -0
- package/fixtures/authoring-lints/06-c2-denumbered-coupling.spec.md +7 -0
- package/fixtures/authoring-lints/07-c3a-t41-line-count-proxy.spec.md +7 -0
- package/fixtures/authoring-lints/08-c3b-t41-governance-referent.spec.md +7 -0
- package/fixtures/authoring-lints/09-c4-universals-without-pointer.spec.md +7 -0
- package/fixtures/authoring-lints/10-c5-t34-conjunct-flood.spec.md +7 -0
- package/fixtures/authoring-lints/11-c6-t34-q3-q9-q20-bundle.spec.md +7 -0
- package/fixtures/authoring-lints/12-c7-t24-prose-seam.spec.md +8 -0
- package/fixtures/wrapped-acceptance.native.md +29 -0
- package/package.json +1 -1
- package/schema/rungraph.schema.json +21 -2
- package/skills/tickmarkr-overseer/SKILL.md +262 -5
- package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +79 -8
- package/skills/tickmarkr-overseer/scripts/watch-contamination.sh +80 -0
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +86 -0
- package/skills/tickmarkr-overseer/scripts/watch-parks.sh +96 -0
- package/skills/tickmarkr-overseer/scripts/watch-pending-input.sh +201 -0
|
@@ -1,109 +1,120 @@
|
|
|
1
1
|
import { userInfo } from "node:os";
|
|
2
2
|
import { GATE_NAMES } from "../../graph/schema.js";
|
|
3
3
|
import { ATTEMPT_CAP_RELEASE, GATE_SATISFIED_RELEASE, Journal, RECHECK_RELEASE, REVIEW_UPHELD_RELEASE } from "../../run/journal.js";
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
// The approval is a JOURNAL EVENT (task-approved) carrying who and when — it touches ONLY the
|
|
7
|
-
// append-only journal. Writing it into tickmarkr's compiled graph artifact would be silently erased by
|
|
8
|
-
// the next recompile (which re-emits humanGate:true from the plan frontmatter) — Phase 42 D-02.
|
|
9
|
-
//
|
|
10
|
-
// v1.24 OBS-18: when the park kind is attempt-cap (not a humanGate pre-dispatch park), the event
|
|
11
|
-
// also carries `release: "attempt-cap"`. replayResumeState zeros the attempt budget on that marker so
|
|
12
|
-
// resume dispatches instead of re-parking in the same tick; tried-list is preserved. Unknown kinds
|
|
13
|
-
// receive no release and remain fail-closed to a human rather than being inferred from prose.
|
|
14
|
-
//
|
|
15
|
-
// Fail-closed (D-05): unknown runId, unknown taskId, a not-parked task, and a double-approve are all
|
|
16
|
-
// LOUD refusals that name the reason and append NO event — never a silent no-op. A handler throw
|
|
17
|
-
// becomes `tickmarkr approve: <message>` at exit 1 (src/cli/index.ts dispatch).
|
|
18
|
-
//
|
|
19
|
-
// Who/when is truthful, not dressed-up auth (D-03): default actor os.userInfo().username; --by overrides
|
|
20
|
-
// for delegated approval; optional --reason; the event's ts (stamped by Journal.append) is the when.
|
|
21
|
-
// OBS-189: `--uphold` is the second decision a review park offers. Plain approve accepts the diff the
|
|
22
|
-
// reviewer rejected (gate-satisfied); --uphold sides WITH the reviewer and funds ONE fixed worker
|
|
23
|
-
// attempt carrying the findings — the park costs an attempt, never the run.
|
|
4
|
+
import { acquireApprovalSerialization, runLockOwner } from "../../run/lock.js";
|
|
5
|
+
/** The production command registered in COMMANDS; its returned bytes are what the CLI prints. */
|
|
24
6
|
export async function approve(argv, cwd = process.cwd()) {
|
|
25
|
-
const { runId, taskId, by, reason, uphold, recheck } = parseArgs(argv);
|
|
7
|
+
const { runId, taskId, by, reason, uphold, recheck, reviewRoundCeiling } = parseArgs(argv);
|
|
26
8
|
if (uphold && recheck)
|
|
27
9
|
throw new Error("--uphold and --recheck are different decisions — pass one");
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
// a silent no-op would be worse than a loud refusal — name the actual status (D-05)
|
|
36
|
-
throw new Error(`task ${taskId} is ${status}, not a parked human gate — refusing (a silent no-op would be worse)`);
|
|
37
|
-
}
|
|
38
|
-
// OBS-18: only the most recent task-human for this task decides whether this approval grants a
|
|
39
|
-
// fresh attempt budget. The closed daemon-issued kind, never a human prose string, controls release.
|
|
40
|
-
const events = journal.read();
|
|
41
|
-
let lastHumanIndex = -1;
|
|
42
|
-
for (let i = events.length - 1; i >= 0; i--) {
|
|
43
|
-
if (events[i].event === "task-human" && events[i].taskId === taskId) {
|
|
44
|
-
lastHumanIndex = i;
|
|
45
|
-
break;
|
|
10
|
+
const serialization = await acquireApprovalSerialization(cwd, runId);
|
|
11
|
+
try {
|
|
12
|
+
// Journal.open throws `no journal for <runId> at <dir>` on an unknown run — that IS the refusal.
|
|
13
|
+
const journal = Journal.open(cwd, runId);
|
|
14
|
+
const status = journal.replayStatuses().get(taskId);
|
|
15
|
+
if (status === undefined) {
|
|
16
|
+
throw new Error(`task ${taskId} has no events in run ${runId} — unknown task or never dispatched`);
|
|
46
17
|
}
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
// Fail-closed on the DATA: uphold applies only when the newest failed gate is the review gate —
|
|
51
|
-
// any other gate has no reviewer to uphold. Never inferred from the park's prose.
|
|
52
|
-
const lastFailed = events.slice(0, lastHumanIndex).reverse().find((e) => e.event === "gate-result" && e.taskId === taskId && e.data.pass === false
|
|
53
|
-
&& typeof e.data.gate === "string" && GATE_NAMES.includes(e.data.gate))?.data.gate;
|
|
54
|
-
if (lastFailed !== "review") {
|
|
55
|
-
throw new Error(`--uphold applies to a review rejection; ${taskId}'s last failed gate is ${lastFailed ?? "none"} — refusing`);
|
|
18
|
+
if (status !== "human") {
|
|
19
|
+
// a silent no-op would be worse than a loud refusal — name the actual status (D-05)
|
|
20
|
+
throw new Error(`task ${taskId} is ${status}, not a parked human gate — refusing (a silent no-op would be worse)`);
|
|
56
21
|
}
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
22
|
+
// OBS-18: only the most recent task-human for this task decides whether this approval grants a
|
|
23
|
+
// fresh attempt budget. The closed daemon-issued kind, never a human prose string, controls release.
|
|
24
|
+
const events = journal.read();
|
|
25
|
+
let lastHumanIndex = -1;
|
|
26
|
+
for (let i = events.length - 1; i >= 0; i--) {
|
|
27
|
+
if (events[i].event === "task-human" && events[i].taskId === taskId) {
|
|
28
|
+
lastHumanIndex = i;
|
|
29
|
+
break;
|
|
30
|
+
}
|
|
31
|
+
}
|
|
32
|
+
const lastHuman = events[lastHumanIndex];
|
|
33
|
+
if (uphold) {
|
|
34
|
+
// Fail-closed on the DATA: uphold applies only when the newest failed gate is the review gate —
|
|
35
|
+
// any other gate has no reviewer to uphold. Never inferred from the park's prose.
|
|
36
|
+
const lastFailed = events.slice(0, lastHumanIndex).reverse().find((e) => e.event === "gate-result" && e.taskId === taskId && e.data.pass === false
|
|
37
|
+
&& typeof e.data.gate === "string" && GATE_NAMES.includes(e.data.gate))?.data.gate;
|
|
38
|
+
if (lastFailed !== "review") {
|
|
39
|
+
throw new Error(`--uphold applies to a review rejection; ${taskId}'s last failed gate is ${lastFailed ?? "none"} — refusing`);
|
|
40
|
+
}
|
|
41
|
+
journal.append("task-approved", taskId, {
|
|
42
|
+
by,
|
|
43
|
+
...(reason ? { reason } : {}),
|
|
44
|
+
via: "cli",
|
|
45
|
+
release: REVIEW_UPHELD_RELEASE,
|
|
46
|
+
gate: "review",
|
|
47
|
+
...(reviewRoundCeiling === undefined ? {} : { reviewRoundCeiling }),
|
|
48
|
+
});
|
|
49
|
+
return disposition(cwd, runId, `upheld the reviewer for ${taskId} in ${runId} — by ${by}; run \`tickmarkr resume ${runId}\` to dispatch a fixed attempt carrying the findings`, serialization.contended);
|
|
50
|
+
}
|
|
51
|
+
const capPark = lastHuman?.data.kind === ATTEMPT_CAP_RELEASE;
|
|
52
|
+
const gateFailPark = lastHuman?.data.kind === "gate-fail";
|
|
53
|
+
if (recheck) {
|
|
54
|
+
// OBS-203: fail-closed on the PARK KIND — only a gate-fail park has a gate to re-run. Refusing
|
|
55
|
+
// elsewhere keeps --recheck from becoming a silent budget reset on a pre-dispatch human gate.
|
|
56
|
+
if (!gateFailPark) {
|
|
57
|
+
throw new Error(`--recheck applies to a gate-fail park; ${taskId}'s park kind is ${lastHuman?.data.kind ?? "none"} — refusing`);
|
|
58
|
+
}
|
|
59
|
+
journal.append("task-approved", taskId, {
|
|
60
|
+
by,
|
|
61
|
+
...(reason ? { reason } : {}),
|
|
62
|
+
via: "cli",
|
|
63
|
+
release: RECHECK_RELEASE,
|
|
64
|
+
...(reviewRoundCeiling === undefined ? {} : { reviewRoundCeiling }),
|
|
65
|
+
});
|
|
66
|
+
return disposition(cwd, runId, `re-checking ${taskId} in ${runId} — by ${by}; no gate marked satisfied, run \`tickmarkr resume ${runId}\` to re-dispatch against the full gate suite`, serialization.contended);
|
|
67
|
+
}
|
|
68
|
+
const failedGate = gateFailPark
|
|
69
|
+
? events.slice(0, lastHumanIndex).reverse().find((e) => e.event === "gate-result" && e.taskId === taskId && e.data.pass === false
|
|
70
|
+
&& typeof e.data.gate === "string" && GATE_NAMES.includes(e.data.gate))?.data.gate
|
|
71
|
+
: undefined;
|
|
72
|
+
if (gateFailPark && !failedGate) {
|
|
73
|
+
throw new Error(`task ${taskId} is parked on gate-fail but has no failed gate result — refusing to infer one`);
|
|
73
74
|
}
|
|
74
75
|
journal.append("task-approved", taskId, {
|
|
75
76
|
by,
|
|
76
77
|
...(reason ? { reason } : {}),
|
|
77
78
|
via: "cli",
|
|
78
|
-
|
|
79
|
+
...(reviewRoundCeiling === undefined ? {} : { reviewRoundCeiling }),
|
|
80
|
+
...(capPark ? { release: ATTEMPT_CAP_RELEASE } : {}),
|
|
81
|
+
...(failedGate ? { release: GATE_SATISFIED_RELEASE, gate: failedGate } : {}),
|
|
79
82
|
});
|
|
80
|
-
return `
|
|
83
|
+
return disposition(cwd, runId, `approved ${taskId} in ${runId} — by ${by}; run \`tickmarkr resume ${runId}\` to ${failedGate ? "continue past the approved gate" : "dispatch it"}`, serialization.contended);
|
|
81
84
|
}
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
&& typeof e.data.gate === "string" && GATE_NAMES.includes(e.data.gate))?.data.gate
|
|
85
|
-
: undefined;
|
|
86
|
-
if (gateFailPark && !failedGate) {
|
|
87
|
-
throw new Error(`task ${taskId} is parked on gate-fail but has no failed gate result — refusing to infer one`);
|
|
85
|
+
finally {
|
|
86
|
+
serialization.release();
|
|
88
87
|
}
|
|
89
|
-
journal.append("task-approved", taskId, {
|
|
90
|
-
by,
|
|
91
|
-
...(reason ? { reason } : {}),
|
|
92
|
-
via: "cli",
|
|
93
|
-
...(capPark ? { release: ATTEMPT_CAP_RELEASE } : {}),
|
|
94
|
-
...(failedGate ? { release: GATE_SATISFIED_RELEASE, gate: failedGate } : {}),
|
|
95
|
-
});
|
|
96
|
-
return `approved ${taskId} in ${runId} — by ${by}; run \`tickmarkr resume ${runId}\` to ${failedGate ? "continue past the approved gate" : "dispatch it"}`;
|
|
97
88
|
}
|
|
98
|
-
|
|
99
|
-
//
|
|
100
|
-
//
|
|
89
|
+
// The status is command OUTPUT, not a typed sibling result: the registered approve function returns
|
|
90
|
+
// one primitive string and the dispatcher prints those bytes unchanged. A compact sentinel followed
|
|
91
|
+
// by JSON makes the contract unambiguous to machines without widening the shared command result type.
|
|
92
|
+
// No-lock approvals keep their historical one-line result for the cockpit; a dead recorded owner and
|
|
93
|
+
// a command delayed behind terminalization both emit recorded-no-owner.
|
|
94
|
+
function disposition(cwd, runId, message, contended) {
|
|
95
|
+
const owner = runLockOwner(cwd);
|
|
96
|
+
const resume = `tickmarkr resume ${runId}`;
|
|
97
|
+
if (!owner && !contended)
|
|
98
|
+
return message;
|
|
99
|
+
const status = owner?.live ? "deferred-live" : "recorded-no-owner";
|
|
100
|
+
const record = {
|
|
101
|
+
status,
|
|
102
|
+
resume,
|
|
103
|
+
...(owner?.pid === undefined ? {} : { ownerPid: owner.pid }),
|
|
104
|
+
...(owner?.runId === undefined ? {} : { ownerRunId: owner.runId }),
|
|
105
|
+
};
|
|
106
|
+
return `${message}\nTICKMARKR_APPROVAL ${JSON.stringify(record)}`;
|
|
107
|
+
}
|
|
108
|
+
const USAGE = "usage: tickmarkr approve <run-id> <task-id> [--uphold|--recheck] [--review-rounds <positive-integer>] [--by <name>] [--reason <text>]";
|
|
109
|
+
// hand-parsed argv — no CLI framework (house style). Positionals are runId then taskId; decision,
|
|
110
|
+
// ceiling, actor and reason are flags. Throws usage on missing positionals (mirrors resume.ts/unlock.ts).
|
|
101
111
|
function parseArgs(argv) {
|
|
102
112
|
const positionals = [];
|
|
103
113
|
let by;
|
|
104
114
|
let reason;
|
|
105
115
|
let uphold = false;
|
|
106
116
|
let recheck = false;
|
|
117
|
+
let reviewRoundCeiling;
|
|
107
118
|
for (let i = 0; i < argv.length; i++) {
|
|
108
119
|
const a = argv[i];
|
|
109
120
|
if (a === "--by") {
|
|
@@ -122,6 +133,15 @@ function parseArgs(argv) {
|
|
|
122
133
|
else if (a === "--recheck") {
|
|
123
134
|
recheck = true;
|
|
124
135
|
}
|
|
136
|
+
else if (a === "--review-rounds") {
|
|
137
|
+
const value = argv[++i];
|
|
138
|
+
if (value === undefined)
|
|
139
|
+
throw new Error(USAGE);
|
|
140
|
+
if (!/^[1-9]\d*$/.test(value) || !Number.isSafeInteger(Number(value))) {
|
|
141
|
+
throw new Error("--review-rounds must be a positive integer");
|
|
142
|
+
}
|
|
143
|
+
reviewRoundCeiling = Number(value);
|
|
144
|
+
}
|
|
125
145
|
else {
|
|
126
146
|
positionals.push(a);
|
|
127
147
|
}
|
|
@@ -130,5 +150,5 @@ function parseArgs(argv) {
|
|
|
130
150
|
if (!runId || !taskId) {
|
|
131
151
|
throw new Error(USAGE);
|
|
132
152
|
}
|
|
133
|
-
return { runId, taskId, by: by ?? userInfo().username, reason, uphold, recheck };
|
|
153
|
+
return { runId, taskId, by: by ?? userInfo().username, reason, uphold, recheck, reviewRoundCeiling };
|
|
134
154
|
}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
export declare function compile(argv: string[], cwd?: string): Promise<string>;
|
|
1
|
+
export declare function compile(argv: string[], cwd?: string, harnessFrom?: string | undefined): Promise<string>;
|
|
@@ -1,28 +1,45 @@
|
|
|
1
1
|
import { isAbsolute, join } from "node:path";
|
|
2
2
|
import { parseArgs } from "node:util";
|
|
3
|
+
import { collateralLints, sourceScopeLints } from "../../compile/collateral.js";
|
|
3
4
|
import { compileSource } from "../../compile/index.js";
|
|
4
5
|
import { saveGraph, stateDirName } from "../../graph/graph.js";
|
|
5
6
|
import { acquireRunLock, releaseRunLock } from "../../run/lock.js";
|
|
6
|
-
|
|
7
|
+
import { harnessLine, resolveHarness } from "../harness.js";
|
|
8
|
+
// v1.89 T4: harnessFrom is the resolver's INPUT (see plan.ts); the default is the INVOKED entrypoint
|
|
9
|
+
// (`process.argv[1]`, the bin symlink), never this module's own url — that names an internal module.
|
|
10
|
+
export async function compile(argv, cwd = process.cwd(), harnessFrom = process.argv[1]) {
|
|
7
11
|
const { values, positionals } = parseArgs({
|
|
8
12
|
args: argv,
|
|
9
|
-
options: {
|
|
13
|
+
options: {
|
|
14
|
+
type: { type: "string" },
|
|
15
|
+
"dry-run": { type: "boolean" },
|
|
16
|
+
},
|
|
10
17
|
allowPositionals: true,
|
|
11
18
|
});
|
|
12
19
|
const src = positionals[0];
|
|
13
20
|
if (!src)
|
|
14
|
-
throw new Error("usage: tickmarkr compile <spec-dir-or-md> [--type speckit|prd|gsd|native]");
|
|
21
|
+
throw new Error("usage: tickmarkr compile <spec-dir-or-md> [--type speckit|prd|gsd|native] [--dry-run]");
|
|
15
22
|
// resolve against the target repo, not the process cwd (the CLI test passes a tmp repo)
|
|
23
|
+
// Both modes reach the same pure compiler; --dry-run only removes the lock/write side effect below.
|
|
16
24
|
const g = compileSource(isAbsolute(src) ? src : join(cwd, src), values.type, cwd);
|
|
17
|
-
// HARD-01 / Sol #3: hold the same link(2) run lock as the daemon around saveGraph so compile
|
|
18
|
-
// cannot swap graph.json under an active run between the daemon's read and act.
|
|
19
25
|
const stateDir = stateDirName(cwd);
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
26
|
+
if (!values["dry-run"]) {
|
|
27
|
+
// HARD-01 / Sol #3: hold the same link(2) run lock as the daemon around saveGraph so compile
|
|
28
|
+
// cannot swap graph.json under an active run between the daemon's read and act.
|
|
29
|
+
acquireRunLock(cwd, "compile");
|
|
30
|
+
try {
|
|
31
|
+
saveGraph(cwd, g);
|
|
32
|
+
}
|
|
33
|
+
finally {
|
|
34
|
+
releaseRunLock(cwd);
|
|
35
|
+
}
|
|
23
36
|
}
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
37
|
+
const summary = values["dry-run"]
|
|
38
|
+
? `validated ${src} (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)}) — dry run; no graph written`
|
|
39
|
+
: `compiled ${src} → ${stateDir}/graph.json (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)})`;
|
|
40
|
+
const scopeLints = [...collateralLints(g.tasks, cwd), ...sourceScopeLints(g.tasks, cwd)];
|
|
41
|
+
const diagnostics = scopeLints.length
|
|
42
|
+
? `\nscope lints:\n${scopeLints.map((lint) => ` ! ${lint}`).join("\n")}`
|
|
43
|
+
: "";
|
|
44
|
+
return `${harnessLine(resolveHarness(harnessFrom))}\n${summary}${diagnostics}`;
|
|
28
45
|
}
|
|
@@ -9,4 +9,20 @@ export type DoctorOpts = {
|
|
|
9
9
|
catalog?: CatalogReadResult;
|
|
10
10
|
catalogNow?: () => Date;
|
|
11
11
|
};
|
|
12
|
+
/**
|
|
13
|
+
* Q122s (TRIAL T-OBS-3): tickmarkr provisions run worktrees INSIDE the repo
|
|
14
|
+
* (.tickmarkr/worktrees.noindex/). A target repo whose test runner collects with
|
|
15
|
+
* repo-wide globs (jest's default `**/__tests__/**` shape, vitest's default include)
|
|
16
|
+
* scans every suite once per live worktree — duplicate-suite interference red the
|
|
17
|
+
* baseline AND tip-verify on SentioQ's first external run, a class invisible to
|
|
18
|
+
* dogfooding (this repo's own includes are rooted at tests/).
|
|
19
|
+
*
|
|
20
|
+
* Text-level heuristic, advisory only — never enters health/doctor.json or routing.
|
|
21
|
+
* ponytail: ceilings — a commented-out ignore false-passes; an include built from
|
|
22
|
+
* variables is judged by its literal text. Both cost one advisory row, nothing more.
|
|
23
|
+
*/
|
|
24
|
+
export declare function runnerIgnoreFinding(cwd: string): {
|
|
25
|
+
verdict: "pass" | "warn";
|
|
26
|
+
detail: string;
|
|
27
|
+
} | undefined;
|
|
12
28
|
export declare function doctor(_argv: string[], cwd?: string, adapters?: WorkerAdapter[], opts?: DoctorOpts): Promise<string>;
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { writeFileSync } from "node:fs";
|
|
2
|
+
import { existsSync, readFileSync } from "node:fs";
|
|
2
3
|
import { join } from "node:path";
|
|
3
4
|
import { allAdapters, binaryShadowWarnings, detectCandidateClis, flagDriftWarnings, modelAliasExclusions, modelAliasLine, probeAll, probeModels, servableExclusions, servabilityLine, writeDoctor } from "../../adapters/registry.js";
|
|
4
5
|
import { CLAUDE_ALIAS_IDENTITY_STAMPS, claudeCode, resolveClaudeAliasIdentity } from "../../adapters/claude-code.js";
|
|
@@ -13,6 +14,53 @@ import { readCachedCatalog, refreshCatalogCommand } from "../../adapters/catalog
|
|
|
13
14
|
const visual = () => process.stdout.isTTY === true && process.env.NO_COLOR === undefined;
|
|
14
15
|
const alignedStatusRow = (verdict, key, value) => ` ${statusRow(verdict, kvRow(key, value).slice(2))}`;
|
|
15
16
|
const attentionRow = (text) => ` ${statusRow("warn", text)}`;
|
|
17
|
+
/**
|
|
18
|
+
* Q122s (TRIAL T-OBS-3): tickmarkr provisions run worktrees INSIDE the repo
|
|
19
|
+
* (.tickmarkr/worktrees.noindex/). A target repo whose test runner collects with
|
|
20
|
+
* repo-wide globs (jest's default `**/__tests__/**` shape, vitest's default include)
|
|
21
|
+
* scans every suite once per live worktree — duplicate-suite interference red the
|
|
22
|
+
* baseline AND tip-verify on SentioQ's first external run, a class invisible to
|
|
23
|
+
* dogfooding (this repo's own includes are rooted at tests/).
|
|
24
|
+
*
|
|
25
|
+
* Text-level heuristic, advisory only — never enters health/doctor.json or routing.
|
|
26
|
+
* ponytail: ceilings — a commented-out ignore false-passes; an include built from
|
|
27
|
+
* variables is judged by its literal text. Both cost one advisory row, nothing more.
|
|
28
|
+
*/
|
|
29
|
+
export function runnerIgnoreFinding(cwd) {
|
|
30
|
+
const readIf = (p) => (existsSync(join(cwd, p)) ? readFileSync(join(cwd, p), "utf8") : "");
|
|
31
|
+
const pkgText = readIf("package.json");
|
|
32
|
+
let pkg = {};
|
|
33
|
+
try {
|
|
34
|
+
pkg = JSON.parse(pkgText || "{}");
|
|
35
|
+
}
|
|
36
|
+
catch { /* unparseable manifest: judge config files alone */ }
|
|
37
|
+
const jestConfig = ["jest.config.js", "jest.config.cjs", "jest.config.mjs", "jest.config.ts", "jest.config.json"].find((p) => existsSync(join(cwd, p)));
|
|
38
|
+
const vitestConfig = ["vitest.config.ts", "vitest.config.js", "vitest.config.mts", "vitest.config.mjs", "vitest.config.cts"].find((p) => existsSync(join(cwd, p)));
|
|
39
|
+
const testScript = pkg.scripts?.test ?? "";
|
|
40
|
+
const isJest = jestConfig !== undefined || pkg.jest !== undefined || /\bjest\b/.test(testScript);
|
|
41
|
+
const isVitest = !isJest && (vitestConfig !== undefined || /\bvitest\b/.test(testScript));
|
|
42
|
+
if (!isJest && !isVitest)
|
|
43
|
+
return undefined;
|
|
44
|
+
const configText = isJest
|
|
45
|
+
? `${jestConfig ? readIf(jestConfig) : ""}${pkg.jest ? JSON.stringify(pkg.jest) : ""}`
|
|
46
|
+
: vitestConfig ? readIf(vitestConfig) : "";
|
|
47
|
+
if (/\.tickmarkr|worktrees\.noindex/.test(configText)) {
|
|
48
|
+
return { verdict: "pass", detail: `${isJest ? "jest" : "vitest"} config ignores .tickmarkr/ — run worktrees stay out of the suite` };
|
|
49
|
+
}
|
|
50
|
+
// Rooted collection globs never reach .tickmarkr/…; only repo-wide `**/`-style patterns (or
|
|
51
|
+
// runner defaults, which are repo-wide) can collect worktree copies.
|
|
52
|
+
const repoWide = configText === "" || /["'`]\*\*\//.test(configText);
|
|
53
|
+
if (!repoWide) {
|
|
54
|
+
return { verdict: "pass", detail: `${isJest ? "jest" : "vitest"} collection globs are rooted — run worktrees are not collectable` };
|
|
55
|
+
}
|
|
56
|
+
const remedy = isJest
|
|
57
|
+
? `add "/.tickmarkr/" to testPathIgnorePatterns in ${jestConfig ?? "the package.json jest block"}`
|
|
58
|
+
: `add "**/.tickmarkr/**" to test.exclude in ${vitestConfig ?? "a vitest config"}`;
|
|
59
|
+
return {
|
|
60
|
+
verdict: "warn",
|
|
61
|
+
detail: `${isJest ? "jest" : "vitest"} collects repo-wide and will scan .tickmarkr/ run worktrees (duplicate suites, false reds) — ${remedy}`,
|
|
62
|
+
};
|
|
63
|
+
}
|
|
16
64
|
export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(), opts = {}) {
|
|
17
65
|
if (_argv.length === 1 && _argv[0] === "--refresh-catalog") {
|
|
18
66
|
const refreshed = await refreshCatalogCommand({ repoRoot: cwd, now: opts.catalogNow });
|
|
@@ -90,6 +138,10 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
|
|
|
90
138
|
rows.push(...detectCandidateClis({ cwd }).map(({ binary, path }) => alignedStatusRow("warn", binary, `detected at ${path} (no drive contract — not routable)`)));
|
|
91
139
|
const herdr = HerdrDriver.available();
|
|
92
140
|
rows.push(alignedStatusRow(herdr ? "pass" : "fail", "herdr", herdr ? "driver available (HERDR_ENV=1)" : "not detected — subprocess driver will be used"));
|
|
141
|
+
// Q122s: the target repo's runner must not scan tickmarkr's own worktrees (advisory row).
|
|
142
|
+
const runnerIgnore = runnerIgnoreFinding(cwd);
|
|
143
|
+
if (runnerIgnore)
|
|
144
|
+
rows.push(alignedStatusRow(runnerIgnore.verdict, "test-runner", runnerIgnore.detail));
|
|
93
145
|
// v1.22 T5: workspace-trust pre-flight — per installed adapter: trusted | seeded | action-required | n/a.
|
|
94
146
|
// action-required names the exact one-time command (or dialog) the operator must run once.
|
|
95
147
|
rows.push(legend("workspace trust:"));
|
|
@@ -87,6 +87,7 @@ tickmarkr compiles repository specs into isolated, independently verified agent
|
|
|
87
87
|
- \`tickmarkr resume <runId>\` — continue a paused or failed run
|
|
88
88
|
- \`tickmarkr approve <runId> <taskId>\` — release a human gate
|
|
89
89
|
- \`tickmarkr report <runId> --md\` — execution record beside the spec
|
|
90
|
+
- \`tickmarkr verify --base <ref>\` — run the gate battery standalone against merge-base(base, HEAD)..HEAD: no daemon, no retries, one fail-closed verdict (\`--criteria <file>\` or \`--task <id>\` adds the semantic gates; \`--no-review\` for deterministic-only)
|
|
90
91
|
|
|
91
92
|
Loop: compile → plan → run → report. Watch the journal for run-end rather than polling workers.
|
|
92
93
|
|
|
@@ -100,7 +101,7 @@ Outside multi-agent environments, run the loop directly.
|
|
|
100
101
|
|
|
101
102
|
### Version preflight
|
|
102
103
|
|
|
103
|
-
Before \`tickmarkr compile\` or \`tickmarkr run\`: run \`tickmarkr version\`, read \`package.json\` version, and if the binary is older on major.minor, stop and tell the operator to update. Never proceed on hope — stale binaries silently skip daemon gates.
|
|
104
|
+
Before \`tickmarkr compile\` or \`tickmarkr run\`: run \`tickmarkr version\`, read \`package.json\` version, and if the binary is older on major.minor, stop and tell the operator to update. Never proceed on hope — stale binaries silently skip daemon gates. Also verify no run is live before starting one: \`pgrep -f "tickmarkr (run|resume)"\` must be empty — match the process, not one install path (\`dist/cli/index.js\` alone misses global and homebrew installs), and treat a held \`.tickmarkr/graph.lock\` as a live run until its holder pid is proven dead.
|
|
104
105
|
|
|
105
106
|
### Tip-verify-before-green
|
|
106
107
|
|
|
@@ -1,2 +1,2 @@
|
|
|
1
1
|
import type { WorkerAdapter } from "../../adapters/types.js";
|
|
2
|
-
export declare function plan(argv: string[], cwd?: string, adapters?: WorkerAdapter[]): Promise<string>;
|
|
2
|
+
export declare function plan(argv: string[], cwd?: string, adapters?: WorkerAdapter[], harnessFrom?: string | undefined): Promise<string>;
|
|
@@ -11,6 +11,7 @@ import { staffLedEvidence } from "../../route/profile.js";
|
|
|
11
11
|
import { route, RoutingError } from "../../route/router.js";
|
|
12
12
|
import { modelId } from "../../gates/review.js";
|
|
13
13
|
import { loadRoutingProfile } from "../../run/journal.js";
|
|
14
|
+
import { harnessLine, resolveHarness } from "../harness.js";
|
|
14
15
|
// T4 (v1.50): TTY-only brand pass — the title helper frames the routing table, lint/unroutable
|
|
15
16
|
// markers carry the attention glyph, section labels dim to chrome (the doctor/status system).
|
|
16
17
|
// Gated on ttyVisual(): the non-TTY surface returns untouched (byte-pinned, machine-consumable).
|
|
@@ -34,7 +35,12 @@ const fleetCanCrossVendorReview = (channels) => {
|
|
|
34
35
|
return true;
|
|
35
36
|
return false;
|
|
36
37
|
};
|
|
37
|
-
|
|
38
|
+
// v1.89 T4: harnessFrom is the resolver's INPUT — a caller (the byte-pinned goldens) fixes the location
|
|
39
|
+
// and keeps this machine's absolute paths out of a fixture. The default is the INVOKED entrypoint,
|
|
40
|
+
// `process.argv[1]`: the bin symlink a global install puts on PATH, which resolves to dist/cli/index.js.
|
|
41
|
+
// It is NOT `import.meta.url` — that names dist/cli/commands/plan.js, an internal module of the harness
|
|
42
|
+
// rather than the harness that was invoked, so the banner would identify the wrong file entirely.
|
|
43
|
+
export async function plan(argv, cwd = process.cwd(), adapters = allAdapters(), harnessFrom = process.argv[1]) {
|
|
38
44
|
// ponytail: hardcoded 24h TTL — promote to config when an operator asks. mtime is the signal because
|
|
39
45
|
// doctor.json has no probe timestamp and a schema field would break the existing-files compat invariant.
|
|
40
46
|
const DOCTOR_STALE_MS = 24 * 60 * 60 * 1000;
|
|
@@ -62,9 +68,12 @@ export async function plan(argv, cwd = process.cwd(), adapters = allAdapters())
|
|
|
62
68
|
let deviations = 0;
|
|
63
69
|
// v1.51 T4: the mode is never invisible — the header names the resolved mode, its winning
|
|
64
70
|
// source, and the explore posture; each task row carries a floor-derivation line below.
|
|
71
|
+
// v1.89 T4: the harness names itself ABOVE the routing table — the table is only as trustworthy as the
|
|
72
|
+
// binary that produced it, and version equality cannot tell an installed package from a checkout.
|
|
65
73
|
const lines = [
|
|
66
74
|
`tickmarkr plan — dry run (${channels.length} channels available)`,
|
|
67
75
|
`mode: ${mode.mode} (${source}) · explore ${cfg.routing.explore?.mode ?? "on"}`,
|
|
76
|
+
harnessLine(resolveHarness(harnessFrom)),
|
|
68
77
|
"",
|
|
69
78
|
];
|
|
70
79
|
const derivation = (shape) => {
|
|
@@ -9,6 +9,7 @@ import { compareRuns } from "../../report/compare.js";
|
|
|
9
9
|
import { estimateCosts } from "../../report/cost.js";
|
|
10
10
|
import { cellsOf, cellSummary } from "../../route/profile.js";
|
|
11
11
|
import { Journal, loadRoutingProfile } from "../../run/journal.js";
|
|
12
|
+
import { deriveRunCockpitData } from "../../tui/cockpit/derive.js";
|
|
12
13
|
const n = (x) => x.toLocaleString("en-US"); // explicit locale — CI/darwin flake guard
|
|
13
14
|
const EM = "—";
|
|
14
15
|
// TokenUsage fields that are actually present — filtered, never coalesced to zero (absent ⇒ unmetered).
|
|
@@ -144,6 +145,53 @@ const outcomeFor = (events, taskId, runEnd) => {
|
|
|
144
145
|
}
|
|
145
146
|
return "not recorded";
|
|
146
147
|
};
|
|
148
|
+
const tipStatusItem = (runId, events) => {
|
|
149
|
+
try {
|
|
150
|
+
return deriveRunCockpitData({ fileName: runId, raw: events.map((e) => JSON.stringify(e)).join("\n") }, "", // binaryVersion — unread here; the tip-verify status item is all this asks for
|
|
151
|
+
{ isDaemonAlive: () => false }).statusItems.find((item) => item.text.startsWith("tip-verify "));
|
|
152
|
+
}
|
|
153
|
+
catch {
|
|
154
|
+
// A capture the cockpit refuses (empty, or no run-start) verified nothing. Absent, never a pass.
|
|
155
|
+
return undefined;
|
|
156
|
+
}
|
|
157
|
+
};
|
|
158
|
+
/**
|
|
159
|
+
* The events the cockpit's verdict speaks for: a verification cycle ends at the `run-end` that
|
|
160
|
+
* closes it (derive.ts tipVerificationPassed slices to `lastRunEnd + 1`), so verify events a later
|
|
161
|
+
* resume appended belong to a cycle nothing has closed. Both readings below take THIS one slice, so
|
|
162
|
+
* the record can never name a cache that belongs to a cycle the state never judged.
|
|
163
|
+
*/
|
|
164
|
+
const closedCycle = (events) => {
|
|
165
|
+
for (let i = events.length - 1; i >= 0; i--) {
|
|
166
|
+
if (events[i].event === "run-end")
|
|
167
|
+
return events.slice(0, i + 1);
|
|
168
|
+
}
|
|
169
|
+
return events;
|
|
170
|
+
};
|
|
171
|
+
const verificationOf = (runId, events) => {
|
|
172
|
+
const cycle = closedCycle(events);
|
|
173
|
+
const tip = tipStatusItem(runId, cycle);
|
|
174
|
+
if (tip?.state === "fail")
|
|
175
|
+
return "failed";
|
|
176
|
+
if (tip?.state !== "pass")
|
|
177
|
+
return "absent";
|
|
178
|
+
// A cached green is a verified green of an EARLIER tip (daemon.ts verifyIntegrationTipCached
|
|
179
|
+
// stamps `cached` on every gate it replays), so the record names it instead of folding it into a
|
|
180
|
+
// plain pass. Still no second window: the cockpit reports "pass" only when the closed cycle
|
|
181
|
+
// carried verdicts, so the last `tip-verify` IN THAT CYCLE is the one it passed on.
|
|
182
|
+
const last = [...cycle].reverse().find((e) => e.event === "tip-verify");
|
|
183
|
+
return last?.data.cached === true ? "cached" : "passed";
|
|
184
|
+
};
|
|
185
|
+
// Absent is a state to render, never a zero to invent — and no reading here calls a run green.
|
|
186
|
+
// NOTE: every reading below is byte-pinned by tests/fixtures/brand-surfaces/report-md.md — the
|
|
187
|
+
// golden freezes the WHOLE markdown record, so changing a reading (or the line that renders it)
|
|
188
|
+
// means regenerating that fixture in the same commit.
|
|
189
|
+
const VERIFICATION_READING = {
|
|
190
|
+
passed: "passed — tickmarkr verified this run's integration tip",
|
|
191
|
+
cached: "cached — carried forward from an earlier verified tip, not re-run for this one",
|
|
192
|
+
failed: "FAILED — the run did not verify its own tip",
|
|
193
|
+
absent: "absent — no tip verification recorded: neither passed nor failed",
|
|
194
|
+
};
|
|
147
195
|
// VIS-07 / REC-01: derived only from the run journal, telemetry, and local configuration.
|
|
148
196
|
export function renderMarkdownRecord(runId, events, prices = [], rows = []) {
|
|
149
197
|
const runStart = events.find((e) => e.event === "run-start");
|
|
@@ -180,6 +228,7 @@ export function renderMarkdownRecord(runId, events, prices = [], rows = []) {
|
|
|
180
228
|
`- **done:** ${count("done")}`,
|
|
181
229
|
`- **failed:** ${count("failed")}`,
|
|
182
230
|
`- **human:** ${count("human")}`,
|
|
231
|
+
`- **verification:** ${VERIFICATION_READING[verificationOf(runId, events)]}`,
|
|
183
232
|
"",
|
|
184
233
|
"## Usage & efficiency",
|
|
185
234
|
"",
|