tickmarkr 2.6.2 → 2.6.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +12 -2
- package/dist/adapters/codex.d.ts +2 -1
- package/dist/adapters/codex.js +26 -5
- package/dist/adapters/model-lints.d.ts +1 -0
- package/dist/adapters/model-lints.js +48 -1
- package/dist/adapters/model-windows.d.ts +2 -1
- package/dist/adapters/model-windows.js +18 -7
- package/dist/adapters/types.d.ts +6 -0
- package/dist/cli/commands/approve.d.ts +6 -3
- package/dist/cli/commands/approve.js +43 -6
- package/dist/cli/commands/doctor.d.ts +8 -2
- package/dist/cli/commands/doctor.js +6 -1
- package/dist/cli/commands/fleet.js +264 -59
- package/dist/cli/commands/init.js +5 -3
- package/dist/cli/commands/plan.js +13 -8
- package/dist/cli/commands/report.d.ts +40 -2
- package/dist/cli/commands/report.js +278 -11
- package/dist/cli/commands/resume.js +2 -4
- package/dist/cli/commands/run.d.ts +11 -0
- package/dist/cli/commands/run.js +33 -5
- package/dist/cli/commands/status.js +49 -4
- package/dist/cli/commands/verify.js +332 -121
- package/dist/compile/native.js +137 -0
- package/dist/config/config.d.ts +40 -9
- package/dist/config/config.js +133 -15
- package/dist/config/fleet-overlay.d.ts +13 -3
- package/dist/config/fleet-overlay.js +12 -8
- package/dist/drivers/herdr.d.ts +12 -0
- package/dist/drivers/herdr.js +51 -0
- package/dist/drivers/orca.d.ts +9 -1
- package/dist/drivers/orca.js +29 -7
- package/dist/drivers/types.d.ts +2 -0
- package/dist/drivers/types.js +2 -2
- package/dist/gates/acceptance.d.ts +7 -0
- package/dist/gates/acceptance.js +27 -5
- package/dist/gates/baseline.d.ts +20 -1
- package/dist/gates/baseline.js +100 -20
- package/dist/gates/cache.d.ts +8 -0
- package/dist/gates/cache.js +12 -2
- package/dist/gates/llm.d.ts +6 -0
- package/dist/gates/llm.js +27 -8
- package/dist/gates/review.d.ts +6 -1
- package/dist/gates/review.js +122 -32
- package/dist/gates/run-gates.d.ts +54 -3
- package/dist/gates/run-gates.js +331 -45
- package/dist/gates/test-manifest.d.ts +42 -0
- package/dist/gates/test-manifest.js +69 -10
- package/dist/route/router.d.ts +23 -1
- package/dist/route/router.js +54 -16
- package/dist/run/consult.d.ts +3 -1
- package/dist/run/consult.js +4 -2
- package/dist/run/daemon.d.ts +2 -1
- package/dist/run/daemon.js +349 -79
- package/dist/run/interactive-seed.d.ts +4 -0
- package/dist/run/interactive-seed.js +35 -9
- package/dist/run/journal.d.ts +123 -1
- package/dist/run/journal.js +480 -17
- package/dist/run/lease.d.ts +13 -0
- package/dist/run/lease.js +45 -0
- package/dist/run/protocol.d.ts +15 -0
- package/dist/run/protocol.js +11 -1
- package/dist/run/receipt-resolver.d.ts +22 -0
- package/dist/run/receipt-resolver.js +40 -1
- package/dist/run/repair-selection.d.ts +11 -1
- package/dist/run/repair-selection.js +17 -9
- package/dist/run/wall-budget.d.ts +48 -0
- package/dist/run/wall-budget.js +280 -0
- package/dist/tui/cockpit/live-store.d.ts +6 -0
- package/dist/tui/cockpit/live-store.js +36 -11
- package/dist/tui/cockpit/run-cockpit.js +2 -2
- package/dist/tui/cockpit/run-view.d.ts +2 -1
- package/dist/tui/cockpit/run-view.js +13 -9
- package/dist/tui/cockpit/setup-cockpit.d.ts +2 -0
- package/dist/tui/cockpit/setup-cockpit.js +4 -0
- package/package.json +2 -1
- package/schema/config.schema.json +8 -1
- package/skills/tickmarkr-auto/SKILL.md +2 -2
- package/skills/tickmarkr-loop/SKILL.md +17 -5
- package/skills/tickmarkr-overseer/scripts/classify-vitest-log.sh +4 -1
|
@@ -7,6 +7,8 @@ import { isPidLive, STALE_MS } from "../../run/lock.js";
|
|
|
7
7
|
import { readTierLiveness, SUPERVISION_TIERS } from "../../run/supervision.js";
|
|
8
8
|
import { OperatorStateFold } from "../../run/operator-state.js";
|
|
9
9
|
export const OBSERVATION_INTERVAL_MS = 1_000;
|
|
10
|
+
/** Stat-to-fstat observations one poll spends on a journal a writer keeps appending to. */
|
|
11
|
+
export const APPEND_RACE_OBSERVATIONS = 3;
|
|
10
12
|
// Graph declarations have a separate 16 MiB bound; journal retention remains unchanged.
|
|
11
13
|
export const STORE_LIMITS = { graphBytes: 16 * 1024 * 1024, history: 256, historyBytes: 2 * 1024 * 1024, recordBytes: 1024 * 1024, readBytes: 1024 * 1024, subscribers: 64, metrics: 12, errors: 32 };
|
|
12
14
|
const errorText = (e) => e instanceof Error ? e.message : String(e);
|
|
@@ -48,6 +50,7 @@ export class JournalTail {
|
|
|
48
50
|
bytesRead = 0;
|
|
49
51
|
lastSuccessfulReadAt;
|
|
50
52
|
failure;
|
|
53
|
+
stale = false;
|
|
51
54
|
constructor(source, hooks = {}) {
|
|
52
55
|
this.source = source;
|
|
53
56
|
this.hooks = hooks;
|
|
@@ -65,18 +68,45 @@ export class JournalTail {
|
|
|
65
68
|
this.generation++;
|
|
66
69
|
this.hooks.reset?.();
|
|
67
70
|
}
|
|
71
|
+
/** A same-inode append between stat and fstat is retried; exhaustion keeps the last good snapshot, pending. */
|
|
68
72
|
poll(now = Date.now()) {
|
|
73
|
+
try {
|
|
74
|
+
for (let observation = 1; !this.observe(); observation++) {
|
|
75
|
+
if (observation === APPEND_RACE_OBSERVATIONS) {
|
|
76
|
+
this.stale = true;
|
|
77
|
+
this.failure = undefined;
|
|
78
|
+
return this.snapshot();
|
|
79
|
+
}
|
|
80
|
+
}
|
|
81
|
+
this.stale = false;
|
|
82
|
+
this.failure = undefined;
|
|
83
|
+
this.lastSuccessfulReadAt = now;
|
|
84
|
+
}
|
|
85
|
+
catch (e) {
|
|
86
|
+
this.failure = { source: this.source, error: errorText(e) };
|
|
87
|
+
}
|
|
88
|
+
return this.snapshot();
|
|
89
|
+
}
|
|
90
|
+
/** One stat-to-fstat observation. False when an append raced it; nothing was consumed or reset. */
|
|
91
|
+
observe() {
|
|
69
92
|
let fd;
|
|
70
93
|
try {
|
|
71
94
|
const st = statSync(this.source, { bigint: true });
|
|
72
95
|
if (!st.isFile())
|
|
73
96
|
throw new Error("journal source is not a regular file");
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
if (Number(st.size) > this.offset) {
|
|
97
|
+
const replaced = !this.st || fileIdentity(st) !== fileIdentity(this.st) || st.size < this.st.size || (st.size === this.st.size && stamp(st) !== stamp(this.st));
|
|
98
|
+
if (Number(st.size) > (replaced ? 0 : this.offset)) {
|
|
77
99
|
fd = openSync(this.source, "r");
|
|
78
|
-
|
|
100
|
+
const opened = fstatSync(fd, { bigint: true });
|
|
101
|
+
if (stamp(opened) !== stamp(st)) {
|
|
102
|
+
if (fileIdentity(opened) === fileIdentity(st) && opened.size > st.size)
|
|
103
|
+
return false;
|
|
79
104
|
throw new Error("journal changed before read; retry observation");
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
if (replaced)
|
|
108
|
+
this.reset();
|
|
109
|
+
if (fd !== undefined) {
|
|
80
110
|
let remaining = STORE_LIMITS.readBytes;
|
|
81
111
|
const buffer = Buffer.allocUnsafe(Math.min(64 * 1024, Number(st.size) - this.offset));
|
|
82
112
|
while (this.offset < Number(st.size) && remaining > 0) {
|
|
@@ -113,17 +143,12 @@ export class JournalTail {
|
|
|
113
143
|
}
|
|
114
144
|
}
|
|
115
145
|
this.st = st;
|
|
116
|
-
|
|
117
|
-
this.lastSuccessfulReadAt = now;
|
|
118
|
-
}
|
|
119
|
-
catch (e) {
|
|
120
|
-
this.failure = { source: this.source, error: errorText(e) };
|
|
146
|
+
return true;
|
|
121
147
|
}
|
|
122
148
|
finally {
|
|
123
149
|
if (fd !== undefined)
|
|
124
150
|
closeSync(fd);
|
|
125
151
|
}
|
|
126
|
-
return this.snapshot();
|
|
127
152
|
}
|
|
128
153
|
appendCarry(bytes) {
|
|
129
154
|
this.carryBytes += bytes.length;
|
|
@@ -137,7 +162,7 @@ export class JournalTail {
|
|
|
137
162
|
history: this.history.slice(), errors: this.errors.slice(), malformedCount: this.malformedCount,
|
|
138
163
|
pending: this.carryBytes ? { line: this.line + 1, bytes: this.carryBytes } : undefined,
|
|
139
164
|
backlogBytes: Math.max(0, Number(this.st?.size ?? 0) - this.offset),
|
|
140
|
-
status: this.failure ? "unreadable" : this.malformedCount ? "corrupt" : this.carryBytes ? "pending" : "readable",
|
|
165
|
+
status: this.failure ? "unreadable" : this.malformedCount ? "corrupt" : this.carryBytes || this.stale ? "pending" : "readable",
|
|
141
166
|
error: this.failure, lastSuccessfulReadAt: this.lastSuccessfulReadAt, bytesRead: this.bytesRead,
|
|
142
167
|
};
|
|
143
168
|
}
|
|
@@ -334,7 +334,7 @@ export function taskProjectionText(row, journal = []) {
|
|
|
334
334
|
`blocker ${blocker ? `${blocker.kind} ${at(evidence.blocker)}` : "none"}`,
|
|
335
335
|
`next ${blocker?.nextAction == null ? fieldReading(undefined) : `${blocker.nextAction} ${at(evidence.nextAction)}`}`,
|
|
336
336
|
...(harvest?.suspectedStalledHarvest ? [`${STALL_MARKER} · launch ${at(evidence.launch)} · ${harvest.nudgeFailures} nudge failed ${at(evidence.nudge)} · ${harvest.pageCount} paged ${at(evidence.page)} · no worker-result`] : []),
|
|
337
|
-
].
|
|
337
|
+
].join(" · ");
|
|
338
338
|
}
|
|
339
339
|
/**
|
|
340
340
|
* Identity, phase, and build are unrecorded until a dispatch. Append the blocker and next action
|
|
@@ -354,7 +354,7 @@ function undispatchedProjectionText(row, journal) {
|
|
|
354
354
|
MISSING_EVIDENCE,
|
|
355
355
|
...(showBlocker ? [`blocker ${blocker.kind} ${at(evidence.blocker)}`] : []),
|
|
356
356
|
...(showNext ? [`next ${blocker.nextAction} ${at(evidence.nextAction)}`] : []),
|
|
357
|
-
].
|
|
357
|
+
].join(" · ");
|
|
358
358
|
}
|
|
359
359
|
/**
|
|
360
360
|
* Correlate each already-derived field with the row that can have produced that value. This is
|
|
@@ -93,8 +93,9 @@ export interface TaskProjection {
|
|
|
93
93
|
readonly build: ProjectionField;
|
|
94
94
|
readonly blocker: ProjectionField;
|
|
95
95
|
readonly nextAction: ProjectionField;
|
|
96
|
-
/** OBS-1048: present only when the newest launched attempt has failed nudges, pages and no worker-result.
|
|
96
|
+
/** OBS-1048: present only when the newest launched attempt has failed nudges, pages and no worker-result. */
|
|
97
97
|
readonly stalled?: ProjectionField;
|
|
98
|
+
/** OBS-1104: a task with no recorded dispatch collapses its line to one missing-evidence clause. */
|
|
98
99
|
readonly neverDispatched?: true;
|
|
99
100
|
}
|
|
100
101
|
export declare const STALL_MARKER = "\u26A0 stalled harvest suspected";
|
|
@@ -251,8 +251,11 @@ export function projectRunTasks(snapshot, rows, graph, decisions, runId) {
|
|
|
251
251
|
const blocker = { label: `blocker ${blk ? `${blk.kind}${blk.diagnostic ? ` · ${blk.diagnostic}` : ""}` : "none"}`, ...(blk && blockerRow ? { line: blockerRow.line } : {}) };
|
|
252
252
|
const nextAction = { label: `next ${blk?.nextAction ?? "none"}`, ...(blk?.nextAction && blockerRow ? { line: blockerRow.line } : {}) };
|
|
253
253
|
// OBS-1048 harvest, chronological: the newest worker-launch opens the attempt; a later worker-result retires it.
|
|
254
|
-
let launched
|
|
255
|
-
|
|
254
|
+
let launched;
|
|
255
|
+
let launchedAttempt;
|
|
256
|
+
let returned = false;
|
|
257
|
+
const nudges = [];
|
|
258
|
+
const pages = [];
|
|
256
259
|
// Finding 1: only rows of the launched attempt count; a row without an attempt belongs to it (derive.ts attemptHarvests).
|
|
257
260
|
const ofLaunched = (d) => launched !== undefined && (ordinal(d.attempt) ?? launchedAttempt) === launchedAttempt;
|
|
258
261
|
for (const r of own) {
|
|
@@ -389,15 +392,16 @@ function undispatchedProjectionLine(p) {
|
|
|
389
392
|
].join(" · ");
|
|
390
393
|
}
|
|
391
394
|
/**
|
|
392
|
-
* Rows the projection panel paints
|
|
393
|
-
*
|
|
394
|
-
*
|
|
395
|
+
* Rows the projection panel paints: each task's clause at its own wrapped height (OBS-1193). A
|
|
396
|
+
* never-dispatched clause is shorter than the field line it replaces, so the block is shorter too;
|
|
397
|
+
* the shell's content counter reads that real height and the pinned run frames record it.
|
|
395
398
|
*/
|
|
396
399
|
function projectionBlockRows(projections, wrap) {
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
400
|
+
return projections.flatMap((p) => wrap(projectionLine(p)).map((line, i) => ({
|
|
401
|
+
key: `${p.taskId}:${i}`,
|
|
402
|
+
line,
|
|
403
|
+
strong: p.stalled !== undefined && i === 0,
|
|
404
|
+
})));
|
|
401
405
|
}
|
|
402
406
|
/** The Run body: the approved board (BD-1) over the fold, then the selected task's detail panels. */
|
|
403
407
|
export function RunView({ snapshot, rows, page, graph, decisions, session, columns, run, now = Date.now }) {
|
|
@@ -73,6 +73,8 @@ export type ParkedDecision = {
|
|
|
73
73
|
readonly tombstone: boolean;
|
|
74
74
|
/** OBS-1178: the park's `<line>@<ts>` token — the confirmed write binds to it via `--park`. */
|
|
75
75
|
readonly park?: string;
|
|
76
|
+
/** OBS-1202: a stall park's recorded reap failure — the one stall park recheck may release. */
|
|
77
|
+
readonly reapFailure?: string;
|
|
76
78
|
};
|
|
77
79
|
/** The parks a verb can release — the rows the surface draws live verbs on. */
|
|
78
80
|
export declare function actionableDecisions(decisions: readonly ParkedDecision[]): readonly ParkedDecision[];
|
|
@@ -607,6 +607,7 @@ export function deriveParkedDecisions(journal) {
|
|
|
607
607
|
failedGate,
|
|
608
608
|
tombstone: isTombstonePark(kind, reason),
|
|
609
609
|
...(typeof parked?.ts === "string" ? { park: bindingToken({ line: lines[parkedIndex], ts: parked.ts }) } : {}),
|
|
610
|
+
...(kind === "stall" && typeof parked?.data.reapFailure === "string" ? { reapFailure: parked.data.reapFailure } : {}),
|
|
610
611
|
});
|
|
611
612
|
}
|
|
612
613
|
return decisions;
|
|
@@ -623,6 +624,9 @@ export function initialSetupDecisionsSession() {
|
|
|
623
624
|
export function setupDecisionVerbs(decision) {
|
|
624
625
|
if (decision.tombstone)
|
|
625
626
|
return [];
|
|
627
|
+
// OBS-1202: the production table's census-recovery row; an ordinary stall stays approve-only.
|
|
628
|
+
if (decision.kind === "stall" && decision.reapFailure !== undefined)
|
|
629
|
+
return ["approve", "recheck"];
|
|
626
630
|
if (decision.kind !== "gate-fail")
|
|
627
631
|
return ["approve"];
|
|
628
632
|
if (decision.failedGate === undefined)
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "tickmarkr",
|
|
3
|
-
"version": "2.6.
|
|
3
|
+
"version": "2.6.4",
|
|
4
4
|
"description": "Spec in, verified work out.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "MIT",
|
|
@@ -81,6 +81,7 @@
|
|
|
81
81
|
"scripts/emit-schema.ts",
|
|
82
82
|
"scripts/probe-rig.mjs",
|
|
83
83
|
"scripts/run-ci-vitest.sh",
|
|
84
|
+
"scripts/vitest-lease.ts",
|
|
84
85
|
"specs/export-selftest.spec.md"
|
|
85
86
|
],
|
|
86
87
|
"prefixes": [
|
|
@@ -10,7 +10,6 @@
|
|
|
10
10
|
"type": "boolean"
|
|
11
11
|
},
|
|
12
12
|
"repairSelection": {
|
|
13
|
-
"default": false,
|
|
14
13
|
"type": "boolean"
|
|
15
14
|
},
|
|
16
15
|
"taskExecutionLimitMs": {
|
|
@@ -619,6 +618,14 @@
|
|
|
619
618
|
"exclusiveMinimum": 0,
|
|
620
619
|
"maximum": 9007199254740991
|
|
621
620
|
},
|
|
621
|
+
"evidenceQuotaBytes": {
|
|
622
|
+
"type": "integer",
|
|
623
|
+
"minimum": 0,
|
|
624
|
+
"maximum": 9007199254740991
|
|
625
|
+
},
|
|
626
|
+
"repairSelection": {
|
|
627
|
+
"type": "boolean"
|
|
628
|
+
},
|
|
622
629
|
"byShape": {
|
|
623
630
|
"type": "object",
|
|
624
631
|
"propertyNames": {
|
|
@@ -92,7 +92,7 @@ After sending, **confirm delivery** by reading the target pane and verifying the
|
|
|
92
92
|
1. **Prepare** — confirm the target list. Run the [binary preflight](#binary-preflight-before-compile-or-run). Check `git status`, confirm no tickmarkr run is active, and work from a non-main branch.
|
|
93
93
|
2. **Compile** — run `tickmarkr compile <spec-or-directory>`. Fix source-spec defects instead of editing the generated graph.
|
|
94
94
|
3. **Plan** — run `tickmarkr plan`. Review routes, capability-floor warnings, and human gates before execution.
|
|
95
|
-
4. **Run** — run `tickmarkr run`. A watch ending the seat's turn is no watch: keep a **blocking journal consumer** alive for the run's terminal events — the shipped watcher below, or a foreground `until grep` on the run's terminal events — and ensure it is re-armed at most every twenty minutes. Never rely on a `Monitor`-only wake. Watch the run journal rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every
|
|
96
|
-
5. **Verify and consolidate** — continue only after a green run. A run is green when the run-end event exists in the journal, the tip verify is not "failed",
|
|
95
|
+
4. **Run** — run `tickmarkr run`. A watch ending the seat's turn is no watch: keep a **blocking journal consumer** alive for the run's terminal events — the shipped watcher below, or a foreground `until grep` on the run's terminal events — and ensure it is re-armed at most every twenty minutes. Never rely on a `Monitor`-only wake. Watch the run journal rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every execution clause for you; the debt clause is yours — read CURRENT `tickmarkr status <runId>` (step 5). Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the relevant agent session.
|
|
96
|
+
5. **Verify and consolidate** — continue only after a green run. A run is green when the run-end event exists in the journal, the tip verify is not "failed", the summary's `failed`, `human`, `blocked` and `pending` buckets are all empty, and CURRENT `tickmarkr status <runId>` reads its owed checks outstanding empty AND known (`outstanding 0`) — a run with a parked task is partial, not green. Empty execution buckets alone are not green either (D-660): every bucket empty and the tip passed, but an operator waived one review, leaves one accepted-risk review check owed — status reads `outstanding 1 (T7 review)` and `run`/`resume` still exit 0 on execution alone, so the run is execution complete, not green. It turns green only when a `tickmarkr verify --record <runId>` discharge moves CURRENT status to `outstanding 0`, including a discharge landing after run-end — the historical run-end record keeps the old count, so never read debt from it; `outstanding unknown` is never green. Tickmarkr consolidates accepted work on `tickmarkr/<runId>` and never signs off to the main branch. A human controls any later release merge.
|
|
97
97
|
6. **Record** — `tickmarkr report <runId> --md` prints Markdown to stdout; redirect explicitly beside the spec (for example `tickmarkr report <runId> --md > feature.record.md`) and commit the execution record when the repository tracks those records.
|
|
98
98
|
7. **Continue** — move to the next requested target. If a target fails or is parked, stop with the journal evidence rather than silently skipping it.
|
|
@@ -88,8 +88,8 @@ When spawning consultants (agents gathering synthesis input for decisions like S
|
|
|
88
88
|
1. **Prepare** — start from the requested spec. Run the [binary preflight](#binary-preflight-before-compile-or-run). Check `git status`, confirm no tickmarkr run is active, and work from a non-main branch.
|
|
89
89
|
2. **Compile** — run `tickmarkr compile <spec>`. Correct compilation errors in the spec, never in the generated graph.
|
|
90
90
|
3. **Plan** — run `tickmarkr plan`. Review the routing table, capability-floor warnings, and every human gate, including work that each gate blocks.
|
|
91
|
-
4. **Run** — run `tickmarkr run`. A watch ending the seat's turn is no watch: keep a **blocking journal consumer** alive for the run's terminal events — the shipped watcher below, or a foreground `until grep` on the run's terminal events — and ensure it is re-armed at most every twenty minutes. Never rely on a `Monitor`-only wake. Watch the run journal rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every
|
|
92
|
-
5. **Verify and consolidate** — accept only a green run. A run is green when the run-end event exists in the journal, the tip verify is not "failed",
|
|
91
|
+
4. **Run** — run `tickmarkr run`. A watch ending the seat's turn is no watch: keep a **blocking journal consumer** alive for the run's terminal events — the shipped watcher below, or a foreground `until grep` on the run's terminal events — and ensure it is re-armed at most every twenty minutes. Never rely on a `Monitor`-only wake. Watch the run journal rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every execution clause for you; the debt clause is yours — read CURRENT `tickmarkr status <runId>` (step 5). Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the agent session; do not turn them into proxy questions.
|
|
92
|
+
5. **Verify and consolidate** — accept only a green run. A run is green when the run-end event exists in the journal, the tip verify is not "failed", the summary's `failed`, `human`, `blocked` and `pending` buckets are all empty, and CURRENT `tickmarkr status <runId>` reads its owed checks outstanding empty AND known (`outstanding 0`) — a run with a parked task is partial, not green. Empty execution buckets alone are not green either (D-660): every bucket empty and the tip passed, but an operator waived one review, leaves one accepted-risk review check owed — status reads `outstanding 1 (T7 review)` and `run`/`resume` still exit 0 on execution alone, so the run is execution complete, not green. It turns green only when a `tickmarkr verify --record <runId>` discharge moves CURRENT status to `outstanding 0`, including a discharge landing after run-end — the historical run-end record keeps the old count, so never read debt from it; `outstanding unknown` is never green. Tickmarkr consolidates accepted task work on `tickmarkr/<runId>`; it never signs off to the main branch. A human may later merge that integration branch through the repository's normal release process.
|
|
93
93
|
6. **Record** — `tickmarkr report <runId> --md` prints Markdown to stdout. Redirect it explicitly beside the source spec (for example `tickmarkr report <runId> --md > feature.record.md`) and commit the execution record when the repository tracks those records. Then [stand down](#stand-down-mission-end-and-retirement).
|
|
94
94
|
|
|
95
95
|
## Cockpit, parked decisions and printed twins
|
|
@@ -122,14 +122,26 @@ own bound failure token: status prints `failed — T3 — failure <line>@<ts>`,
|
|
|
122
122
|
`tickmarkr approve <runId> T3 --recheck --park <line>@<ts>` re-gates its landed commits. Preserve resume refusals and repair the named source/config issue
|
|
123
123
|
(including deny/prefer conflicts); never edit the compiled graph to force a result. Resume
|
|
124
124
|
makes CURRENT TIP PENDING; historical GATES RAN does not prove completion. The completed
|
|
125
|
-
case has 3/3 recorded merges, a latest run-end, a nonfailed known tip result
|
|
126
|
-
`failed`, `human`, `blocked`, `pending` buckets
|
|
125
|
+
case has 3/3 recorded merges, a latest run-end, a nonfailed known tip result, empty
|
|
126
|
+
`failed`, `human`, `blocked`, `pending` buckets, and CURRENT status reading `outstanding 0`
|
|
127
|
+
— owed checks outstanding empty AND known. Replay D-660 before calling it green: the same
|
|
128
|
+
3/3 with every bucket empty, after T2's review was waived, reads `outstanding 1 (T2 review)`;
|
|
129
|
+
`resume` exits 0 and its final line says `execution complete; outstanding 1 (T2 review)`, not
|
|
130
|
+
`verified`. It is green only once `tickmarkr verify --record <runId>` discharges that check and
|
|
131
|
+
CURRENT status reads `outstanding 0` — even after run-end, whose record still carries 1.
|
|
132
|
+
`outstanding unknown` (a legacy waiver) is never green. A mismatched graph is “not comparable.”
|
|
127
133
|
|
|
128
134
|
Run offers only validated park verbs: human/attempt-cap/other non-gate parks allow approve;
|
|
129
135
|
infra allows approve or `--recheck`; review gate-fail allows `--waive`, `--uphold` or
|
|
130
136
|
`--recheck`; other gate-fail allows waive/recheck. Waive satisfies only the identified
|
|
131
137
|
failed gate, uphold funds a fixed attempt carrying review findings, and recheck reruns the
|
|
132
|
-
declared battery without satisfying a gate.
|
|
138
|
+
declared battery without satisfying a gate. A stall park that recorded a `reapFailure`
|
|
139
|
+
(unreadable or surviving worker census) allows approve or `--recheck --park <line>@<ts>`:
|
|
140
|
+
recheck re-verifies that attempt's owned census and gates its harvested commits with no
|
|
141
|
+
worker only with an explicitly recorded empty survivors array (`[]`) — a missing,
|
|
142
|
+
unreadable or surviving census
|
|
143
|
+
re-parks the stall under a new token — while plain approve dispatches a worker. An
|
|
144
|
+
ordinary stall park (no `reapFailure`) stays approve-only; `--recheck` refuses it. Attempt-cap approval resets the budget while
|
|
133
145
|
retaining routing exclusions. Tombstones and failures without identified gate evidence
|
|
134
146
|
are diagnostic-only. Decisions cannot be undone; stale or duplicate decisions refuse.
|
|
135
147
|
|
|
@@ -44,7 +44,10 @@ awk -v esc="$esc" '
|
|
|
44
44
|
gsub(/\^\[\[[0-9;]*m/, "", line); gsub(esc "\\[[0-9;]*m", "", line)
|
|
45
45
|
sub(/^[^\t]*\t[^\t]*\t[0-9T:.-]+Z[ ]?/, "", line)
|
|
46
46
|
}
|
|
47
|
-
|
|
47
|
+
# The Vitest timeout message carries its millisecond count; a test that PRINTS a fingerprint normalized
|
|
48
|
+
# to #ms (the daemon OBS-1106 notifications) is output, not a timed-out test. A real timeout also fails
|
|
49
|
+
# its test, so the summary failed count still reds it.
|
|
50
|
+
line ~ /(Test|Hook) timed out in [0-9]+ ?ms/ { timedout++ }
|
|
48
51
|
line ~ /ERROR: Coverage for .* does not meet .*threshold/ { coverage++ }
|
|
49
52
|
line ~ /^npm (error|ERR!) signal / { signals++ }
|
|
50
53
|
line ~ /^(⎯)+ Unhandled Errors (⎯)+[ \t]*$/ { if (armed) close_block(); opening = 1; next }
|