tickmarkr 2.1.8 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/commands/approve.d.ts +24 -0
- package/dist/cli/commands/approve.js +121 -34
- package/dist/compile/collateral.js +149 -4
- package/dist/compile/native.js +16 -0
- package/dist/drivers/herdr.d.ts +7 -5
- package/dist/drivers/herdr.js +142 -11
- package/dist/drivers/types.d.ts +5 -6
- package/dist/drivers/types.js +2 -48
- package/dist/run/daemon.d.ts +10 -6
- package/dist/run/daemon.js +275 -22
- package/dist/run/journal.d.ts +16 -0
- package/dist/run/journal.js +51 -6
- package/dist/tui/cockpit/setup-cockpit.d.ts +8 -10
- package/dist/tui/cockpit/setup-cockpit.js +92 -53
- package/package.json +1 -1
- package/skills/tickmarkr-auto/SKILL.md +1 -1
- package/skills/tickmarkr-loop/SKILL.md +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +202 -24
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +109 -4
- package/skills/tickmarkr-overseer/scripts/watch-journal.sh +137 -0
|
@@ -4,8 +4,8 @@ import { userInfo } from "node:os";
|
|
|
4
4
|
import { join } from "node:path";
|
|
5
5
|
import { cloneElement, createElement, useState, } from "react";
|
|
6
6
|
import { parse } from "yaml";
|
|
7
|
-
import { approve } from "../../cli/commands/approve.js";
|
|
8
|
-
import { ATTEMPT_CAP_RELEASE, GATE_SATISFIED_RELEASE, Journal, REVIEW_UPHELD_RELEASE, } from "../../run/journal.js";
|
|
7
|
+
import { approvalDispositionForRelease, approvalEnactment, approvalRunOwner, approve, } from "../../cli/commands/approve.js";
|
|
8
|
+
import { ATTEMPT_CAP_RELEASE, GATE_SATISFIED_RELEASE, Journal, RECHECK_RELEASE, REVIEW_UPHELD_RELEASE, } from "../../run/journal.js";
|
|
9
9
|
import { BANNER, GLYPHS, PLAIN_BANNER, } from "../../brand.js";
|
|
10
10
|
import { DEFAULT_CONFIG, INTEGRITY_FLOOR_SHAPES, TIER_RANK, TickmarkrConfigSchema, unifiedYamlDiff, } from "../../config/config.js";
|
|
11
11
|
import { GATE_NAMES, SHAPES } from "../../graph/schema.js";
|
|
@@ -560,6 +560,20 @@ export function actionableDecisions(decisions) {
|
|
|
560
560
|
* replay is the production one (`Journal.replayStatuses`) so this surface and
|
|
561
561
|
* `tickmarkr status` can never disagree about what is parked.
|
|
562
562
|
*/
|
|
563
|
+
function failedGateForNewestPark(events, taskId, parkedIndex) {
|
|
564
|
+
for (let index = parkedIndex - 1; index >= 0; index -= 1) {
|
|
565
|
+
const event = events[index];
|
|
566
|
+
if (event.taskId !== taskId)
|
|
567
|
+
continue;
|
|
568
|
+
if (event.event === "task-approved" || event.event === "task-human" || event.event === "task-dispatch")
|
|
569
|
+
return undefined;
|
|
570
|
+
if (event.event === "gate-result" && event.data.pass === false
|
|
571
|
+
&& typeof event.data.gate === "string" && GATE_NAMES.includes(event.data.gate)) {
|
|
572
|
+
return event.data.gate;
|
|
573
|
+
}
|
|
574
|
+
}
|
|
575
|
+
return undefined;
|
|
576
|
+
}
|
|
563
577
|
export function deriveParkedDecisions(journal) {
|
|
564
578
|
const statuses = journal.replayStatuses();
|
|
565
579
|
const events = journal.read();
|
|
@@ -579,12 +593,7 @@ export function deriveParkedDecisions(journal) {
|
|
|
579
593
|
parkedIndex = index;
|
|
580
594
|
}
|
|
581
595
|
const parked = parkedIndex >= 0 ? events[parkedIndex] : undefined;
|
|
582
|
-
|
|
583
|
-
// gate-result before the newest task-human controls what approve records
|
|
584
|
-
// and whether uphold exists at all. Never inferred from the park's prose.
|
|
585
|
-
const failedGate = events.slice(0, Math.max(0, parkedIndex)).reverse().find((event) => event.event === "gate-result" && event.taskId === taskId && event.data.pass === false
|
|
586
|
-
&& typeof event.data.gate === "string"
|
|
587
|
-
&& GATE_NAMES.includes(event.data.gate))?.data.gate;
|
|
596
|
+
const failedGate = failedGateForNewestPark(events, taskId, parkedIndex);
|
|
588
597
|
const kind = typeof parked?.data.kind === "string" ? parked.data.kind : "human-gate";
|
|
589
598
|
const reason = typeof parked?.data.reason === "string" ? parked.data.reason : undefined;
|
|
590
599
|
decisions.push({
|
|
@@ -610,8 +619,17 @@ export function initialSetupDecisionsSession() {
|
|
|
610
619
|
* inert and the keybar does not advertise it — the surface never promises a
|
|
611
620
|
* decision the command must refuse.
|
|
612
621
|
*/
|
|
622
|
+
export function setupDecisionVerbs(decision) {
|
|
623
|
+
if (decision.tombstone)
|
|
624
|
+
return [];
|
|
625
|
+
if (decision.kind !== "gate-fail")
|
|
626
|
+
return ["approve"];
|
|
627
|
+
if (decision.failedGate === undefined)
|
|
628
|
+
return [];
|
|
629
|
+
return decision.failedGate === "review" ? ["waive", "uphold", "recheck"] : ["waive", "recheck"];
|
|
630
|
+
}
|
|
613
631
|
export function upholdAvailable(decision) {
|
|
614
|
-
return decision.
|
|
632
|
+
return setupDecisionVerbs(decision).includes("uphold");
|
|
615
633
|
}
|
|
616
634
|
/**
|
|
617
635
|
* The decisions surface's whole key grammar: ↑↓ move, a approve, u uphold,
|
|
@@ -650,25 +668,30 @@ export function applySetupDecisionsKey(session, event, parked) {
|
|
|
650
668
|
},
|
|
651
669
|
};
|
|
652
670
|
}
|
|
653
|
-
|
|
671
|
+
const verb = { a: "approve", w: "waive", u: "uphold", r: "recheck" }[event.input ?? ""];
|
|
672
|
+
if (verb !== undefined) {
|
|
654
673
|
const decision = decisions[selection];
|
|
655
|
-
if (decision === undefined)
|
|
656
|
-
return { session };
|
|
657
|
-
if (event.input === "u" && !upholdAvailable(decision))
|
|
674
|
+
if (decision === undefined || !setupDecisionVerbs(decision).includes(verb))
|
|
658
675
|
return { session };
|
|
659
676
|
return {
|
|
660
677
|
session: {
|
|
661
678
|
...session,
|
|
662
679
|
selection,
|
|
663
|
-
confirming: {
|
|
664
|
-
verb: event.input === "a" ? "approve" : "uphold",
|
|
665
|
-
taskId: decision.taskId,
|
|
666
|
-
},
|
|
680
|
+
confirming: { verb, taskId: decision.taskId },
|
|
667
681
|
},
|
|
668
682
|
};
|
|
669
683
|
}
|
|
670
684
|
return { session };
|
|
671
685
|
}
|
|
686
|
+
const decisionFlag = (verb) => {
|
|
687
|
+
if (verb === "waive")
|
|
688
|
+
return ["--waive"];
|
|
689
|
+
if (verb === "uphold")
|
|
690
|
+
return ["--uphold"];
|
|
691
|
+
if (verb === "recheck")
|
|
692
|
+
return ["--recheck"];
|
|
693
|
+
return [];
|
|
694
|
+
};
|
|
672
695
|
/**
|
|
673
696
|
* THE write. It calls the production `approve` command — validation, fail-closed
|
|
674
697
|
* refusals and the journal append are inherited, never re-implemented — and then
|
|
@@ -680,7 +703,7 @@ export async function executeSetupDecision(command, { cwd, runId, by }) {
|
|
|
680
703
|
const message = await approve([
|
|
681
704
|
runId,
|
|
682
705
|
command.taskId,
|
|
683
|
-
...(command.verb
|
|
706
|
+
...decisionFlag(command.verb),
|
|
684
707
|
"--by",
|
|
685
708
|
by,
|
|
686
709
|
], cwd);
|
|
@@ -693,11 +716,12 @@ export async function executeSetupDecision(command, { cwd, runId, by }) {
|
|
|
693
716
|
break;
|
|
694
717
|
}
|
|
695
718
|
}
|
|
719
|
+
const disposition = approvalDispositionForRelease(release);
|
|
696
720
|
return {
|
|
697
721
|
ok: true,
|
|
698
722
|
command,
|
|
699
723
|
message,
|
|
700
|
-
write: { taskId: command.taskId, verb: command.verb, release, by },
|
|
724
|
+
write: { taskId: command.taskId, verb: command.verb, release, disposition, by },
|
|
701
725
|
};
|
|
702
726
|
}
|
|
703
727
|
catch (error) {
|
|
@@ -726,52 +750,58 @@ function decisionRowText(decision) {
|
|
|
726
750
|
* append — the release markers are the command's own constants, so this text
|
|
727
751
|
* cannot drift from the write. An effect the operator cannot read here is an
|
|
728
752
|
* effect this surface does not have.
|
|
753
|
+
*
|
|
754
|
+
* WHO enacts it is the command's own sentence (`approvalEnactment`), read from
|
|
755
|
+
* the same run-owner fact: predicting "the next resume" over a LIVE run was
|
|
756
|
+
* true only while the daemon read approvals once at startup. It now sweeps them
|
|
757
|
+
* at every task boundary, so that prediction would send the operator to start a
|
|
758
|
+
* second run in this repository — forbidden — for work already scheduled.
|
|
729
759
|
*/
|
|
730
|
-
function decisionEffectLines(command, decision) {
|
|
731
|
-
if (command.verb === "
|
|
760
|
+
function decisionEffectLines(command, decision, run) {
|
|
761
|
+
if (command.verb === "waive") {
|
|
762
|
+
const gate = decision?.failedGate;
|
|
763
|
+
return gate === undefined ? ["effect none — waive refuses without a failed gate"] : [
|
|
764
|
+
`effect disposition waive-gate; appends release ${GATE_SATISFIED_RELEASE} for gate ${gate}`,
|
|
765
|
+
`it marks gate ${gate} satisfied; it does NOT mark the task done`,
|
|
766
|
+
approvalEnactment("waive-gate", run),
|
|
767
|
+
];
|
|
768
|
+
}
|
|
769
|
+
if (command.verb === "recheck") {
|
|
732
770
|
return [
|
|
733
|
-
`effect
|
|
734
|
-
|
|
735
|
-
"
|
|
771
|
+
`effect disposition re-dispatch; appends release ${RECHECK_RELEASE}`,
|
|
772
|
+
"it marks no gate satisfied",
|
|
773
|
+
approvalEnactment("re-dispatch", run),
|
|
736
774
|
];
|
|
737
775
|
}
|
|
738
|
-
if (
|
|
739
|
-
if (decision.failedGate === undefined) {
|
|
740
|
-
// The command's own refusal, predicted: no write is promised here.
|
|
741
|
-
return [
|
|
742
|
-
"effect none — the command refuses: parked on gate-fail"
|
|
743
|
-
+ " but no failed gate result is recorded",
|
|
744
|
-
];
|
|
745
|
-
}
|
|
746
|
-
const gate = decision.failedGate;
|
|
776
|
+
if (command.verb === "uphold") {
|
|
747
777
|
return [
|
|
748
|
-
`effect
|
|
749
|
-
|
|
750
|
-
`it marks gate ${gate} satisfied; it does NOT mark the task done`,
|
|
778
|
+
`effect disposition fund-fixed-attempt; appends release ${REVIEW_UPHELD_RELEASE}`,
|
|
779
|
+
approvalEnactment("fund-fixed-attempt", run),
|
|
751
780
|
];
|
|
752
781
|
}
|
|
782
|
+
if (decision?.kind === "gate-fail") {
|
|
783
|
+
return ["effect none — approve is not offered on gate-fail parks; choose waive, recheck or uphold"];
|
|
784
|
+
}
|
|
753
785
|
if (decision?.kind === ATTEMPT_CAP_RELEASE) {
|
|
754
786
|
return [
|
|
755
|
-
`effect
|
|
756
|
-
|
|
757
|
-
"it does NOT mark the task done or pass any gate",
|
|
787
|
+
`effect disposition fresh-budget; appends release ${ATTEMPT_CAP_RELEASE}`,
|
|
788
|
+
approvalEnactment("fresh-budget", run),
|
|
758
789
|
];
|
|
759
790
|
}
|
|
760
791
|
return [
|
|
761
|
-
|
|
762
|
-
|
|
763
|
-
"it does NOT mark the task done or pass any gate",
|
|
792
|
+
`effect disposition dispatch; appends one task-approved event with no release`,
|
|
793
|
+
approvalEnactment("dispatch", run),
|
|
764
794
|
];
|
|
765
795
|
}
|
|
766
796
|
/**
|
|
767
797
|
* The confirm inset's lines: the task, the actor, the effect and the file,
|
|
768
798
|
* stated before the single key that can write.
|
|
769
799
|
*/
|
|
770
|
-
export function setupDecisionConfirmLines(command, decision, actor, journalFile) {
|
|
800
|
+
export function setupDecisionConfirmLines(command, decision, actor, journalFile, run) {
|
|
771
801
|
return [
|
|
772
802
|
`task ${command.taskId}${decision === undefined ? "" : ` · ${decision.kind} · attempt ${decision.attempts}`}`,
|
|
773
803
|
`actor ${actor}`,
|
|
774
|
-
...decisionEffectLines(command, decision),
|
|
804
|
+
...decisionEffectLines(command, decision, run),
|
|
775
805
|
`file ${journalFile} · append only`,
|
|
776
806
|
`y ${command.verb} · n cancel`,
|
|
777
807
|
];
|
|
@@ -783,8 +813,13 @@ function setupDecisionsKeybar(session, decisions) {
|
|
|
783
813
|
if (decisions.length === 0)
|
|
784
814
|
return "q Quit";
|
|
785
815
|
const selected = Math.min(session.selection, decisions.length - 1);
|
|
786
|
-
const
|
|
787
|
-
|
|
816
|
+
const verbs = setupDecisionVerbs(decisions[selected]).map((verb) => ({
|
|
817
|
+
approve: "a Approve",
|
|
818
|
+
waive: "w Waive",
|
|
819
|
+
uphold: "u Uphold",
|
|
820
|
+
recheck: "r Recheck",
|
|
821
|
+
})[verb]).join(" · ");
|
|
822
|
+
return `↑↓ Move · ${verbs} · q Quit`;
|
|
788
823
|
}
|
|
789
824
|
/**
|
|
790
825
|
* What the tab is, drawn on the tab it names. The operator's v1.83 UAT read
|
|
@@ -793,7 +828,7 @@ function setupDecisionsKeybar(session, decisions) {
|
|
|
793
828
|
* keybar hint also draw.
|
|
794
829
|
*/
|
|
795
830
|
const DECISIONS_TAB_LINE = `tab ${DECISIONS_TAB_HINT} · journal-parked human decisions`
|
|
796
|
-
+ " · approve
|
|
831
|
+
+ " · approve waive uphold and recheck are its only writes";
|
|
797
832
|
/**
|
|
798
833
|
* What the section says when no park needs the operator. It names its own
|
|
799
834
|
* trigger and both verbs, so an operator who arrives at a quiet surface learns
|
|
@@ -801,8 +836,8 @@ const DECISIONS_TAB_LINE = `tab ${DECISIONS_TAB_HINT} · journal-parked human de
|
|
|
801
836
|
*/
|
|
802
837
|
const DECISIONS_EMPTY_STATE = [
|
|
803
838
|
"nothing needs you now — no task is parked on a human gate",
|
|
804
|
-
"a task that parks on one appears here with its
|
|
805
|
-
+ " approve
|
|
839
|
+
"a task that parks on one appears here with its decisions —"
|
|
840
|
+
+ " approve dispatches, waive accepts a gate, uphold funds a fix, recheck re-runs",
|
|
806
841
|
];
|
|
807
842
|
/**
|
|
808
843
|
* The decisions surface. Every row it draws — decisions, the confirm inset,
|
|
@@ -811,15 +846,16 @@ const DECISIONS_EMPTY_STATE = [
|
|
|
811
846
|
* neither overflow its row nor be cut mid-cluster. The one unclipped element
|
|
812
847
|
* is `notice`: the production command's own words, verbatim.
|
|
813
848
|
*/
|
|
814
|
-
export function SetupDecisionsSurface({ decisions, session, columns, actor, journalFile, }) {
|
|
849
|
+
export function SetupDecisionsSurface({ decisions, session, columns, actor, journalFile, run, }) {
|
|
815
850
|
const inner = Math.max(1, Math.floor(columns) - 4);
|
|
816
851
|
const actionable = actionableDecisions(decisions);
|
|
817
852
|
const tombstones = decisions.filter((decision) => decision.tombstone);
|
|
818
853
|
const selected = Math.min(session.selection, Math.max(0, actionable.length - 1));
|
|
819
854
|
const confirming = session.confirming;
|
|
820
|
-
return (_jsxs(Box, { flexDirection: "column", width: columns, children: [_jsxs(Panel, { title: `${DECISIONS_TAB_TITLE} · ${decisions.length} parked`, focused: true, children: [_jsx(BodyText, { emphasis: "dim", children: fitCells(DECISIONS_TAB_LINE, inner) }), actionable.map((decision, index) => (_jsx(BodyText, { emphasis: index === selected ? "strong" : "normal", children: fitCells(`${index === selected ? `${GLYPHS.pointer} ` : " "}${decisionRowText(decision)}`, inner) }, decision.taskId))), actionable.length === 0 && DECISIONS_EMPTY_STATE.map((line) => (_jsx(BodyText, { children: fitCells(line, inner) }, line))), tombstones.map((decision) => (_jsx(BodyText, { emphasis: "dim", children: fitCells(` read-only · permanent by design · ${decisionRowText(decision)}`, inner) }, decision.taskId)))] }), confirming !== null && (_jsx(Panel, { title: `CONFIRM ${confirming.verb.toUpperCase()}`, focused: true, children: setupDecisionConfirmLines(confirming, decisions.find((decision) => decision.taskId === confirming.taskId), actor, journalFile).map((line) => (_jsx(BodyText, { children: fitCells(line, inner) }, line))) })), session.notice !== null && _jsx(BodyText, { children: session.notice }), _jsx(Panel, { title: "PENDING WRITES", children: session.writes.length === 0
|
|
855
|
+
return (_jsxs(Box, { flexDirection: "column", width: columns, children: [_jsxs(Panel, { title: `${DECISIONS_TAB_TITLE} · ${decisions.length} parked`, focused: true, children: [_jsx(BodyText, { emphasis: "dim", children: fitCells(DECISIONS_TAB_LINE, inner) }), actionable.map((decision, index) => (_jsx(BodyText, { emphasis: index === selected ? "strong" : "normal", children: fitCells(`${index === selected ? `${GLYPHS.pointer} ` : " "}${decisionRowText(decision)}`, inner) }, decision.taskId))), actionable.length === 0 && DECISIONS_EMPTY_STATE.map((line) => (_jsx(BodyText, { children: fitCells(line, inner) }, line))), tombstones.map((decision) => (_jsx(BodyText, { emphasis: "dim", children: fitCells(` read-only · permanent by design · ${decisionRowText(decision)}`, inner) }, decision.taskId)))] }), confirming !== null && (_jsx(Panel, { title: `CONFIRM ${confirming.verb.toUpperCase()}`, focused: true, children: setupDecisionConfirmLines(confirming, decisions.find((decision) => decision.taskId === confirming.taskId), actor, journalFile, run).map((line) => (_jsx(BodyText, { children: fitCells(line, inner) }, line))) })), session.notice !== null && _jsx(BodyText, { children: session.notice }), _jsx(Panel, { title: "PENDING WRITES", children: session.writes.length === 0
|
|
821
856
|
? _jsx(BodyText, { children: fitCells("nothing written this session", inner) })
|
|
822
|
-
: session.writes.map((write) => (_jsx(BodyText, { children: fitCells(`${write.verb} ${write.taskId} ·
|
|
857
|
+
: session.writes.map((write) => (_jsx(BodyText, { children: fitCells(`${write.verb} ${write.taskId} · disposition ${write.disposition}`
|
|
858
|
+
+ ` · release ${write.release ?? "none"} · by ${write.by}`, inner) }, `${write.verb}:${write.taskId}`))) }), _jsx(BodyText, { emphasis: "dim", children: fitCells(setupDecisionsKeybar(session, actionable), Math.max(1, Math.floor(columns))) })] }));
|
|
823
859
|
}
|
|
824
860
|
function SetupDecisionsApp({ cwd, runId, actor, }) {
|
|
825
861
|
const { exit } = useApp();
|
|
@@ -846,7 +882,10 @@ function SetupDecisionsApp({ cwd, runId, actor, }) {
|
|
|
846
882
|
});
|
|
847
883
|
}
|
|
848
884
|
});
|
|
849
|
-
return (_jsx(SetupDecisionsSurface, { decisions: decisions, session: session, columns: stdout.columns ?? 80, actor: actor, journalFile: join(stateDirName(cwd), "runs", runId, "journal.jsonl")
|
|
885
|
+
return (_jsx(SetupDecisionsSurface, { decisions: decisions, session: session, columns: stdout.columns ?? 80, actor: actor, journalFile: join(stateDirName(cwd), "runs", runId, "journal.jsonl"),
|
|
886
|
+
// read per render, not once at mount: a daemon can start or end while the
|
|
887
|
+
// operator is deciding, and the inset must not promise the wrong enactor.
|
|
888
|
+
run: approvalRunOwner(cwd, runId) }));
|
|
850
889
|
}
|
|
851
890
|
/**
|
|
852
891
|
* The setup decisions surface on a live engagement. This is the only cockpit
|
package/package.json
CHANGED
|
@@ -83,7 +83,7 @@ After sending, **confirm delivery** by reading the target pane and verifying the
|
|
|
83
83
|
1. **Prepare** — confirm the target list. Run the [binary preflight](#binary-preflight-before-compile-or-run). Check `git status`, confirm no tickmarkr run is active, and work from a non-main branch.
|
|
84
84
|
2. **Compile** — run `tickmarkr compile <spec-or-directory>`. Fix source-spec defects instead of editing the generated graph.
|
|
85
85
|
3. **Plan** — run `tickmarkr plan`. Review routes, capability-floor warnings, and human gates before execution.
|
|
86
|
-
4. **Run** — run `tickmarkr run`. Watch the run journal for its terminal
|
|
86
|
+
4. **Run** — run `tickmarkr run`. Watch the run journal for its terminal events rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every green clause for you. Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the relevant agent session.
|
|
87
87
|
5. **Verify and consolidate** — continue only after a green run. A run is green when the run-end event exists in the journal, the tip verify is not "failed", and the summary's `failed`, `human`, `blocked` and `pending` buckets are all empty — a run with a parked task is partial, not green. Tickmarkr consolidates accepted work on `tickmarkr/<runId>` and never signs off to the main branch. A human controls any later release merge.
|
|
88
88
|
6. **Record** — write `tickmarkr report <runId> --md` beside the source spec and commit the execution record when the repository tracks those records.
|
|
89
89
|
7. **Continue** — move to the next requested target. If a target fails or is parked, stop with the journal evidence rather than silently skipping it.
|
|
@@ -87,6 +87,6 @@ When spawning consultants (agents gathering synthesis input for decisions like S
|
|
|
87
87
|
1. **Prepare** — start from the requested spec. Run the [binary preflight](#binary-preflight-before-compile-or-run). Check `git status`, confirm no tickmarkr run is active, and work from a non-main branch.
|
|
88
88
|
2. **Compile** — run `tickmarkr compile <spec>`. Correct compilation errors in the spec, never in the generated graph.
|
|
89
89
|
3. **Plan** — run `tickmarkr plan`. Review the routing table, capability-floor warnings, and every human gate, including work that each gate blocks.
|
|
90
|
-
4. **Run** — run `tickmarkr run`. Watch the run journal for its terminal
|
|
90
|
+
4. **Run** — run `tickmarkr run`. Watch the run journal for its terminal events rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every green clause for you. Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the agent session; do not turn them into proxy questions.
|
|
91
91
|
5. **Verify and consolidate** — accept only a green run. A run is green when the run-end event exists in the journal, the tip verify is not "failed", and the summary's `failed`, `human`, `blocked` and `pending` buckets are all empty — a run with a parked task is partial, not green. Tickmarkr consolidates accepted task work on `tickmarkr/<runId>`; it never signs off to the main branch. A human may later merge that integration branch through the repository's normal release process.
|
|
92
92
|
6. **Record** — write `tickmarkr report <runId> --md` beside the source spec and commit the execution record when the repository tracks those records. Then [stand down](#stand-down-mission-end-and-retirement).
|
|
@@ -41,8 +41,11 @@ through brief lineage. **An executor choice nobody made is still an executor cho
|
|
|
41
41
|
is strictly WEAKER than the live seat's report rule 11 already forbids trusting — and it reads as
|
|
42
42
|
settled fact. So at every adopt, walk the predecessor's watchers by class — **journal watchers,
|
|
43
43
|
artifact watchers, dialog watchers and beat loops, which is the closed set a session owns** — probe
|
|
44
|
-
each from the process table yourself
|
|
45
|
-
|
|
44
|
+
each from the process table yourself **twice, a second apart, keeping only what appears in both**
|
|
45
|
+
(rule 11's liveness test — a loop of single `pgrep -f <token>` snapshots is what returned eight
|
|
46
|
+
phantom pids at an adopt on 2026-08-31, every one of them the probing shell itself), and re-arm every
|
|
47
|
+
one the table does not show. A zero here is not absence until the same probe has been run once against
|
|
48
|
+
a watcher you know is alive. Earned 2026-08-25 (OBS-622): a handoff recorded *"artifact watcher armed"*
|
|
46
49
|
over two live consult verdicts; at adopt the only `watch-artifacts.sh` on the machine belonged to a
|
|
47
50
|
different repository, and nothing had been watching either file.
|
|
48
51
|
**An adopted seat ANNOUNCES itself, in the same act as re-arming:** tell the adopted orchestrator the
|
|
@@ -165,6 +168,10 @@ journal tail to decide what happens next, or sweeping orphans — you have taken
|
|
|
165
168
|
- **The journal is the source of truth**, not panes. Watchers go on `run-end` / `task-human` /
|
|
166
169
|
`task-failed` / `consult-verdict`; never sleep-poll inside an agent turn. **Never key a watcher on an
|
|
167
170
|
agent's `done`** — that is turn end and fires the moment a seat finishes acknowledging you.
|
|
171
|
+
**All four are covered by one shipped instrument** — `scripts/watch-journal.sh <runs-dir> [poll] [cap]
|
|
172
|
+
[events-csv]` — which arms on a line baseline, wakes once, and grades a `run-end` against every green
|
|
173
|
+
clause. `scripts/watch-parks.sh` stays the park-specific wake for THIS seat (it counts parks and speaks
|
|
174
|
+
about rulings); the two overlap on `task-human` deliberately, and arming both is coverage, not a bug.
|
|
168
175
|
- **Daemon liveness ≠ journal activity.** A dead daemon emits no events, so journal watchers sleep through
|
|
169
176
|
its death. Liveness comes from the lock's OWN pid (`kill -0`), never a command-name grep. Recovery is
|
|
170
177
|
`tickmarkr resume <runId>` — **the orchestrator's command, not yours** — and note that resume REPLAYS the
|
|
@@ -199,6 +206,58 @@ Read the evidence file the orchestrator writes, rule on it against your pre-comm
|
|
|
199
206
|
record the ruling with what it set aside, and hand the ruling back for execution. That is the whole job,
|
|
200
207
|
and it is the only work that cannot be delegated — which is exactly why nothing else should occupy you.
|
|
201
208
|
|
|
209
|
+
#### The release criterion, and the two clauses without which it grades work it never saw
|
|
210
|
+
|
|
211
|
+
**A pre-commitment is a file. A re-scope is a different file. Nothing joins them** — so a criterion
|
|
212
|
+
survives its own subject's rewrite and keeps grading, which is **worse than having no criterion at all**:
|
|
213
|
+
it carries the authority of a seal over work the seal never saw. Both clauses below, or the join stays
|
|
214
|
+
broken in the direction the missing half covers.
|
|
215
|
+
|
|
216
|
+
**Measured 2026-08-30 (OBS-804).** `RELEASE-CRITERION-v2.1.8.md` was sealed at 10:18:25, nine minutes
|
|
217
|
+
before its run, with zero gate results in existence — an honest pre-commitment that named its trigger AND
|
|
218
|
+
enumerated its subject set, exactly as rule 23 demands. **Two of its four named subjects were then
|
|
219
|
+
materially edited underneath it**: T3 narrowed at 11:49:58, T1 rewritten at 13:37:59 — *three hours and
|
|
220
|
+
nineteen minutes after sealing*. The run that actually delivered the work was, in the criterion's own
|
|
221
|
+
vocabulary, both *"a re-plan"* and *"a follow-on run"* — **two of its own named exclusions. The document
|
|
222
|
+
excluded the only run that ever ran it.** Three of five clauses had already graded MET before anyone asked
|
|
223
|
+
the subject question, and it was caught because a downstream seat refused to decide a ruling that was not
|
|
224
|
+
its own — **not because any instrument detected it.**
|
|
225
|
+
|
|
226
|
+
**Every sealed criterion carries a VOID CONDITIONS section. It is not optional and it is not boilerplate:**
|
|
227
|
+
|
|
228
|
+
```markdown
|
|
229
|
+
## SUBJECT
|
|
230
|
+
<what this criterion is ABOUT — see the identity rule below>
|
|
231
|
+
|
|
232
|
+
## TRIGGER
|
|
233
|
+
<what fires the grading>
|
|
234
|
+
|
|
235
|
+
## CLAUSES
|
|
236
|
+
1. …
|
|
237
|
+
|
|
238
|
+
## VOID CONDITIONS — this document is VOID, with no ruling required, if any of these occur
|
|
239
|
+
- any subject named above is re-scoped, narrowed, widened, split, merged or re-owned
|
|
240
|
+
- the work is delivered by a run this document's own exclusions would exclude
|
|
241
|
+
- <the specific things that would make these clauses grade something else>
|
|
242
|
+
```
|
|
243
|
+
|
|
244
|
+
> **Naming a subject does not freeze it. A pre-commitment must name what VOIDS it, not only what it
|
|
245
|
+
> covers — and a re-scope of any named subject voids it AUTOMATICALLY, with no ruling required.**
|
|
246
|
+
> A void condition that needs a ruling to fire is not a void condition; it is a second thing to forget.
|
|
247
|
+
|
|
248
|
+
**⚡ IDENTIFY THE SUBJECT BY WHAT THE CLAIM IS ABOUT. Half of the failure above was a category error, and
|
|
249
|
+
it is the cheap half to fix:** a graph hash identifies a **PLAN**, and a plan is recompiled, re-cut and
|
|
250
|
+
re-owned as a matter of course. **A criterion about a SHIPPED TREE names the COMMIT** — or the tag, or the
|
|
251
|
+
export tree hash — **never a graph hash, never a run id, never a task list.** Ask what a reader would have
|
|
252
|
+
to hold in their hand to check the clause: if it is bytes, name the bytes.
|
|
253
|
+
|
|
254
|
+
**THE RECIPROCAL DUTY, and it is yours because you write both documents:** when you issue a ruling that
|
|
255
|
+
re-scopes, narrows, splits or re-owns anything, **the ruling must name every sealed document its subject
|
|
256
|
+
appears in** — and say, in the ruling, whether each one is now void. You are the only seat that can do
|
|
257
|
+
this: the criterion cannot watch for the ruling, and the product cannot know that an English document
|
|
258
|
+
elsewhere sealed a claim about a graph it is recompiling. **A re-scope ruling that names no sealed
|
|
259
|
+
documents is asserting there are none. Check before you assert it.**
|
|
260
|
+
|
|
202
261
|
#### The one operational duty that IS yours: a verdict produced under starvation is not a verdict
|
|
203
262
|
|
|
204
263
|
**Operator, 2026-08-07: *"that is the kind of job I need overseer to be vigilant about."*** Do not read the
|
|
@@ -428,6 +487,12 @@ they are left implicit:
|
|
|
428
487
|
were authed on 2026-08-17 while every seat ran claude). Priority when independence is scarce:
|
|
429
488
|
**verifier > checker > planner > executors** — the independent seat goes cross-vendor
|
|
430
489
|
(`herdr agent start … --kind codex`), ruled at dispatch, never debated under time pressure.
|
|
490
|
+
**A codex seat inside a git WORKTREE cannot commit and cannot write outside the worktree** (OBS-824, measured
|
|
491
|
+
twice on 2026-09-01): its sandbox pins writes to the worktree's own path, and a worktree's `.git` is a FILE pointing
|
|
492
|
+
at the main repository's object store, so every `git commit` from inside it is refused. Brief such a seat to leave its
|
|
493
|
+
report INSIDE the worktree and to commit nothing — the overseer commits from the main checkout — or give the work to a
|
|
494
|
+
claude seat, or to a throwaway CLONE (a real `.git` directory). A brief that tells a codex-in-worktree seat to commit
|
|
495
|
+
buys a stall, not a commit.
|
|
431
496
|
4. **Gate every exec lane with the shipped battery, not hand-rolled greps.**
|
|
432
497
|
`tickmarkr verify --base <ref> --criteria <file>` is the standalone form of the engine's own gates —
|
|
433
498
|
build/test/lint diffed against a recorded baseline, evidence, scope, plus the semantic judges — one
|
|
@@ -536,6 +601,53 @@ they are left implicit:
|
|
|
536
601
|
|
|
537
602
|
## Supervision watcher
|
|
538
603
|
|
|
604
|
+
### ⛔ EDITING A WATCHER WHILE WATCHERS ARE ARMED: REPLACE BY RENAME, NEVER IN PLACE
|
|
605
|
+
|
|
606
|
+
**`bash` reads a running script BY BYTE OFFSET.** Edit the file a live watcher is executing and every
|
|
607
|
+
offset after your edit shifts — a comment-only insertion is enough — and the process runs garbage from
|
|
608
|
+
wherever it happens to be. **The failure signature is SILENCE: a corrupted watcher and a correctly-quiet
|
|
609
|
+
one emit byte-identical evidence.**
|
|
610
|
+
|
|
611
|
+
> **Write a temp file, then `mv` it over the target.** `mv` swaps the directory entry and yields a NEW
|
|
612
|
+
> inode; the running `bash` keeps its old inode open and finishes unharmed on the old bytes. **An in-place
|
|
613
|
+
> `sed -i`, or a `>` truncate-and-rewrite, corrupts a running script mid-flight.** Re-arm afterwards to
|
|
614
|
+
> pick up the new bytes — the running process will not.
|
|
615
|
+
|
|
616
|
+
**And `skills/` (canonical) versus `.claude/skills/` (installed) is a MIXED tree — symlinks for some
|
|
617
|
+
files, independent copies for others — so a blanket rule in EITHER direction is wrong** (OBS-809). It
|
|
618
|
+
cuts both ways: for a **shared inode**, an edit in `skills/` reaches into the running process; for a
|
|
619
|
+
**copy**, a fix in `skills/` does **not** reach the running watcher at all, so a repaired watcher keeps
|
|
620
|
+
running the old bytes while the tree says it is fixed. Sync both trees or `skills-single-source.test.ts`
|
|
621
|
+
reds.
|
|
622
|
+
|
|
623
|
+
⚠ **PROBE IT WITH `readlink` AND `stat -L`. NEVER BARE `stat -f %i`: on macOS that reports the SYMLINK'S
|
|
624
|
+
OWN inode, not its target's**, so every symlink reads as a separate file. Measured 2026-08-31 — one seat
|
|
625
|
+
probed **one** file that way, got differing inodes, and wrote *"they are all copies, editing them cannot
|
|
626
|
+
corrupt the running watchers"* into a seat brief as a blanket rule; following it would have silently
|
|
627
|
+
killed the watcher then handing a milestone's spec to its orchestrator. A second seat's first pass
|
|
628
|
+
returned **seven** false *"separate file"* verdicts before it re-probed.
|
|
629
|
+
|
|
630
|
+
```bash
|
|
631
|
+
# Identity, correctly. It must compare the TWO TREES: pointed at either one alone it renders nothing,
|
|
632
|
+
# because the canonical files are all real files and the links live on the installed side.
|
|
633
|
+
cd <repo>
|
|
634
|
+
for f in $(cd skills && find . -type f -o -type l | sed 's|^\./||' | sort); do
|
|
635
|
+
a="skills/$f"; b=".claude/skills/$f"
|
|
636
|
+
[ -e "$b" ] || { printf '%-52s CANON-ONLY\n' "$f"; continue; }
|
|
637
|
+
ia=$(stat -L -f %i "$a"); ib=$(stat -L -f %i "$b") # -L: resolve, or symlinks read as separate files
|
|
638
|
+
[ "$ia" = "$ib" ] \
|
|
639
|
+
&& printf '%-52s SHARED INODE %s -> mv-replace ONLY; a live reader is on these bytes\n' "$f" "$ia" \
|
|
640
|
+
|| printf '%-52s copies (%s/%s) -> a fix in skills/ does NOT reach the running copy\n' "$f" "$ia" "$ib"
|
|
641
|
+
done
|
|
642
|
+
```
|
|
643
|
+
|
|
644
|
+
**Two failures worth separating, and the second is the durable one:** *a probe that cannot render the
|
|
645
|
+
evidence cannot fail* — rule 11 aimed at your own instrument; and **a narrow verified fact was generalised
|
|
646
|
+
to a population it was never sampled over.** One file checked, seven ruled on. **The generalisation ran in
|
|
647
|
+
the REASSURING direction, which is worse than the alarming one: an alarming overclaim gets challenged, a
|
|
648
|
+
reassuring one gets acted on.**
|
|
649
|
+
|
|
650
|
+
|
|
539
651
|
**Arm your OWN tier first, in the same call chain that arms everything else.** `status` derives each
|
|
540
652
|
tier's state from a beat file the tier itself writes, so a seat that never beats reads `ABSENT` — and
|
|
541
653
|
`ABSENT` means *never armed*, which is a lie about a seat that is working the run. Measured on the P99
|
|
@@ -575,7 +687,8 @@ one owned by an unrelated session. So:
|
|
|
575
687
|
beat <tier> --seat <seat>` in this repo). Neither liveness claim is read from a recorded pid: a pid
|
|
576
688
|
recorded earlier can be stale, reused, or detached from the beat now holding the tier green.
|
|
577
689
|
- **At every adopt, clear, or re-brief, sweep for pre-existing loops on YOUR tier before arming one**
|
|
578
|
-
(`pgrep -f "tickmarkr beat <tier>"
|
|
690
|
+
(`pgrep -f "tickmarkr beat <tier>"`, **read twice and intersected** — this exact probe returned its own
|
|
691
|
+
shell as pid 14680 on 2026-08-31), trace each survivor to its parent session, and kill the **loop only**
|
|
579
692
|
— never the parent — then verify the parent survived.
|
|
580
693
|
- **`ARMED (<seat>)` is an attributable claim, not proof that the named seat is still alive.** Before
|
|
581
694
|
trusting it, ask whose session owns the beater; an orphan loop can keep naming a departed seat
|
|
@@ -996,27 +1109,56 @@ twice.** They are mission-independent on purpose: nothing here names a task, a l
|
|
|
996
1109
|
the same root read the other way: a DETACHED loop outlives its seat and holds a tier `ARMED` with
|
|
997
1110
|
nobody home (OBS-583). Neither direction may be assumed; the lifetime is a property of how the watcher
|
|
998
1111
|
was launched, and it belongs in writing next to every claim that one is armed.
|
|
999
|
-
**And
|
|
1000
|
-
|
|
1001
|
-
|
|
1002
|
-
|
|
1003
|
-
|
|
1004
|
-
|
|
1005
|
-
|
|
1006
|
-
|
|
1007
|
-
|
|
1008
|
-
|
|
1009
|
-
|
|
1010
|
-
|
|
1011
|
-
|
|
1012
|
-
|
|
1013
|
-
|
|
1014
|
-
|
|
1015
|
-
|
|
1016
|
-
|
|
1017
|
-
|
|
1018
|
-
|
|
1019
|
-
|
|
1112
|
+
**And EVERY process-table probe has an idiom that defeats it, so the rule above needs the one test
|
|
1113
|
+
that survives all of them. Lead with this; it is not the last resort, it is the first move:**
|
|
1114
|
+
|
|
1115
|
+
> ### A REAL WATCHER DOES NOT EXIT BETWEEN TWO READS.
|
|
1116
|
+
> Read the process table twice, a second or two apart, and keep only what appears in both —
|
|
1117
|
+
> `kill -0 <pid>` on each candidate is the cheap form. Nothing else is needed to kill a phantom.
|
|
1118
|
+
|
|
1119
|
+
It works because it tests a **property of the thing you are hunting** (a watcher persists) rather than
|
|
1120
|
+
a **property of your probe** (how its argv happens to look). A probe-shaped exclusion has to enumerate
|
|
1121
|
+
every way a probe can look; the liveness test does not care what the probe looked like — which is why
|
|
1122
|
+
it is immune to both failures below instead of to one of them.
|
|
1123
|
+
**Measured 2026-08-31 (OBS-807), twice inside one hour, by two seats independently on one machine.**
|
|
1124
|
+
One seat swept for inherited watchers with a loop of `pgrep -f "<script>.sh"` and got **eight
|
|
1125
|
+
live-looking pids**; the other probed `pgrep -f "tickmarkr beat orchestrator"` and got **pid 14680**.
|
|
1126
|
+
**All nine were the probing shell's own argv, and one `kill -0` sweep one command later killed all
|
|
1127
|
+
nine at once** — no cwd resolution, no argv parsing, no per-hit judgement. Two supporting tells, both
|
|
1128
|
+
free: phantom pids arrive **sequential** (6781, 6786, 6791 … one per iteration of the probing loop),
|
|
1129
|
+
and a phantom's `argv` reads **empty** by the time you inspect it.
|
|
1130
|
+
⛔ **Both seats were following this skill's own previous text correctly when they produced a phantom.**
|
|
1131
|
+
That text named `pgrep -f` as the *remedy* for `grep -v grep`. A remedy with that recurrence rate is
|
|
1132
|
+
not a remedy; it is a second trap wearing the first one's clothes.
|
|
1133
|
+
|
|
1134
|
+
**The two idioms, and the direction each one lies in — you still need to know these, because the
|
|
1135
|
+
liveness test tells you a hit is REAL, not that it is YOURS:**
|
|
1136
|
+
- `ps … | grep <token> | grep -v grep` fails toward ***not there***. A poll-grep watcher carries the
|
|
1137
|
+
word `grep` in its own argv, so the filter whose job is removing the *probing* grep removes the
|
|
1138
|
+
*watched* one. Measured 2026-08-06 against a positive control (OBS-415):
|
|
1139
|
+
`ps -eo pid,ppid,etime,command | grep -F <token>` returned **4 matches**, and adding `| grep -v grep`
|
|
1140
|
+
returned **0**. The seat concluded its watcher had died silently, reported that to the operator, and
|
|
1141
|
+
filed it as a defect — and was corrected forty minutes later when the watcher fired normally, having
|
|
1142
|
+
been alive throughout. Two hypotheses (`ps` truncation; multi-column truncation) were formed and
|
|
1143
|
+
killed by measurement first, and the first falsification was itself run against the wrong `ps` form.
|
|
1144
|
+
- `pgrep -f <token>` fails toward ***there twice***, because it matches the **argv of the shell running
|
|
1145
|
+
it**. Measured 2026-08-25: a load-ceiling wake was investigated, a second `npm test` "outside the
|
|
1146
|
+
gate's worktree" was found, and it was the probe; the escalation would have been a false
|
|
1147
|
+
contamination alarm during a task's last attempt. **This is the worse direction, because *there
|
|
1148
|
+
twice* gets ACTED ON** — reported to the operator as a defect, or swept.
|
|
1149
|
+
|
|
1150
|
+
**FALLBACK, for a hit that SURVIVES two reads and still might be yours:** resolve each surviving pid's
|
|
1151
|
+
own `cwd` (`lsof -a -p <pid> -d cwd`) and count only what belongs to the tree you are asking about, or
|
|
1152
|
+
match a pattern the target has and the probe cannot — the binary's real path rather than the words you
|
|
1153
|
+
typed. This costs one `lsof` per candidate plus a judgement call, which is why it is second and not
|
|
1154
|
+
first: after two reads there are usually no candidates left to spend it on.
|
|
1155
|
+
|
|
1156
|
+
⚠ **AND A ZERO IS NOT ABSENCE UNTIL A POSITIVE CONTROL SAYS SO.** Rule 11 demands this of every guard
|
|
1157
|
+
whose failure is silence, and a process probe is exactly that guard: *no matches* and *my filter is
|
|
1158
|
+
broken* are byte-identical outputs. **Before you report a watcher dead, a tree clean, or a machine
|
|
1159
|
+
idle, run the same probe once against something you KNOW is alive** — a watcher you just armed, a
|
|
1160
|
+
`sleep 300` you just launched — and see it come back non-empty. The `grep -v grep` case above was
|
|
1161
|
+
caught by exactly this and by nothing else.
|
|
1020
1162
|
The general rule: **an exclusion filter is exactly as
|
|
1021
1163
|
dangerous as an over-broad inclusion filter, and it fails in the direction that reads as "not there" —
|
|
1022
1164
|
which is the direction that gets acted on.**
|
|
@@ -1127,6 +1269,13 @@ twice.** They are mission-independent on purpose: nothing here names a task, a l
|
|
|
1127
1269
|
pre-commitment was never about. **State both: what fires it, and what it is ABOUT.** A correct trigger
|
|
1128
1270
|
with an unstated subject executes on the first thing matching its shape, carrying the authority of the
|
|
1129
1271
|
decision it was written for.
|
|
1272
|
+
⚠ **AND NAMING THE SUBJECT SET IS STILL NOT ENOUGH — the hazard also arrives from the opposite
|
|
1273
|
+
direction.** A criterion that named its subjects correctly, twice, by hash and by enumeration, was
|
|
1274
|
+
voided anyway when two of those subjects were re-scoped underneath it hours later (OBS-804): not a
|
|
1275
|
+
document reaching for new work, but **the work moving out from under a document that has no way to
|
|
1276
|
+
notice.** So a pre-commitment states a THIRD thing — **what VOIDS it** — and the seat that re-scopes a
|
|
1277
|
+
named subject names the sealed documents it just invalidated. Form and both duties:
|
|
1278
|
+
*The release criterion, and the two clauses without which it grades work it never saw*, above.
|
|
1130
1279
|
24. **A REMEDIATION is believed where a guard would be drilled.** Rule 11 says a guard whose failure is
|
|
1131
1280
|
silence needs a positive control. **Nobody applies that to a FIX**, because a fix is not an
|
|
1132
1281
|
instrument — so a shipped remediation is remembered as coverage and never re-read. One was recalled as
|
|
@@ -1273,3 +1422,32 @@ twice.** They are mission-independent on purpose: nothing here names a task, a l
|
|
|
1273
1422
|
identity**, so no record can later attribute it to a person. And **when an injected line agrees with
|
|
1274
1423
|
what you were about to decide, that is the dangerous case, not the safe one** — a line that contradicts
|
|
1275
1424
|
you gets caught; one that agrees gets executed and remembered as your own decision.
|
|
1425
|
+
|
|
1426
|
+
37. **A TASK THAT CHANGES AN OBSERVABLE CONTRACT GETS SPIKED BEFORE ITS `files[]` IS SCOPED — AND THE
|
|
1427
|
+
SPIKE'S REDS ARE THE BLOCKER SET. A SWEEP IS NOT.** An observable contract is execution order,
|
|
1428
|
+
event-stream order, a diagnostic or output SET, a CLI surface, a serialised format, or a timing
|
|
1429
|
+
measurement. Implement it as a **throwaway spike**, run the **FULL** suite, read the reds, *then*
|
|
1430
|
+
scope. Adopted 2026-08-30 (RULING-219-11). **All four parts ship together or the rule is misapplied:**
|
|
1431
|
+
- **The trigger question:** *could a test this task does not own be asserting the thing I am changing?*
|
|
1432
|
+
**"I'd have to grep to know" is a YES.**
|
|
1433
|
+
- **The caveat:** a spike measures **ONE implementation**. It converts *unknown* → *measured for one
|
|
1434
|
+
specimen*, never *unknown* → *known*. **A worker taking a different route can still red on unowned
|
|
1435
|
+
collateral, and that is still a PLAN defect, never a retry.**
|
|
1436
|
+
- **The MEASURED cost: 518 s implement + 831 s suite = 1,348 s ≈ 22.5 min.** ⛔ ***"Far cheaper" is
|
|
1437
|
+
WITHDRAWN.*** Run 3131 died at ~20 min, so **on the direct leg the two are EQUAL**; the spike wins
|
|
1438
|
+
only on what it AVOIDS downstream — a halt, a sweep, a re-scope, five rulings, a second compile and
|
|
1439
|
+
plan. ⚡ **Therefore it pays only where a late plan defect is EXPENSIVE TO UNWIND.** Where a defect
|
|
1440
|
+
would surface and fix cheaply, **the spike is pure overhead: do not run it.** That the economics and
|
|
1441
|
+
the trigger name the same class, derived independently — one from a stopwatch, one from a taxonomy —
|
|
1442
|
+
is the best evidence this rule is real, and it is why neither half may be quoted without the other.
|
|
1443
|
+
- **The sweep matrix, as the REASON a better sweep is not the remedy** (RULING-219-09): concept **2/3**,
|
|
1444
|
+
symbol **1/3**, union **3/3 on files but only 2/3 on ACTIONABLE SIGNAL** — `gate-telemetry`'s single
|
|
1445
|
+
symbol hit is an order-INSENSITIVE sorted comparison at `:54` that correct triage *discards*, while
|
|
1446
|
+
the failing test at `:101` references it not at all. ⚠ **Rigorous triage makes that discard MORE
|
|
1447
|
+
likely, not less.** The sweep does not fail from sloppiness, so it cannot be fixed with care.
|
|
1448
|
+
|
|
1449
|
+
**Corroboration from the same halt:** of 13 stale-order carriers, the **9** answered from real
|
|
1450
|
+
execution were all proven; the only **2** labelled *"inspection, not execution"* were exactly the 2 the
|
|
1451
|
+
suite never reached. **Evidence quality tracked execution coverage with no exceptions**, while every
|
|
1452
|
+
sweep-based estimate — *"~15"*, *"13"*, *"union 3/3"* — was wrong, and one 512 s suite run gave the
|
|
1453
|
+
right answer (**3**) first time.
|