tickmarkr 2.1.8 → 2.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -4,8 +4,8 @@ import { userInfo } from "node:os";
4
4
  import { join } from "node:path";
5
5
  import { cloneElement, createElement, useState, } from "react";
6
6
  import { parse } from "yaml";
7
- import { approve } from "../../cli/commands/approve.js";
8
- import { ATTEMPT_CAP_RELEASE, GATE_SATISFIED_RELEASE, Journal, REVIEW_UPHELD_RELEASE, } from "../../run/journal.js";
7
+ import { approvalDispositionForRelease, approvalEnactment, approvalRunOwner, approve, } from "../../cli/commands/approve.js";
8
+ import { ATTEMPT_CAP_RELEASE, GATE_SATISFIED_RELEASE, Journal, RECHECK_RELEASE, REVIEW_UPHELD_RELEASE, } from "../../run/journal.js";
9
9
  import { BANNER, GLYPHS, PLAIN_BANNER, } from "../../brand.js";
10
10
  import { DEFAULT_CONFIG, INTEGRITY_FLOOR_SHAPES, TIER_RANK, TickmarkrConfigSchema, unifiedYamlDiff, } from "../../config/config.js";
11
11
  import { GATE_NAMES, SHAPES } from "../../graph/schema.js";
@@ -560,6 +560,20 @@ export function actionableDecisions(decisions) {
560
560
  * replay is the production one (`Journal.replayStatuses`) so this surface and
561
561
  * `tickmarkr status` can never disagree about what is parked.
562
562
  */
563
+ function failedGateForNewestPark(events, taskId, parkedIndex) {
564
+ for (let index = parkedIndex - 1; index >= 0; index -= 1) {
565
+ const event = events[index];
566
+ if (event.taskId !== taskId)
567
+ continue;
568
+ if (event.event === "task-approved" || event.event === "task-human" || event.event === "task-dispatch")
569
+ return undefined;
570
+ if (event.event === "gate-result" && event.data.pass === false
571
+ && typeof event.data.gate === "string" && GATE_NAMES.includes(event.data.gate)) {
572
+ return event.data.gate;
573
+ }
574
+ }
575
+ return undefined;
576
+ }
563
577
  export function deriveParkedDecisions(journal) {
564
578
  const statuses = journal.replayStatuses();
565
579
  const events = journal.read();
@@ -579,12 +593,7 @@ export function deriveParkedDecisions(journal) {
579
593
  parkedIndex = index;
580
594
  }
581
595
  const parked = parkedIndex >= 0 ? events[parkedIndex] : undefined;
582
- // The same rule the production approve command applies: the newest failed
583
- // gate-result before the newest task-human controls what approve records
584
- // and whether uphold exists at all. Never inferred from the park's prose.
585
- const failedGate = events.slice(0, Math.max(0, parkedIndex)).reverse().find((event) => event.event === "gate-result" && event.taskId === taskId && event.data.pass === false
586
- && typeof event.data.gate === "string"
587
- && GATE_NAMES.includes(event.data.gate))?.data.gate;
596
+ const failedGate = failedGateForNewestPark(events, taskId, parkedIndex);
588
597
  const kind = typeof parked?.data.kind === "string" ? parked.data.kind : "human-gate";
589
598
  const reason = typeof parked?.data.reason === "string" ? parked.data.reason : undefined;
590
599
  decisions.push({
@@ -610,8 +619,17 @@ export function initialSetupDecisionsSession() {
610
619
  * inert and the keybar does not advertise it — the surface never promises a
611
620
  * decision the command must refuse.
612
621
  */
622
+ export function setupDecisionVerbs(decision) {
623
+ if (decision.tombstone)
624
+ return [];
625
+ if (decision.kind !== "gate-fail")
626
+ return ["approve"];
627
+ if (decision.failedGate === undefined)
628
+ return [];
629
+ return decision.failedGate === "review" ? ["waive", "uphold", "recheck"] : ["waive", "recheck"];
630
+ }
613
631
  export function upholdAvailable(decision) {
614
- return decision.failedGate === "review";
632
+ return setupDecisionVerbs(decision).includes("uphold");
615
633
  }
616
634
  /**
617
635
  * The decisions surface's whole key grammar: ↑↓ move, a approve, u uphold,
@@ -650,25 +668,30 @@ export function applySetupDecisionsKey(session, event, parked) {
650
668
  },
651
669
  };
652
670
  }
653
- if (event.input === "a" || event.input === "u") {
671
+ const verb = { a: "approve", w: "waive", u: "uphold", r: "recheck" }[event.input ?? ""];
672
+ if (verb !== undefined) {
654
673
  const decision = decisions[selection];
655
- if (decision === undefined)
656
- return { session };
657
- if (event.input === "u" && !upholdAvailable(decision))
674
+ if (decision === undefined || !setupDecisionVerbs(decision).includes(verb))
658
675
  return { session };
659
676
  return {
660
677
  session: {
661
678
  ...session,
662
679
  selection,
663
- confirming: {
664
- verb: event.input === "a" ? "approve" : "uphold",
665
- taskId: decision.taskId,
666
- },
680
+ confirming: { verb, taskId: decision.taskId },
667
681
  },
668
682
  };
669
683
  }
670
684
  return { session };
671
685
  }
686
+ const decisionFlag = (verb) => {
687
+ if (verb === "waive")
688
+ return ["--waive"];
689
+ if (verb === "uphold")
690
+ return ["--uphold"];
691
+ if (verb === "recheck")
692
+ return ["--recheck"];
693
+ return [];
694
+ };
672
695
  /**
673
696
  * THE write. It calls the production `approve` command — validation, fail-closed
674
697
  * refusals and the journal append are inherited, never re-implemented — and then
@@ -680,7 +703,7 @@ export async function executeSetupDecision(command, { cwd, runId, by }) {
680
703
  const message = await approve([
681
704
  runId,
682
705
  command.taskId,
683
- ...(command.verb === "uphold" ? ["--uphold"] : []),
706
+ ...decisionFlag(command.verb),
684
707
  "--by",
685
708
  by,
686
709
  ], cwd);
@@ -693,11 +716,12 @@ export async function executeSetupDecision(command, { cwd, runId, by }) {
693
716
  break;
694
717
  }
695
718
  }
719
+ const disposition = approvalDispositionForRelease(release);
696
720
  return {
697
721
  ok: true,
698
722
  command,
699
723
  message,
700
- write: { taskId: command.taskId, verb: command.verb, release, by },
724
+ write: { taskId: command.taskId, verb: command.verb, release, disposition, by },
701
725
  };
702
726
  }
703
727
  catch (error) {
@@ -726,52 +750,58 @@ function decisionRowText(decision) {
726
750
  * append — the release markers are the command's own constants, so this text
727
751
  * cannot drift from the write. An effect the operator cannot read here is an
728
752
  * effect this surface does not have.
753
+ *
754
+ * WHO enacts it is the command's own sentence (`approvalEnactment`), read from
755
+ * the same run-owner fact: predicting "the next resume" over a LIVE run was
756
+ * true only while the daemon read approvals once at startup. It now sweeps them
757
+ * at every task boundary, so that prediction would send the operator to start a
758
+ * second run in this repository — forbidden — for work already scheduled.
729
759
  */
730
- function decisionEffectLines(command, decision) {
731
- if (command.verb === "uphold") {
760
+ function decisionEffectLines(command, decision, run) {
761
+ if (command.verb === "waive") {
762
+ const gate = decision?.failedGate;
763
+ return gate === undefined ? ["effect none — waive refuses without a failed gate"] : [
764
+ `effect disposition waive-gate; appends release ${GATE_SATISFIED_RELEASE} for gate ${gate}`,
765
+ `it marks gate ${gate} satisfied; it does NOT mark the task done`,
766
+ approvalEnactment("waive-gate", run),
767
+ ];
768
+ }
769
+ if (command.verb === "recheck") {
732
770
  return [
733
- `effect appends one task-approved event with release ${REVIEW_UPHELD_RELEASE};`
734
- + " the next resume funds one fixed attempt carrying the findings",
735
- "it does NOT mark the task done or pass any gate",
771
+ `effect disposition re-dispatch; appends release ${RECHECK_RELEASE}`,
772
+ "it marks no gate satisfied",
773
+ approvalEnactment("re-dispatch", run),
736
774
  ];
737
775
  }
738
- if (decision?.kind === "gate-fail") {
739
- if (decision.failedGate === undefined) {
740
- // The command's own refusal, predicted: no write is promised here.
741
- return [
742
- "effect none — the command refuses: parked on gate-fail"
743
- + " but no failed gate result is recorded",
744
- ];
745
- }
746
- const gate = decision.failedGate;
776
+ if (command.verb === "uphold") {
747
777
  return [
748
- `effect appends one task-approved event with release ${GATE_SATISFIED_RELEASE}`
749
- + ` for gate ${gate}; the next resume continues past the approved gate`,
750
- `it marks gate ${gate} satisfied; it does NOT mark the task done`,
778
+ `effect disposition fund-fixed-attempt; appends release ${REVIEW_UPHELD_RELEASE}`,
779
+ approvalEnactment("fund-fixed-attempt", run),
751
780
  ];
752
781
  }
782
+ if (decision?.kind === "gate-fail") {
783
+ return ["effect none — approve is not offered on gate-fail parks; choose waive, recheck or uphold"];
784
+ }
753
785
  if (decision?.kind === ATTEMPT_CAP_RELEASE) {
754
786
  return [
755
- `effect appends one task-approved event with release ${ATTEMPT_CAP_RELEASE};`
756
- + ` the next resume dispatches ${command.taskId} with a fresh attempt budget`,
757
- "it does NOT mark the task done or pass any gate",
787
+ `effect disposition fresh-budget; appends release ${ATTEMPT_CAP_RELEASE}`,
788
+ approvalEnactment("fresh-budget", run),
758
789
  ];
759
790
  }
760
791
  return [
761
- "effect appends one task-approved event with no release;"
762
- + ` the next resume dispatches ${command.taskId}`,
763
- "it does NOT mark the task done or pass any gate",
792
+ `effect disposition dispatch; appends one task-approved event with no release`,
793
+ approvalEnactment("dispatch", run),
764
794
  ];
765
795
  }
766
796
  /**
767
797
  * The confirm inset's lines: the task, the actor, the effect and the file,
768
798
  * stated before the single key that can write.
769
799
  */
770
- export function setupDecisionConfirmLines(command, decision, actor, journalFile) {
800
+ export function setupDecisionConfirmLines(command, decision, actor, journalFile, run) {
771
801
  return [
772
802
  `task ${command.taskId}${decision === undefined ? "" : ` · ${decision.kind} · attempt ${decision.attempts}`}`,
773
803
  `actor ${actor}`,
774
- ...decisionEffectLines(command, decision),
804
+ ...decisionEffectLines(command, decision, run),
775
805
  `file ${journalFile} · append only`,
776
806
  `y ${command.verb} · n cancel`,
777
807
  ];
@@ -783,8 +813,13 @@ function setupDecisionsKeybar(session, decisions) {
783
813
  if (decisions.length === 0)
784
814
  return "q Quit";
785
815
  const selected = Math.min(session.selection, decisions.length - 1);
786
- const uphold = upholdAvailable(decisions[selected]) ? " · u Uphold" : "";
787
- return `↑↓ Move · a Approve${uphold} · q Quit`;
816
+ const verbs = setupDecisionVerbs(decisions[selected]).map((verb) => ({
817
+ approve: "a Approve",
818
+ waive: "w Waive",
819
+ uphold: "u Uphold",
820
+ recheck: "r Recheck",
821
+ })[verb]).join(" · ");
822
+ return `↑↓ Move · ${verbs} · q Quit`;
788
823
  }
789
824
  /**
790
825
  * What the tab is, drawn on the tab it names. The operator's v1.83 UAT read
@@ -793,7 +828,7 @@ function setupDecisionsKeybar(session, decisions) {
793
828
  * keybar hint also draw.
794
829
  */
795
830
  const DECISIONS_TAB_LINE = `tab ${DECISIONS_TAB_HINT} · journal-parked human decisions`
796
- + " · approve and uphold are its only writes";
831
+ + " · approve waive uphold and recheck are its only writes";
797
832
  /**
798
833
  * What the section says when no park needs the operator. It names its own
799
834
  * trigger and both verbs, so an operator who arrives at a quiet surface learns
@@ -801,8 +836,8 @@ const DECISIONS_TAB_LINE = `tab ${DECISIONS_TAB_HINT} · journal-parked human de
801
836
  */
802
837
  const DECISIONS_EMPTY_STATE = [
803
838
  "nothing needs you now — no task is parked on a human gate",
804
- "a task that parks on one appears here with its two decisions —"
805
- + " approve releases it, uphold sides with the reviewer",
839
+ "a task that parks on one appears here with its decisions —"
840
+ + " approve dispatches, waive accepts a gate, uphold funds a fix, recheck re-runs",
806
841
  ];
807
842
  /**
808
843
  * The decisions surface. Every row it draws — decisions, the confirm inset,
@@ -811,15 +846,16 @@ const DECISIONS_EMPTY_STATE = [
811
846
  * neither overflow its row nor be cut mid-cluster. The one unclipped element
812
847
  * is `notice`: the production command's own words, verbatim.
813
848
  */
814
- export function SetupDecisionsSurface({ decisions, session, columns, actor, journalFile, }) {
849
+ export function SetupDecisionsSurface({ decisions, session, columns, actor, journalFile, run, }) {
815
850
  const inner = Math.max(1, Math.floor(columns) - 4);
816
851
  const actionable = actionableDecisions(decisions);
817
852
  const tombstones = decisions.filter((decision) => decision.tombstone);
818
853
  const selected = Math.min(session.selection, Math.max(0, actionable.length - 1));
819
854
  const confirming = session.confirming;
820
- return (_jsxs(Box, { flexDirection: "column", width: columns, children: [_jsxs(Panel, { title: `${DECISIONS_TAB_TITLE} · ${decisions.length} parked`, focused: true, children: [_jsx(BodyText, { emphasis: "dim", children: fitCells(DECISIONS_TAB_LINE, inner) }), actionable.map((decision, index) => (_jsx(BodyText, { emphasis: index === selected ? "strong" : "normal", children: fitCells(`${index === selected ? `${GLYPHS.pointer} ` : " "}${decisionRowText(decision)}`, inner) }, decision.taskId))), actionable.length === 0 && DECISIONS_EMPTY_STATE.map((line) => (_jsx(BodyText, { children: fitCells(line, inner) }, line))), tombstones.map((decision) => (_jsx(BodyText, { emphasis: "dim", children: fitCells(` read-only · permanent by design · ${decisionRowText(decision)}`, inner) }, decision.taskId)))] }), confirming !== null && (_jsx(Panel, { title: `CONFIRM ${confirming.verb.toUpperCase()}`, focused: true, children: setupDecisionConfirmLines(confirming, decisions.find((decision) => decision.taskId === confirming.taskId), actor, journalFile).map((line) => (_jsx(BodyText, { children: fitCells(line, inner) }, line))) })), session.notice !== null && _jsx(BodyText, { children: session.notice }), _jsx(Panel, { title: "PENDING WRITES", children: session.writes.length === 0
855
+ return (_jsxs(Box, { flexDirection: "column", width: columns, children: [_jsxs(Panel, { title: `${DECISIONS_TAB_TITLE} · ${decisions.length} parked`, focused: true, children: [_jsx(BodyText, { emphasis: "dim", children: fitCells(DECISIONS_TAB_LINE, inner) }), actionable.map((decision, index) => (_jsx(BodyText, { emphasis: index === selected ? "strong" : "normal", children: fitCells(`${index === selected ? `${GLYPHS.pointer} ` : " "}${decisionRowText(decision)}`, inner) }, decision.taskId))), actionable.length === 0 && DECISIONS_EMPTY_STATE.map((line) => (_jsx(BodyText, { children: fitCells(line, inner) }, line))), tombstones.map((decision) => (_jsx(BodyText, { emphasis: "dim", children: fitCells(` read-only · permanent by design · ${decisionRowText(decision)}`, inner) }, decision.taskId)))] }), confirming !== null && (_jsx(Panel, { title: `CONFIRM ${confirming.verb.toUpperCase()}`, focused: true, children: setupDecisionConfirmLines(confirming, decisions.find((decision) => decision.taskId === confirming.taskId), actor, journalFile, run).map((line) => (_jsx(BodyText, { children: fitCells(line, inner) }, line))) })), session.notice !== null && _jsx(BodyText, { children: session.notice }), _jsx(Panel, { title: "PENDING WRITES", children: session.writes.length === 0
821
856
  ? _jsx(BodyText, { children: fitCells("nothing written this session", inner) })
822
- : session.writes.map((write) => (_jsx(BodyText, { children: fitCells(`${write.verb} ${write.taskId} · release ${write.release ?? "none"} · by ${write.by}`, inner) }, `${write.verb}:${write.taskId}`))) }), _jsx(BodyText, { emphasis: "dim", children: fitCells(setupDecisionsKeybar(session, actionable), Math.max(1, Math.floor(columns))) })] }));
857
+ : session.writes.map((write) => (_jsx(BodyText, { children: fitCells(`${write.verb} ${write.taskId} · disposition ${write.disposition}`
858
+ + ` · release ${write.release ?? "none"} · by ${write.by}`, inner) }, `${write.verb}:${write.taskId}`))) }), _jsx(BodyText, { emphasis: "dim", children: fitCells(setupDecisionsKeybar(session, actionable), Math.max(1, Math.floor(columns))) })] }));
823
859
  }
824
860
  function SetupDecisionsApp({ cwd, runId, actor, }) {
825
861
  const { exit } = useApp();
@@ -846,7 +882,10 @@ function SetupDecisionsApp({ cwd, runId, actor, }) {
846
882
  });
847
883
  }
848
884
  });
849
- return (_jsx(SetupDecisionsSurface, { decisions: decisions, session: session, columns: stdout.columns ?? 80, actor: actor, journalFile: join(stateDirName(cwd), "runs", runId, "journal.jsonl") }));
885
+ return (_jsx(SetupDecisionsSurface, { decisions: decisions, session: session, columns: stdout.columns ?? 80, actor: actor, journalFile: join(stateDirName(cwd), "runs", runId, "journal.jsonl"),
886
+ // read per render, not once at mount: a daemon can start or end while the
887
+ // operator is deciding, and the inset must not promise the wrong enactor.
888
+ run: approvalRunOwner(cwd, runId) }));
850
889
  }
851
890
  /**
852
891
  * The setup decisions surface on a live engagement. This is the only cockpit
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "tickmarkr",
3
- "version": "2.1.8",
3
+ "version": "2.2.0",
4
4
  "description": "Spec in, verified work out.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -83,7 +83,7 @@ After sending, **confirm delivery** by reading the target pane and verifying the
83
83
  1. **Prepare** — confirm the target list. Run the [binary preflight](#binary-preflight-before-compile-or-run). Check `git status`, confirm no tickmarkr run is active, and work from a non-main branch.
84
84
  2. **Compile** — run `tickmarkr compile <spec-or-directory>`. Fix source-spec defects instead of editing the generated graph.
85
85
  3. **Plan** — run `tickmarkr plan`. Review routes, capability-floor warnings, and human gates before execution.
86
- 4. **Run** — run `tickmarkr run`. Watch the run journal for its terminal event rather than polling agents — a self-terminating poll (`until grep -q '"event":"run-end"' <state-dir>/runs/<runId>/journal.jsonl; do sleep 20; done`), never `tail -F | grep -m1` (wedges on the journal's final line) and never a pane-level done wait (turn-end flaps). Resolve blocked interactions in the relevant agent session.
86
+ 4. **Run** — run `tickmarkr run`. Watch the run journal for its terminal events rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every green clause for you. Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the relevant agent session.
87
87
  5. **Verify and consolidate** — continue only after a green run. A run is green when the run-end event exists in the journal, the tip verify is not "failed", and the summary's `failed`, `human`, `blocked` and `pending` buckets are all empty — a run with a parked task is partial, not green. Tickmarkr consolidates accepted work on `tickmarkr/<runId>` and never signs off to the main branch. A human controls any later release merge.
88
88
  6. **Record** — write `tickmarkr report <runId> --md` beside the source spec and commit the execution record when the repository tracks those records.
89
89
  7. **Continue** — move to the next requested target. If a target fails or is parked, stop with the journal evidence rather than silently skipping it.
@@ -87,6 +87,6 @@ When spawning consultants (agents gathering synthesis input for decisions like S
87
87
  1. **Prepare** — start from the requested spec. Run the [binary preflight](#binary-preflight-before-compile-or-run). Check `git status`, confirm no tickmarkr run is active, and work from a non-main branch.
88
88
  2. **Compile** — run `tickmarkr compile <spec>`. Correct compilation errors in the spec, never in the generated graph.
89
89
  3. **Plan** — run `tickmarkr plan`. Review the routing table, capability-floor warnings, and every human gate, including work that each gate blocks.
90
- 4. **Run** — run `tickmarkr run`. Watch the run journal for its terminal event rather than repeatedly polling agents. Use a self-terminating poll — `until grep -q '"event":"run-end"' <state-dir>/runs/<runId>/journal.jsonl; do sleep 20; done` — never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). Resolve blocked interactions in the agent session; do not turn them into proxy questions.
90
+ 4. **Run** — run `tickmarkr run`. Watch the run journal for its terminal events rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every green clause for you. Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the agent session; do not turn them into proxy questions.
91
91
  5. **Verify and consolidate** — accept only a green run. A run is green when the run-end event exists in the journal, the tip verify is not "failed", and the summary's `failed`, `human`, `blocked` and `pending` buckets are all empty — a run with a parked task is partial, not green. Tickmarkr consolidates accepted task work on `tickmarkr/<runId>`; it never signs off to the main branch. A human may later merge that integration branch through the repository's normal release process.
92
92
  6. **Record** — write `tickmarkr report <runId> --md` beside the source spec and commit the execution record when the repository tracks those records. Then [stand down](#stand-down-mission-end-and-retirement).
@@ -41,8 +41,11 @@ through brief lineage. **An executor choice nobody made is still an executor cho
41
41
  is strictly WEAKER than the live seat's report rule 11 already forbids trusting — and it reads as
42
42
  settled fact. So at every adopt, walk the predecessor's watchers by class — **journal watchers,
43
43
  artifact watchers, dialog watchers and beat loops, which is the closed set a session owns** — probe
44
- each from the process table yourself (`pgrep -f <token>`, discriminated per rule 11), and re-arm every
45
- one the table does not show. Earned 2026-08-25 (OBS-622): a handoff recorded *"artifact watcher armed"*
44
+ each from the process table yourself **twice, a second apart, keeping only what appears in both**
45
+ (rule 11's liveness test — a loop of single `pgrep -f <token>` snapshots is what returned eight
46
+ phantom pids at an adopt on 2026-08-31, every one of them the probing shell itself), and re-arm every
47
+ one the table does not show. A zero here is not absence until the same probe has been run once against
48
+ a watcher you know is alive. Earned 2026-08-25 (OBS-622): a handoff recorded *"artifact watcher armed"*
46
49
  over two live consult verdicts; at adopt the only `watch-artifacts.sh` on the machine belonged to a
47
50
  different repository, and nothing had been watching either file.
48
51
  **An adopted seat ANNOUNCES itself, in the same act as re-arming:** tell the adopted orchestrator the
@@ -165,6 +168,10 @@ journal tail to decide what happens next, or sweeping orphans — you have taken
165
168
  - **The journal is the source of truth**, not panes. Watchers go on `run-end` / `task-human` /
166
169
  `task-failed` / `consult-verdict`; never sleep-poll inside an agent turn. **Never key a watcher on an
167
170
  agent's `done`** — that is turn end and fires the moment a seat finishes acknowledging you.
171
+ **All four are covered by one shipped instrument** — `scripts/watch-journal.sh <runs-dir> [poll] [cap]
172
+ [events-csv]` — which arms on a line baseline, wakes once, and grades a `run-end` against every green
173
+ clause. `scripts/watch-parks.sh` stays the park-specific wake for THIS seat (it counts parks and speaks
174
+ about rulings); the two overlap on `task-human` deliberately, and arming both is coverage, not a bug.
168
175
  - **Daemon liveness ≠ journal activity.** A dead daemon emits no events, so journal watchers sleep through
169
176
  its death. Liveness comes from the lock's OWN pid (`kill -0`), never a command-name grep. Recovery is
170
177
  `tickmarkr resume <runId>` — **the orchestrator's command, not yours** — and note that resume REPLAYS the
@@ -199,6 +206,58 @@ Read the evidence file the orchestrator writes, rule on it against your pre-comm
199
206
  record the ruling with what it set aside, and hand the ruling back for execution. That is the whole job,
200
207
  and it is the only work that cannot be delegated — which is exactly why nothing else should occupy you.
201
208
 
209
+ #### The release criterion, and the two clauses without which it grades work it never saw
210
+
211
+ **A pre-commitment is a file. A re-scope is a different file. Nothing joins them** — so a criterion
212
+ survives its own subject's rewrite and keeps grading, which is **worse than having no criterion at all**:
213
+ it carries the authority of a seal over work the seal never saw. Both clauses below, or the join stays
214
+ broken in the direction the missing half covers.
215
+
216
+ **Measured 2026-08-30 (OBS-804).** `RELEASE-CRITERION-v2.1.8.md` was sealed at 10:18:25, nine minutes
217
+ before its run, with zero gate results in existence — an honest pre-commitment that named its trigger AND
218
+ enumerated its subject set, exactly as rule 23 demands. **Two of its four named subjects were then
219
+ materially edited underneath it**: T3 narrowed at 11:49:58, T1 rewritten at 13:37:59 — *three hours and
220
+ nineteen minutes after sealing*. The run that actually delivered the work was, in the criterion's own
221
+ vocabulary, both *"a re-plan"* and *"a follow-on run"* — **two of its own named exclusions. The document
222
+ excluded the only run that ever ran it.** Three of five clauses had already graded MET before anyone asked
223
+ the subject question, and it was caught because a downstream seat refused to decide a ruling that was not
224
+ its own — **not because any instrument detected it.**
225
+
226
+ **Every sealed criterion carries a VOID CONDITIONS section. It is not optional and it is not boilerplate:**
227
+
228
+ ```markdown
229
+ ## SUBJECT
230
+ <what this criterion is ABOUT — see the identity rule below>
231
+
232
+ ## TRIGGER
233
+ <what fires the grading>
234
+
235
+ ## CLAUSES
236
+ 1. …
237
+
238
+ ## VOID CONDITIONS — this document is VOID, with no ruling required, if any of these occur
239
+ - any subject named above is re-scoped, narrowed, widened, split, merged or re-owned
240
+ - the work is delivered by a run this document's own exclusions would exclude
241
+ - <the specific things that would make these clauses grade something else>
242
+ ```
243
+
244
+ > **Naming a subject does not freeze it. A pre-commitment must name what VOIDS it, not only what it
245
+ > covers — and a re-scope of any named subject voids it AUTOMATICALLY, with no ruling required.**
246
+ > A void condition that needs a ruling to fire is not a void condition; it is a second thing to forget.
247
+
248
+ **⚡ IDENTIFY THE SUBJECT BY WHAT THE CLAIM IS ABOUT. Half of the failure above was a category error, and
249
+ it is the cheap half to fix:** a graph hash identifies a **PLAN**, and a plan is recompiled, re-cut and
250
+ re-owned as a matter of course. **A criterion about a SHIPPED TREE names the COMMIT** — or the tag, or the
251
+ export tree hash — **never a graph hash, never a run id, never a task list.** Ask what a reader would have
252
+ to hold in their hand to check the clause: if it is bytes, name the bytes.
253
+
254
+ **THE RECIPROCAL DUTY, and it is yours because you write both documents:** when you issue a ruling that
255
+ re-scopes, narrows, splits or re-owns anything, **the ruling must name every sealed document its subject
256
+ appears in** — and say, in the ruling, whether each one is now void. You are the only seat that can do
257
+ this: the criterion cannot watch for the ruling, and the product cannot know that an English document
258
+ elsewhere sealed a claim about a graph it is recompiling. **A re-scope ruling that names no sealed
259
+ documents is asserting there are none. Check before you assert it.**
260
+
202
261
  #### The one operational duty that IS yours: a verdict produced under starvation is not a verdict
203
262
 
204
263
  **Operator, 2026-08-07: *"that is the kind of job I need overseer to be vigilant about."*** Do not read the
@@ -428,6 +487,12 @@ they are left implicit:
428
487
  were authed on 2026-08-17 while every seat ran claude). Priority when independence is scarce:
429
488
  **verifier > checker > planner > executors** — the independent seat goes cross-vendor
430
489
  (`herdr agent start … --kind codex`), ruled at dispatch, never debated under time pressure.
490
+ **A codex seat inside a git WORKTREE cannot commit and cannot write outside the worktree** (OBS-824, measured
491
+ twice on 2026-09-01): its sandbox pins writes to the worktree's own path, and a worktree's `.git` is a FILE pointing
492
+ at the main repository's object store, so every `git commit` from inside it is refused. Brief such a seat to leave its
493
+ report INSIDE the worktree and to commit nothing — the overseer commits from the main checkout — or give the work to a
494
+ claude seat, or to a throwaway CLONE (a real `.git` directory). A brief that tells a codex-in-worktree seat to commit
495
+ buys a stall, not a commit.
431
496
  4. **Gate every exec lane with the shipped battery, not hand-rolled greps.**
432
497
  `tickmarkr verify --base <ref> --criteria <file>` is the standalone form of the engine's own gates —
433
498
  build/test/lint diffed against a recorded baseline, evidence, scope, plus the semantic judges — one
@@ -536,6 +601,53 @@ they are left implicit:
536
601
 
537
602
  ## Supervision watcher
538
603
 
604
+ ### ⛔ EDITING A WATCHER WHILE WATCHERS ARE ARMED: REPLACE BY RENAME, NEVER IN PLACE
605
+
606
+ **`bash` reads a running script BY BYTE OFFSET.** Edit the file a live watcher is executing and every
607
+ offset after your edit shifts — a comment-only insertion is enough — and the process runs garbage from
608
+ wherever it happens to be. **The failure signature is SILENCE: a corrupted watcher and a correctly-quiet
609
+ one emit byte-identical evidence.**
610
+
611
+ > **Write a temp file, then `mv` it over the target.** `mv` swaps the directory entry and yields a NEW
612
+ > inode; the running `bash` keeps its old inode open and finishes unharmed on the old bytes. **An in-place
613
+ > `sed -i`, or a `>` truncate-and-rewrite, corrupts a running script mid-flight.** Re-arm afterwards to
614
+ > pick up the new bytes — the running process will not.
615
+
616
+ **And `skills/` (canonical) versus `.claude/skills/` (installed) is a MIXED tree — symlinks for some
617
+ files, independent copies for others — so a blanket rule in EITHER direction is wrong** (OBS-809). It
618
+ cuts both ways: for a **shared inode**, an edit in `skills/` reaches into the running process; for a
619
+ **copy**, a fix in `skills/` does **not** reach the running watcher at all, so a repaired watcher keeps
620
+ running the old bytes while the tree says it is fixed. Sync both trees or `skills-single-source.test.ts`
621
+ reds.
622
+
623
+ ⚠ **PROBE IT WITH `readlink` AND `stat -L`. NEVER BARE `stat -f %i`: on macOS that reports the SYMLINK'S
624
+ OWN inode, not its target's**, so every symlink reads as a separate file. Measured 2026-08-31 — one seat
625
+ probed **one** file that way, got differing inodes, and wrote *"they are all copies, editing them cannot
626
+ corrupt the running watchers"* into a seat brief as a blanket rule; following it would have silently
627
+ killed the watcher then handing a milestone's spec to its orchestrator. A second seat's first pass
628
+ returned **seven** false *"separate file"* verdicts before it re-probed.
629
+
630
+ ```bash
631
+ # Identity, correctly. It must compare the TWO TREES: pointed at either one alone it renders nothing,
632
+ # because the canonical files are all real files and the links live on the installed side.
633
+ cd <repo>
634
+ for f in $(cd skills && find . -type f -o -type l | sed 's|^\./||' | sort); do
635
+ a="skills/$f"; b=".claude/skills/$f"
636
+ [ -e "$b" ] || { printf '%-52s CANON-ONLY\n' "$f"; continue; }
637
+ ia=$(stat -L -f %i "$a"); ib=$(stat -L -f %i "$b") # -L: resolve, or symlinks read as separate files
638
+ [ "$ia" = "$ib" ] \
639
+ && printf '%-52s SHARED INODE %s -> mv-replace ONLY; a live reader is on these bytes\n' "$f" "$ia" \
640
+ || printf '%-52s copies (%s/%s) -> a fix in skills/ does NOT reach the running copy\n' "$f" "$ia" "$ib"
641
+ done
642
+ ```
643
+
644
+ **Two failures worth separating, and the second is the durable one:** *a probe that cannot render the
645
+ evidence cannot fail* — rule 11 aimed at your own instrument; and **a narrow verified fact was generalised
646
+ to a population it was never sampled over.** One file checked, seven ruled on. **The generalisation ran in
647
+ the REASSURING direction, which is worse than the alarming one: an alarming overclaim gets challenged, a
648
+ reassuring one gets acted on.**
649
+
650
+
539
651
  **Arm your OWN tier first, in the same call chain that arms everything else.** `status` derives each
540
652
  tier's state from a beat file the tier itself writes, so a seat that never beats reads `ABSENT` — and
541
653
  `ABSENT` means *never armed*, which is a lie about a seat that is working the run. Measured on the P99
@@ -575,7 +687,8 @@ one owned by an unrelated session. So:
575
687
  beat <tier> --seat <seat>` in this repo). Neither liveness claim is read from a recorded pid: a pid
576
688
  recorded earlier can be stale, reused, or detached from the beat now holding the tier green.
577
689
  - **At every adopt, clear, or re-brief, sweep for pre-existing loops on YOUR tier before arming one**
578
- (`pgrep -f "tickmarkr beat <tier>"`), trace each to its parent session, and kill the **loop only**
690
+ (`pgrep -f "tickmarkr beat <tier>"`, **read twice and intersected** — this exact probe returned its own
691
+ shell as pid 14680 on 2026-08-31), trace each survivor to its parent session, and kill the **loop only**
579
692
  — never the parent — then verify the parent survived.
580
693
  - **`ARMED (<seat>)` is an attributable claim, not proof that the named seat is still alive.** Before
581
694
  trusting it, ask whose session owns the beater; an orphan loop can keep naming a departed seat
@@ -996,27 +1109,56 @@ twice.** They are mission-independent on purpose: nothing here names a task, a l
996
1109
  the same root read the other way: a DETACHED loop outlives its seat and holds a tier `ARMED` with
997
1110
  nobody home (OBS-583). Neither direction may be assumed; the lifetime is a property of how the watcher
998
1111
  was launched, and it belongs in writing next to every claim that one is armed.
999
- **And the process-table probe has a standard idiom that DEFEATS it, so the rule above needs one more
1000
- line to be usable.** Never probe for a watcher with `ps … | grep <token> | grep -v grep`: a poll-grep
1001
- watcher carries the word `grep` in its own argv, so the filter whose job is removing the *probing* grep
1002
- removes the *watched* one. Measured 2026-08-06 against a positive control (OBS-415):
1003
- `ps -eo pid,ppid,etime,command | grep -F <token>` returned **4 matches**, and adding `| grep -v grep`
1004
- returned **0**. The seat concluded its watcher had died silently, reported that to the operator, filed
1005
- it as a defect — and was corrected forty minutes later when the watcher fired normally, having been
1006
- alive throughout. Two hypotheses (`ps` truncation; multi-column truncation) were formed and killed by
1007
- measurement first, and the first falsification was itself run against the wrong `ps` form. **Use
1008
- `pgrep -f <token>`, or read the lock's own pid.**
1009
- ⚠ **AND `pgrep -f` HAS ITS OWN INVERSE FAILURE, so the recommended fix is not free: it matches the
1010
- ARGV OF THE SHELL RUNNING IT.** A probe written as `pgrep -f "npm test"` is itself a process whose
1011
- command line contains `npm test`, so it returns its own shell — a PHANTOM that looks exactly like the
1012
- contamination you are hunting. Measured 2026-08-25: a load-ceiling wake was investigated, a second
1013
- `npm test` "outside the gate's worktree" was found, and it was the probe. It had vanished by the next
1014
- command, which is the tell — a real second suite does not exit between two reads. The escalation would
1015
- have been a false contamination alarm during a task's last attempt.
1016
- Discriminate before you believe a hit: **resolve each pid's `cwd` AND drop any whose own command
1017
- contains the probe** (`pgrep`/`bash -c`), or match a pattern the target has and the probe cannot —
1018
- the binary's real path rather than the words you typed. `grep -v grep` fails toward *not there*;
1019
- `pgrep -f` fails toward *there twice*, and this direction gets ACTED ON, which is worse.
1112
+ **And EVERY process-table probe has an idiom that defeats it, so the rule above needs the one test
1113
+ that survives all of them. Lead with this; it is not the last resort, it is the first move:**
1114
+
1115
+ > ### A REAL WATCHER DOES NOT EXIT BETWEEN TWO READS.
1116
+ > Read the process table twice, a second or two apart, and keep only what appears in both —
1117
+ > `kill -0 <pid>` on each candidate is the cheap form. Nothing else is needed to kill a phantom.
1118
+
1119
+ It works because it tests a **property of the thing you are hunting** (a watcher persists) rather than
1120
+ a **property of your probe** (how its argv happens to look). A probe-shaped exclusion has to enumerate
1121
+ every way a probe can look; the liveness test does not care what the probe looked like — which is why
1122
+ it is immune to both failures below instead of to one of them.
1123
+ **Measured 2026-08-31 (OBS-807), twice inside one hour, by two seats independently on one machine.**
1124
+ One seat swept for inherited watchers with a loop of `pgrep -f "<script>.sh"` and got **eight
1125
+ live-looking pids**; the other probed `pgrep -f "tickmarkr beat orchestrator"` and got **pid 14680**.
1126
+ **All nine were the probing shell's own argv, and one `kill -0` sweep one command later killed all
1127
+ nine at once** — no cwd resolution, no argv parsing, no per-hit judgement. Two supporting tells, both
1128
+ free: phantom pids arrive **sequential** (6781, 6786, 6791 … one per iteration of the probing loop),
1129
+ and a phantom's `argv` reads **empty** by the time you inspect it.
1130
+ ⛔ **Both seats were following this skill's own previous text correctly when they produced a phantom.**
1131
+ That text named `pgrep -f` as the *remedy* for `grep -v grep`. A remedy with that recurrence rate is
1132
+ not a remedy; it is a second trap wearing the first one's clothes.
1133
+
1134
+ **The two idioms, and the direction each one lies in — you still need to know these, because the
1135
+ liveness test tells you a hit is REAL, not that it is YOURS:**
1136
+ - `ps … | grep <token> | grep -v grep` fails toward ***not there***. A poll-grep watcher carries the
1137
+ word `grep` in its own argv, so the filter whose job is removing the *probing* grep removes the
1138
+ *watched* one. Measured 2026-08-06 against a positive control (OBS-415):
1139
+ `ps -eo pid,ppid,etime,command | grep -F <token>` returned **4 matches**, and adding `| grep -v grep`
1140
+ returned **0**. The seat concluded its watcher had died silently, reported that to the operator, and
1141
+ filed it as a defect — and was corrected forty minutes later when the watcher fired normally, having
1142
+ been alive throughout. Two hypotheses (`ps` truncation; multi-column truncation) were formed and
1143
+ killed by measurement first, and the first falsification was itself run against the wrong `ps` form.
1144
+ - `pgrep -f <token>` fails toward ***there twice***, because it matches the **argv of the shell running
1145
+ it**. Measured 2026-08-25: a load-ceiling wake was investigated, a second `npm test` "outside the
1146
+ gate's worktree" was found, and it was the probe; the escalation would have been a false
1147
+ contamination alarm during a task's last attempt. **This is the worse direction, because *there
1148
+ twice* gets ACTED ON** — reported to the operator as a defect, or swept.
1149
+
1150
+ **FALLBACK, for a hit that SURVIVES two reads and still might be yours:** resolve each surviving pid's
1151
+ own `cwd` (`lsof -a -p <pid> -d cwd`) and count only what belongs to the tree you are asking about, or
1152
+ match a pattern the target has and the probe cannot — the binary's real path rather than the words you
1153
+ typed. This costs one `lsof` per candidate plus a judgement call, which is why it is second and not
1154
+ first: after two reads there are usually no candidates left to spend it on.
1155
+
1156
+ ⚠ **AND A ZERO IS NOT ABSENCE UNTIL A POSITIVE CONTROL SAYS SO.** Rule 11 demands this of every guard
1157
+ whose failure is silence, and a process probe is exactly that guard: *no matches* and *my filter is
1158
+ broken* are byte-identical outputs. **Before you report a watcher dead, a tree clean, or a machine
1159
+ idle, run the same probe once against something you KNOW is alive** — a watcher you just armed, a
1160
+ `sleep 300` you just launched — and see it come back non-empty. The `grep -v grep` case above was
1161
+ caught by exactly this and by nothing else.
1020
1162
  The general rule: **an exclusion filter is exactly as
1021
1163
  dangerous as an over-broad inclusion filter, and it fails in the direction that reads as "not there" —
1022
1164
  which is the direction that gets acted on.**
@@ -1127,6 +1269,13 @@ twice.** They are mission-independent on purpose: nothing here names a task, a l
1127
1269
  pre-commitment was never about. **State both: what fires it, and what it is ABOUT.** A correct trigger
1128
1270
  with an unstated subject executes on the first thing matching its shape, carrying the authority of the
1129
1271
  decision it was written for.
1272
+ ⚠ **AND NAMING THE SUBJECT SET IS STILL NOT ENOUGH — the hazard also arrives from the opposite
1273
+ direction.** A criterion that named its subjects correctly, twice, by hash and by enumeration, was
1274
+ voided anyway when two of those subjects were re-scoped underneath it hours later (OBS-804): not a
1275
+ document reaching for new work, but **the work moving out from under a document that has no way to
1276
+ notice.** So a pre-commitment states a THIRD thing — **what VOIDS it** — and the seat that re-scopes a
1277
+ named subject names the sealed documents it just invalidated. Form and both duties:
1278
+ *The release criterion, and the two clauses without which it grades work it never saw*, above.
1130
1279
  24. **A REMEDIATION is believed where a guard would be drilled.** Rule 11 says a guard whose failure is
1131
1280
  silence needs a positive control. **Nobody applies that to a FIX**, because a fix is not an
1132
1281
  instrument — so a shipped remediation is remembered as coverage and never re-read. One was recalled as
@@ -1273,3 +1422,32 @@ twice.** They are mission-independent on purpose: nothing here names a task, a l
1273
1422
  identity**, so no record can later attribute it to a person. And **when an injected line agrees with
1274
1423
  what you were about to decide, that is the dangerous case, not the safe one** — a line that contradicts
1275
1424
  you gets caught; one that agrees gets executed and remembered as your own decision.
1425
+
1426
+ 37. **A TASK THAT CHANGES AN OBSERVABLE CONTRACT GETS SPIKED BEFORE ITS `files[]` IS SCOPED — AND THE
1427
+ SPIKE'S REDS ARE THE BLOCKER SET. A SWEEP IS NOT.** An observable contract is execution order,
1428
+ event-stream order, a diagnostic or output SET, a CLI surface, a serialised format, or a timing
1429
+ measurement. Implement it as a **throwaway spike**, run the **FULL** suite, read the reds, *then*
1430
+ scope. Adopted 2026-08-30 (RULING-219-11). **All four parts ship together or the rule is misapplied:**
1431
+ - **The trigger question:** *could a test this task does not own be asserting the thing I am changing?*
1432
+ **"I'd have to grep to know" is a YES.**
1433
+ - **The caveat:** a spike measures **ONE implementation**. It converts *unknown* → *measured for one
1434
+ specimen*, never *unknown* → *known*. **A worker taking a different route can still red on unowned
1435
+ collateral, and that is still a PLAN defect, never a retry.**
1436
+ - **The MEASURED cost: 518 s implement + 831 s suite = 1,348 s ≈ 22.5 min.** ⛔ ***"Far cheaper" is
1437
+ WITHDRAWN.*** Run 3131 died at ~20 min, so **on the direct leg the two are EQUAL**; the spike wins
1438
+ only on what it AVOIDS downstream — a halt, a sweep, a re-scope, five rulings, a second compile and
1439
+ plan. ⚡ **Therefore it pays only where a late plan defect is EXPENSIVE TO UNWIND.** Where a defect
1440
+ would surface and fix cheaply, **the spike is pure overhead: do not run it.** That the economics and
1441
+ the trigger name the same class, derived independently — one from a stopwatch, one from a taxonomy —
1442
+ is the best evidence this rule is real, and it is why neither half may be quoted without the other.
1443
+ - **The sweep matrix, as the REASON a better sweep is not the remedy** (RULING-219-09): concept **2/3**,
1444
+ symbol **1/3**, union **3/3 on files but only 2/3 on ACTIONABLE SIGNAL** — `gate-telemetry`'s single
1445
+ symbol hit is an order-INSENSITIVE sorted comparison at `:54` that correct triage *discards*, while
1446
+ the failing test at `:101` references it not at all. ⚠ **Rigorous triage makes that discard MORE
1447
+ likely, not less.** The sweep does not fail from sloppiness, so it cannot be fixed with care.
1448
+
1449
+ **Corroboration from the same halt:** of 13 stale-order carriers, the **9** answered from real
1450
+ execution were all proven; the only **2** labelled *"inspection, not execution"* were exactly the 2 the
1451
+ suite never reached. **Evidence quality tracked execution coverage with no exceptions**, while every
1452
+ sweep-based estimate — *"~15"*, *"13"*, *"union 3/3"* — was wrong, and one 512 s suite run gave the
1453
+ right answer (**3**) first time.