tickmarkr 2.1.8 → 2.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -38,6 +38,8 @@
38
38
  # TKR_HANDOFF_MAX_AGE_S how fresh "fresh" is (default 900).
39
39
  # TKR_CLEAR_SETTLE_S seconds to let a cleared seat settle before the re-brief (default 6).
40
40
  # TKR_BLIND_ALARM_S seconds unreadable before CONTEXT_BLIND alarms (default 120).
41
+ # TKR_CONTEXT_WINDOW context window in tokens. Set it when the banner does not name one;
42
+ # NEVER guessed — a wrong denominator makes every warn and act line wrong.
41
43
 
42
44
  set -u
43
45
  ROLE="${1:?supervising seat role required: orchestrator|overseer}"
@@ -100,10 +102,96 @@ stand_down() { tickmarkr beat "$TIER" --stand-down --seat "$SEAT" >/dev/null 2>&
100
102
  # record the hand-off. A killed watcher never runs it, which is the one case that must read STALE.
101
103
  trap stand_down EXIT
102
104
 
103
- # The seat's rendered truth lives on the model banner. Select that line first, then read a percentage
104
- # only from it: a bare numeric search across the visible window can borrow an old N% from scrollback
105
- # when a long live-run segment pushes the real field past the pane's visible width (OBS-780).
106
- context_pct() {
105
+ # ── WHERE THE NUMBER COMES FROM, and this is the whole 2.1.7-series lesson ────────────────────────────
106
+ # The shipped version read a TERMINAL RENDERING and nothing else. Every context-measurement failure of
107
+ # that series is downstream of that one choice: a rendering can be truncated by pane width, can push the
108
+ # real field out of the visible window, and can leave a stale N% in scrollback for a bare numeric search
109
+ # to borrow (OBS-780). **The value is not on the screen. It is in the session JSONL**, which is exact,
110
+ # append-only, and indifferent to how wide the pane is.
111
+ #
112
+ # ⚠ NAME THE QUANTITY, because two numbers live in that file and they differ by two orders of magnitude:
113
+ # * CONTEXT FILL = the LAST usage-bearing record's input_tokens + cache_creation + cache_read.
114
+ # This is what is in the window right now. **This is the only one a clear threshold may use.**
115
+ # * CUMULATIVE CONSUMPTION = the same sum added up over every record, plus output.
116
+ # Measured 2026-08-31 on two live sessions in this workspace: **118.5x the fill over 175
117
+ # usage records, and 28.4x over 35.**
118
+ # ⛔ **THE FACTOR IS NOT A CONSTANT — IT GROWS WITH SESSION LENGTH**, because cumulative adds a
119
+ # term per request while fill is what ONE request holds. The two numbers above are two points on
120
+ # that curve, NOT a range and NOT a property of the quantity. **Never quote a single figure as
121
+ # "the" ratio**, and never average two: a longer session yields a larger one, without limit.
122
+ # The banner shows this one too, as `sum N tok`, right beside the fill percentage.
123
+ # **Quoting fill as consumption, or consumption as fill, is wrong by that factor. Say which you mean.**
124
+ #
125
+ # ⛔ AND THE DENOMINATOR IS NOT IN THE JSONL. Its `model` field reads `claude-opus-5` with no context-size
126
+ # suffix, so a 200k default would have read three live 1M seats here at 106%, 150% and 90% — the 150% seat
127
+ # would have been auto-cleared on the first tick. **A window is resolved explicitly or read once from the
128
+ # banner; it is never assumed.** Reading a per-session CONSTANT once from the fragile surface, and the
129
+ # per-tick VARIABLE from the robust one, is the trade this makes.
130
+ # Calibrated 2026-08-31 against three live seats: banner 21/30/18% vs JSONL fill 21.2/30.0/17.9%. Same
131
+ # quantity, same direction, same denominator — so the existing WARN/ACT thresholds carry over unchanged.
132
+
133
+ CTX_JSONL=""
134
+ CTX_WINDOW="${TKR_CONTEXT_WINDOW:-0}"
135
+
136
+ # TARGET is an agent name or a pane id; `herdr agent list` carries both, plus the session uuid that names
137
+ # the JSONL. The uuid is unique, so glob for it rather than reconstructing the project-dir slug — a slug
138
+ # rule is a second thing that can rot.
139
+ resolve_jsonl() {
140
+ local sid
141
+ sid=$(herdr agent list 2>/dev/null | python3 -c '
142
+ import sys, json
143
+ try: d = json.load(sys.stdin)
144
+ except Exception: sys.exit(0)
145
+ t = sys.argv[1]
146
+ for a in d.get("result", {}).get("agents", []):
147
+ if t in (a.get("pane_id"), a.get("name")):
148
+ print((a.get("agent_session") or {}).get("value") or "")
149
+ break
150
+ ' "$TARGET" 2>/dev/null) || return 1
151
+ [ -n "$sid" ] || return 1
152
+ CTX_JSONL=$(ls "$HOME"/.claude/projects/*/"$sid".jsonl 2>/dev/null | head -1)
153
+ [ -n "$CTX_JSONL" ]
154
+ }
155
+
156
+ # The banner names its own window ("Opus 5 (1M context)"). Read it ONCE; it cannot change mid-session.
157
+ resolve_window() {
158
+ [ "$CTX_WINDOW" -gt 0 ] 2>/dev/null && return 0
159
+ local w
160
+ w=$(herdr agent read "$TARGET" --source visible --lines 8 2>/dev/null |
161
+ grep -oEi '\(([0-9]+(\.[0-9]+)?)[KM] context\)' | tail -1 |
162
+ grep -oEi '[0-9]+(\.[0-9]+)?[KM]') || return 1
163
+ case "$w" in
164
+ *[Kk]) CTX_WINDOW=$(awk -v n="${w%[Kk]}" 'BEGIN{printf "%d", n*1000}') ;;
165
+ *[Mm]) CTX_WINDOW=$(awk -v n="${w%[Mm]}" 'BEGIN{printf "%d", n*1000000}') ;;
166
+ *) return 1 ;;
167
+ esac
168
+ [ "$CTX_WINDOW" -gt 0 ] 2>/dev/null
169
+ }
170
+
171
+ jsonl_pct() {
172
+ [ -n "$CTX_JSONL" ] && [ -f "$CTX_JSONL" ] && [ "$CTX_WINDOW" -gt 0 ] 2>/dev/null || return 1
173
+ python3 -c '
174
+ import sys, json
175
+ path, window = sys.argv[1], int(sys.argv[2])
176
+ fill = 0
177
+ with open(path, encoding="utf-8", errors="replace") as fh:
178
+ for line in fh:
179
+ try: rec = json.loads(line)
180
+ except Exception: continue
181
+ usage = (rec.get("message") or {}).get("usage")
182
+ if not isinstance(usage, dict): continue
183
+ # CONTEXT FILL — this request s window occupancy. Never a running total.
184
+ n = sum(int(usage.get(k) or 0) for k in
185
+ ("input_tokens", "cache_creation_input_tokens", "cache_read_input_tokens"))
186
+ if n: fill = n
187
+ if not fill: sys.exit(1)
188
+ print(int(round(fill * 100.0 / window)))
189
+ ' "$CTX_JSONL" "$CTX_WINDOW" 2>/dev/null
190
+ }
191
+
192
+ # The banner path stays as the FALLBACK, unchanged: select the model line first, then read a percentage
193
+ # only from it, so a bare numeric search cannot borrow an old N% from scrollback (OBS-780).
194
+ banner_pct() {
107
195
  local screen banner pct
108
196
  screen=$(herdr agent read "$TARGET" --source visible --lines 8 2>/dev/null) || return 1
109
197
  banner=$(printf '%s\n' "$screen" |
@@ -114,6 +202,12 @@ context_pct() {
114
202
  [ -n "$pct" ] && printf '%s\n' "$pct" || printf 'UNREADABLE\n'
115
203
  }
116
204
 
205
+ context_pct() {
206
+ local p
207
+ if p=$(jsonl_pct) && [ -n "$p" ]; then printf '%s\n' "$p"; return 0; fi
208
+ banner_pct
209
+ }
210
+
117
211
  handoff_fresh() {
118
212
  [ -n "$HANDOFF" ] || return 1
119
213
  [ -f "$HANDOFF" ] || return 1
@@ -144,6 +238,17 @@ act_on() {
144
238
  exit 0
145
239
  }
146
240
 
241
+ # Resolve the JSONL source ONCE, and SAY which surface is in use. A watcher that silently fell back to
242
+ # the fragile path looks identical to one on the robust path, and the difference is the whole point.
243
+ if resolve_jsonl && resolve_window; then
244
+ echo "CONTEXT_SOURCE jsonl $TARGET — CONTEXT FILL from $CTX_JSONL against a ${CTX_WINDOW}-token window"
245
+ echo " this is FILL (what is in the window now), never cumulative consumption — measured 28x and 118x"
246
+ echo " apart here, and that factor GROWS with session length; never quote one figure as the ratio"
247
+ else
248
+ echo "CONTEXT_SOURCE banner $TARGET — JSONL unresolved (session=${CTX_JSONL:-none} window=${CTX_WINDOW})"
249
+ echo " falling back to the rendered banner; set TKR_CONTEXT_WINDOW to use the JSONL"
250
+ fi
251
+
147
252
  warned=0
148
253
  elapsed=0
149
254
  blind=0 # seconds in the current failed-read spell (OBS-739)
@@ -0,0 +1,137 @@
1
+ #!/bin/bash
2
+ # watch-journal.sh — wake on a run's TERMINAL and DECISION events, print one reason, and exit.
3
+ #
4
+ # WHY THIS EXISTS, 2026-08-31 (OBS-808): `tickmarkr-auto` and `tickmarkr-loop` both tell the operator to
5
+ # *"watch the run journal for its terminal event rather than polling"* — and neither skill shipped a single
6
+ # script. Correct instruction, no means to follow it, so the reader polls or sleeps through the run's end.
7
+ # Worse, `task-failed` and `consult-verdict` appeared in NO shipped script at all, while the overseer skill
8
+ # instructs seats to arm watchers on all four events. This is that instrument, for the reader who has no
9
+ # supervising tier.
10
+ #
11
+ # ⚠ It is NOT a second implementation of the scoping idiom: the arm-time line baseline below is
12
+ # `watch-parks.sh`'s, kept deliberately identical, because that file paid for two traps that a fresh
13
+ # implementation re-earns (see ARM-TIME SCOPING). `watch-parks.sh` remains the AUTHORITY seat's park
14
+ # watcher — it counts parks and speaks about rulings; this one is the general four-event watcher.
15
+ #
16
+ # usage: watch-journal.sh <runs-dir> [poll-s] [cap-s] [events-csv]
17
+ # <runs-dir> e.g. .tickmarkr/runs — tracks the NEWEST run directory, so it follows a resume or a
18
+ # fresh run without being re-aimed.
19
+ # [events-csv] default `run-end,task-human,task-failed,consult-verdict` — the four the overseer skill
20
+ # names. Narrow it only when you have a reason; a watcher on fewer events is less coverage
21
+ # wearing the same name.
22
+ #
23
+ # Prints ONE wake reason and exits. RE-ARM AFTER EVERY WAKE — the gap between a wake and its re-arm is
24
+ # unwatched, and its width is however long the reader stays busy.
25
+
26
+ set -u
27
+ RUNS="${1:?runs dir required (e.g. .tickmarkr/runs)}"
28
+ POLL="${2:-20}"
29
+ CAP="${3:-28800}"
30
+ EVENTS="${4:-run-end,task-human,task-failed,consult-verdict}"
31
+
32
+ # Config flows into a regex, so validate the shape rather than trusting it (an unquoted or unchecked
33
+ # event list is a shell/regex injection and a silently-never-matching pattern at the same time).
34
+ #
35
+ # ⚠ VALIDATE THE RAW INPUT, AND NEVER NORMALISE FIRST. An earlier draft ran `tr -d '[:space:]'` before
36
+ # this check, so `run end` passed as the event name `runend` — a watcher armed on a name no journal will
37
+ # ever carry, which polls to its cap and reports "nothing happened". Caught by a control, not by use.
38
+ # **A cleanup that rescues a typo converts a loud exit into a silent never-match**, which is the exact
39
+ # failure this file exists to prevent, so whitespace is rejected rather than stripped.
40
+ case "$EVENTS" in
41
+ *[!a-z0-9,-]*|''|*,,*|,*|*,)
42
+ echo "watch-journal.sh: events must be a comma-separated list of [a-z0-9-] names with no spaces, got '$EVENTS'" >&2
43
+ exit 64 ;;
44
+ esac
45
+ ALT=$(printf '%s' "$EVENTS" | tr ',' '|')
46
+ PAT="\"event\":\"($ALT)\""
47
+
48
+ newest_journal() {
49
+ local d
50
+ d=$(ls -t "$RUNS" 2>/dev/null | grep '^run-' | head -1)
51
+ [ -n "$d" ] && [ -f "$RUNS/$d/journal.jsonl" ] && printf '%s' "$RUNS/$d/journal.jsonl"
52
+ }
53
+
54
+ # ── ARM-TIME SCOPING — the whole correctness argument, inherited from watch-parks.sh:57-64 ─────────────
55
+ # `run-end` is a HISTORICAL RECORD once written. A whole-file grep for it finds the PREVIOUS run's
56
+ # run-end on every resume, and on any re-arm after a run has ended — so the watcher exits in its first
57
+ # poll, a supervisor re-execs it into the same instant exit, and the process table shows coverage that
58
+ # does not exist. Only lines appended AFTER arming are evidence about now.
59
+ #
60
+ # Scoping by LINE POSITION rather than by an event COUNT also sidesteps the second trap that file
61
+ # documents: `grep -c` EXITS 1 WHEN THE COUNT IS ZERO while still printing `0`, so `n=$(grep -c P f ||
62
+ # echo 0)` yields the two-line string "0\n0" and every later `[ "$n" -gt … ]` dies with `integer
63
+ # expression expected` and evaluates FALSE — i.e. the armed-at-run-start watcher, the one case that
64
+ # matters, could never report its first event. This file never counts, so it cannot inherit that.
65
+ J=$(newest_journal)
66
+ base=0
67
+ [ -n "${J:-}" ] && base=$(wc -l < "$J" 2>/dev/null || echo 0)
68
+ seen_run="${J:-}"
69
+ since_arm() { tail -n +$((base + 1)) "$1" 2>/dev/null; }
70
+
71
+ field() { printf '%s' "$2" | sed -n "s/.*\"$1\":\"\([^\"]*\)\".*/\1/p" | head -1; }
72
+ bucket() { printf '%s' "$2" | sed -n "s/.*\"$1\":\[\([^]]*\)\].*/\1/p" | head -1; }
73
+
74
+ report() {
75
+ local line="$1" run="$2" ev
76
+ ev=$(field event "$line")
77
+ case "$ev" in
78
+ run-end)
79
+ local tv done failed human blocked pending verdict
80
+ tv=$(field tipVerify "$line")
81
+ done=$(bucket done "$line"); failed=$(bucket failed "$line")
82
+ human=$(bucket human "$line"); blocked=$(bucket blocked "$line")
83
+ pending=$(bucket pending "$line")
84
+ # GREEN IS A CONJUNCTION, AND THE SHORT FORM OF IT IS WRONG. "run-end plus tip verify" passes a run
85
+ # that ended `done=[T1,T3,T4] human=[T2]` — three delivered, one PARKED — and calling that green is
86
+ # how a park becomes invisible. Grade every clause here so the reader never has to remember to.
87
+ if [ "$tv" != "failed" ] && [ -z "$failed$human$blocked$pending" ]; then
88
+ verdict="GREEN"
89
+ else
90
+ verdict="NOT GREEN"
91
+ fi
92
+ echo "RUN_END $run — $verdict (tipVerify=${tv:-unknown})"
93
+ echo " done=[${done}] failed=[${failed}] human=[${human}] blocked=[${blocked}] pending=[${pending}]"
94
+ [ "$verdict" = "GREEN" ] \
95
+ && echo " all four buckets empty and tip verify is not failed — this run is green" \
96
+ || echo " a non-empty bucket above is the reason; name it, never report this run as green"
97
+ ;;
98
+ task-human)
99
+ echo "TASK_HUMAN $(field taskId "$line") — $run"
100
+ echo " $(printf '%s' "$line" | sed -n 's/.*"reason":"\([^"]\{0,160\}\).*/\1/p')"
101
+ echo " a park waits for a DECISION; read the gate evidence, then \`tickmarkr approve $run $(field taskId "$line")\` or re-scope"
102
+ ;;
103
+ task-failed)
104
+ echo "TASK_FAILED $(field taskId "$line") — $run"
105
+ echo " $(printf '%s' "$line" | sed -n 's/.*"error":"\([^"]\{0,160\}\).*/\1/p')"
106
+ echo " the run may continue on independent tasks; this task did not deliver"
107
+ ;;
108
+ consult-verdict)
109
+ echo "CONSULT_VERDICT $(field taskId "$line") action=$(field action "$line") — $run"
110
+ echo " $(printf '%s' "$line" | sed -n 's/.*"notes":"\([^"]\{0,160\}\).*/\1/p')"
111
+ ;;
112
+ *)
113
+ echo "JOURNAL_EVENT ${ev:-unparseable} — $run"
114
+ echo " ${line:0:200}"
115
+ ;;
116
+ esac
117
+ }
118
+
119
+ elapsed=0
120
+ while [ "$elapsed" -lt "$CAP" ]; do
121
+ sleep "$POLL"
122
+ elapsed=$((elapsed + POLL))
123
+
124
+ J=$(newest_journal)
125
+ [ -z "${J:-}" ] && continue
126
+
127
+ # A new run resets the baseline — its whole journal is unseen by definition.
128
+ if [ "$J" != "$seen_run" ]; then seen_run="$J"; base=0; fi
129
+
130
+ hit=$(since_arm "$J" | grep -E "$PAT" | head -1)
131
+ if [ -n "$hit" ]; then
132
+ report "$hit" "$(basename "$(dirname "$J")")"
133
+ exit 0
134
+ fi
135
+ done
136
+
137
+ echo "WATCH_CAP_REACHED — no ${EVENTS} in ${CAP}s (newest: $(basename "$(dirname "${J:-none/none}")"))"