tickmarkr 2.1.8 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/commands/approve.d.ts +24 -0
- package/dist/cli/commands/approve.js +121 -34
- package/dist/compile/collateral.js +149 -4
- package/dist/compile/native.js +16 -0
- package/dist/drivers/herdr.d.ts +7 -5
- package/dist/drivers/herdr.js +142 -11
- package/dist/drivers/types.d.ts +5 -6
- package/dist/drivers/types.js +2 -48
- package/dist/run/daemon.d.ts +10 -6
- package/dist/run/daemon.js +275 -22
- package/dist/run/journal.d.ts +16 -0
- package/dist/run/journal.js +51 -6
- package/dist/tui/cockpit/setup-cockpit.d.ts +8 -10
- package/dist/tui/cockpit/setup-cockpit.js +92 -53
- package/package.json +1 -1
- package/skills/tickmarkr-auto/SKILL.md +1 -1
- package/skills/tickmarkr-loop/SKILL.md +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +202 -24
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +109 -4
- package/skills/tickmarkr-overseer/scripts/watch-journal.sh +137 -0
|
@@ -38,6 +38,8 @@
|
|
|
38
38
|
# TKR_HANDOFF_MAX_AGE_S how fresh "fresh" is (default 900).
|
|
39
39
|
# TKR_CLEAR_SETTLE_S seconds to let a cleared seat settle before the re-brief (default 6).
|
|
40
40
|
# TKR_BLIND_ALARM_S seconds unreadable before CONTEXT_BLIND alarms (default 120).
|
|
41
|
+
# TKR_CONTEXT_WINDOW context window in tokens. Set it when the banner does not name one;
|
|
42
|
+
# NEVER guessed — a wrong denominator makes every warn and act line wrong.
|
|
41
43
|
|
|
42
44
|
set -u
|
|
43
45
|
ROLE="${1:?supervising seat role required: orchestrator|overseer}"
|
|
@@ -100,10 +102,96 @@ stand_down() { tickmarkr beat "$TIER" --stand-down --seat "$SEAT" >/dev/null 2>&
|
|
|
100
102
|
# record the hand-off. A killed watcher never runs it, which is the one case that must read STALE.
|
|
101
103
|
trap stand_down EXIT
|
|
102
104
|
|
|
103
|
-
#
|
|
104
|
-
#
|
|
105
|
-
#
|
|
106
|
-
|
|
105
|
+
# ── WHERE THE NUMBER COMES FROM, and this is the whole 2.1.7-series lesson ────────────────────────────
|
|
106
|
+
# The shipped version read a TERMINAL RENDERING and nothing else. Every context-measurement failure of
|
|
107
|
+
# that series is downstream of that one choice: a rendering can be truncated by pane width, can push the
|
|
108
|
+
# real field out of the visible window, and can leave a stale N% in scrollback for a bare numeric search
|
|
109
|
+
# to borrow (OBS-780). **The value is not on the screen. It is in the session JSONL**, which is exact,
|
|
110
|
+
# append-only, and indifferent to how wide the pane is.
|
|
111
|
+
#
|
|
112
|
+
# ⚠ NAME THE QUANTITY, because two numbers live in that file and they differ by two orders of magnitude:
|
|
113
|
+
# * CONTEXT FILL = the LAST usage-bearing record's input_tokens + cache_creation + cache_read.
|
|
114
|
+
# This is what is in the window right now. **This is the only one a clear threshold may use.**
|
|
115
|
+
# * CUMULATIVE CONSUMPTION = the same sum added up over every record, plus output.
|
|
116
|
+
# Measured 2026-08-31 on two live sessions in this workspace: **118.5x the fill over 175
|
|
117
|
+
# usage records, and 28.4x over 35.**
|
|
118
|
+
# ⛔ **THE FACTOR IS NOT A CONSTANT — IT GROWS WITH SESSION LENGTH**, because cumulative adds a
|
|
119
|
+
# term per request while fill is what ONE request holds. The two numbers above are two points on
|
|
120
|
+
# that curve, NOT a range and NOT a property of the quantity. **Never quote a single figure as
|
|
121
|
+
# "the" ratio**, and never average two: a longer session yields a larger one, without limit.
|
|
122
|
+
# The banner shows this one too, as `sum N tok`, right beside the fill percentage.
|
|
123
|
+
# **Quoting fill as consumption, or consumption as fill, is wrong by that factor. Say which you mean.**
|
|
124
|
+
#
|
|
125
|
+
# ⛔ AND THE DENOMINATOR IS NOT IN THE JSONL. Its `model` field reads `claude-opus-5` with no context-size
|
|
126
|
+
# suffix, so a 200k default would have read three live 1M seats here at 106%, 150% and 90% — the 150% seat
|
|
127
|
+
# would have been auto-cleared on the first tick. **A window is resolved explicitly or read once from the
|
|
128
|
+
# banner; it is never assumed.** Reading a per-session CONSTANT once from the fragile surface, and the
|
|
129
|
+
# per-tick VARIABLE from the robust one, is the trade this makes.
|
|
130
|
+
# Calibrated 2026-08-31 against three live seats: banner 21/30/18% vs JSONL fill 21.2/30.0/17.9%. Same
|
|
131
|
+
# quantity, same direction, same denominator — so the existing WARN/ACT thresholds carry over unchanged.
|
|
132
|
+
|
|
133
|
+
CTX_JSONL=""
|
|
134
|
+
CTX_WINDOW="${TKR_CONTEXT_WINDOW:-0}"
|
|
135
|
+
|
|
136
|
+
# TARGET is an agent name or a pane id; `herdr agent list` carries both, plus the session uuid that names
|
|
137
|
+
# the JSONL. The uuid is unique, so glob for it rather than reconstructing the project-dir slug — a slug
|
|
138
|
+
# rule is a second thing that can rot.
|
|
139
|
+
resolve_jsonl() {
|
|
140
|
+
local sid
|
|
141
|
+
sid=$(herdr agent list 2>/dev/null | python3 -c '
|
|
142
|
+
import sys, json
|
|
143
|
+
try: d = json.load(sys.stdin)
|
|
144
|
+
except Exception: sys.exit(0)
|
|
145
|
+
t = sys.argv[1]
|
|
146
|
+
for a in d.get("result", {}).get("agents", []):
|
|
147
|
+
if t in (a.get("pane_id"), a.get("name")):
|
|
148
|
+
print((a.get("agent_session") or {}).get("value") or "")
|
|
149
|
+
break
|
|
150
|
+
' "$TARGET" 2>/dev/null) || return 1
|
|
151
|
+
[ -n "$sid" ] || return 1
|
|
152
|
+
CTX_JSONL=$(ls "$HOME"/.claude/projects/*/"$sid".jsonl 2>/dev/null | head -1)
|
|
153
|
+
[ -n "$CTX_JSONL" ]
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
# The banner names its own window ("Opus 5 (1M context)"). Read it ONCE; it cannot change mid-session.
|
|
157
|
+
resolve_window() {
|
|
158
|
+
[ "$CTX_WINDOW" -gt 0 ] 2>/dev/null && return 0
|
|
159
|
+
local w
|
|
160
|
+
w=$(herdr agent read "$TARGET" --source visible --lines 8 2>/dev/null |
|
|
161
|
+
grep -oEi '\(([0-9]+(\.[0-9]+)?)[KM] context\)' | tail -1 |
|
|
162
|
+
grep -oEi '[0-9]+(\.[0-9]+)?[KM]') || return 1
|
|
163
|
+
case "$w" in
|
|
164
|
+
*[Kk]) CTX_WINDOW=$(awk -v n="${w%[Kk]}" 'BEGIN{printf "%d", n*1000}') ;;
|
|
165
|
+
*[Mm]) CTX_WINDOW=$(awk -v n="${w%[Mm]}" 'BEGIN{printf "%d", n*1000000}') ;;
|
|
166
|
+
*) return 1 ;;
|
|
167
|
+
esac
|
|
168
|
+
[ "$CTX_WINDOW" -gt 0 ] 2>/dev/null
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
jsonl_pct() {
|
|
172
|
+
[ -n "$CTX_JSONL" ] && [ -f "$CTX_JSONL" ] && [ "$CTX_WINDOW" -gt 0 ] 2>/dev/null || return 1
|
|
173
|
+
python3 -c '
|
|
174
|
+
import sys, json
|
|
175
|
+
path, window = sys.argv[1], int(sys.argv[2])
|
|
176
|
+
fill = 0
|
|
177
|
+
with open(path, encoding="utf-8", errors="replace") as fh:
|
|
178
|
+
for line in fh:
|
|
179
|
+
try: rec = json.loads(line)
|
|
180
|
+
except Exception: continue
|
|
181
|
+
usage = (rec.get("message") or {}).get("usage")
|
|
182
|
+
if not isinstance(usage, dict): continue
|
|
183
|
+
# CONTEXT FILL — this request s window occupancy. Never a running total.
|
|
184
|
+
n = sum(int(usage.get(k) or 0) for k in
|
|
185
|
+
("input_tokens", "cache_creation_input_tokens", "cache_read_input_tokens"))
|
|
186
|
+
if n: fill = n
|
|
187
|
+
if not fill: sys.exit(1)
|
|
188
|
+
print(int(round(fill * 100.0 / window)))
|
|
189
|
+
' "$CTX_JSONL" "$CTX_WINDOW" 2>/dev/null
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
# The banner path stays as the FALLBACK, unchanged: select the model line first, then read a percentage
|
|
193
|
+
# only from it, so a bare numeric search cannot borrow an old N% from scrollback (OBS-780).
|
|
194
|
+
banner_pct() {
|
|
107
195
|
local screen banner pct
|
|
108
196
|
screen=$(herdr agent read "$TARGET" --source visible --lines 8 2>/dev/null) || return 1
|
|
109
197
|
banner=$(printf '%s\n' "$screen" |
|
|
@@ -114,6 +202,12 @@ context_pct() {
|
|
|
114
202
|
[ -n "$pct" ] && printf '%s\n' "$pct" || printf 'UNREADABLE\n'
|
|
115
203
|
}
|
|
116
204
|
|
|
205
|
+
context_pct() {
|
|
206
|
+
local p
|
|
207
|
+
if p=$(jsonl_pct) && [ -n "$p" ]; then printf '%s\n' "$p"; return 0; fi
|
|
208
|
+
banner_pct
|
|
209
|
+
}
|
|
210
|
+
|
|
117
211
|
handoff_fresh() {
|
|
118
212
|
[ -n "$HANDOFF" ] || return 1
|
|
119
213
|
[ -f "$HANDOFF" ] || return 1
|
|
@@ -144,6 +238,17 @@ act_on() {
|
|
|
144
238
|
exit 0
|
|
145
239
|
}
|
|
146
240
|
|
|
241
|
+
# Resolve the JSONL source ONCE, and SAY which surface is in use. A watcher that silently fell back to
|
|
242
|
+
# the fragile path looks identical to one on the robust path, and the difference is the whole point.
|
|
243
|
+
if resolve_jsonl && resolve_window; then
|
|
244
|
+
echo "CONTEXT_SOURCE jsonl $TARGET — CONTEXT FILL from $CTX_JSONL against a ${CTX_WINDOW}-token window"
|
|
245
|
+
echo " this is FILL (what is in the window now), never cumulative consumption — measured 28x and 118x"
|
|
246
|
+
echo " apart here, and that factor GROWS with session length; never quote one figure as the ratio"
|
|
247
|
+
else
|
|
248
|
+
echo "CONTEXT_SOURCE banner $TARGET — JSONL unresolved (session=${CTX_JSONL:-none} window=${CTX_WINDOW})"
|
|
249
|
+
echo " falling back to the rendered banner; set TKR_CONTEXT_WINDOW to use the JSONL"
|
|
250
|
+
fi
|
|
251
|
+
|
|
147
252
|
warned=0
|
|
148
253
|
elapsed=0
|
|
149
254
|
blind=0 # seconds in the current failed-read spell (OBS-739)
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
#!/bin/bash
|
|
2
|
+
# watch-journal.sh — wake on a run's TERMINAL and DECISION events, print one reason, and exit.
|
|
3
|
+
#
|
|
4
|
+
# WHY THIS EXISTS, 2026-08-31 (OBS-808): `tickmarkr-auto` and `tickmarkr-loop` both tell the operator to
|
|
5
|
+
# *"watch the run journal for its terminal event rather than polling"* — and neither skill shipped a single
|
|
6
|
+
# script. Correct instruction, no means to follow it, so the reader polls or sleeps through the run's end.
|
|
7
|
+
# Worse, `task-failed` and `consult-verdict` appeared in NO shipped script at all, while the overseer skill
|
|
8
|
+
# instructs seats to arm watchers on all four events. This is that instrument, for the reader who has no
|
|
9
|
+
# supervising tier.
|
|
10
|
+
#
|
|
11
|
+
# ⚠ It is NOT a second implementation of the scoping idiom: the arm-time line baseline below is
|
|
12
|
+
# `watch-parks.sh`'s, kept deliberately identical, because that file paid for two traps that a fresh
|
|
13
|
+
# implementation re-earns (see ARM-TIME SCOPING). `watch-parks.sh` remains the AUTHORITY seat's park
|
|
14
|
+
# watcher — it counts parks and speaks about rulings; this one is the general four-event watcher.
|
|
15
|
+
#
|
|
16
|
+
# usage: watch-journal.sh <runs-dir> [poll-s] [cap-s] [events-csv]
|
|
17
|
+
# <runs-dir> e.g. .tickmarkr/runs — tracks the NEWEST run directory, so it follows a resume or a
|
|
18
|
+
# fresh run without being re-aimed.
|
|
19
|
+
# [events-csv] default `run-end,task-human,task-failed,consult-verdict` — the four the overseer skill
|
|
20
|
+
# names. Narrow it only when you have a reason; a watcher on fewer events is less coverage
|
|
21
|
+
# wearing the same name.
|
|
22
|
+
#
|
|
23
|
+
# Prints ONE wake reason and exits. RE-ARM AFTER EVERY WAKE — the gap between a wake and its re-arm is
|
|
24
|
+
# unwatched, and its width is however long the reader stays busy.
|
|
25
|
+
|
|
26
|
+
set -u
|
|
27
|
+
RUNS="${1:?runs dir required (e.g. .tickmarkr/runs)}"
|
|
28
|
+
POLL="${2:-20}"
|
|
29
|
+
CAP="${3:-28800}"
|
|
30
|
+
EVENTS="${4:-run-end,task-human,task-failed,consult-verdict}"
|
|
31
|
+
|
|
32
|
+
# Config flows into a regex, so validate the shape rather than trusting it (an unquoted or unchecked
|
|
33
|
+
# event list is a shell/regex injection and a silently-never-matching pattern at the same time).
|
|
34
|
+
#
|
|
35
|
+
# ⚠ VALIDATE THE RAW INPUT, AND NEVER NORMALISE FIRST. An earlier draft ran `tr -d '[:space:]'` before
|
|
36
|
+
# this check, so `run end` passed as the event name `runend` — a watcher armed on a name no journal will
|
|
37
|
+
# ever carry, which polls to its cap and reports "nothing happened". Caught by a control, not by use.
|
|
38
|
+
# **A cleanup that rescues a typo converts a loud exit into a silent never-match**, which is the exact
|
|
39
|
+
# failure this file exists to prevent, so whitespace is rejected rather than stripped.
|
|
40
|
+
case "$EVENTS" in
|
|
41
|
+
*[!a-z0-9,-]*|''|*,,*|,*|*,)
|
|
42
|
+
echo "watch-journal.sh: events must be a comma-separated list of [a-z0-9-] names with no spaces, got '$EVENTS'" >&2
|
|
43
|
+
exit 64 ;;
|
|
44
|
+
esac
|
|
45
|
+
ALT=$(printf '%s' "$EVENTS" | tr ',' '|')
|
|
46
|
+
PAT="\"event\":\"($ALT)\""
|
|
47
|
+
|
|
48
|
+
newest_journal() {
|
|
49
|
+
local d
|
|
50
|
+
d=$(ls -t "$RUNS" 2>/dev/null | grep '^run-' | head -1)
|
|
51
|
+
[ -n "$d" ] && [ -f "$RUNS/$d/journal.jsonl" ] && printf '%s' "$RUNS/$d/journal.jsonl"
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
# ── ARM-TIME SCOPING — the whole correctness argument, inherited from watch-parks.sh:57-64 ─────────────
|
|
55
|
+
# `run-end` is a HISTORICAL RECORD once written. A whole-file grep for it finds the PREVIOUS run's
|
|
56
|
+
# run-end on every resume, and on any re-arm after a run has ended — so the watcher exits in its first
|
|
57
|
+
# poll, a supervisor re-execs it into the same instant exit, and the process table shows coverage that
|
|
58
|
+
# does not exist. Only lines appended AFTER arming are evidence about now.
|
|
59
|
+
#
|
|
60
|
+
# Scoping by LINE POSITION rather than by an event COUNT also sidesteps the second trap that file
|
|
61
|
+
# documents: `grep -c` EXITS 1 WHEN THE COUNT IS ZERO while still printing `0`, so `n=$(grep -c P f ||
|
|
62
|
+
# echo 0)` yields the two-line string "0\n0" and every later `[ "$n" -gt … ]` dies with `integer
|
|
63
|
+
# expression expected` and evaluates FALSE — i.e. the armed-at-run-start watcher, the one case that
|
|
64
|
+
# matters, could never report its first event. This file never counts, so it cannot inherit that.
|
|
65
|
+
J=$(newest_journal)
|
|
66
|
+
base=0
|
|
67
|
+
[ -n "${J:-}" ] && base=$(wc -l < "$J" 2>/dev/null || echo 0)
|
|
68
|
+
seen_run="${J:-}"
|
|
69
|
+
since_arm() { tail -n +$((base + 1)) "$1" 2>/dev/null; }
|
|
70
|
+
|
|
71
|
+
field() { printf '%s' "$2" | sed -n "s/.*\"$1\":\"\([^\"]*\)\".*/\1/p" | head -1; }
|
|
72
|
+
bucket() { printf '%s' "$2" | sed -n "s/.*\"$1\":\[\([^]]*\)\].*/\1/p" | head -1; }
|
|
73
|
+
|
|
74
|
+
report() {
|
|
75
|
+
local line="$1" run="$2" ev
|
|
76
|
+
ev=$(field event "$line")
|
|
77
|
+
case "$ev" in
|
|
78
|
+
run-end)
|
|
79
|
+
local tv done failed human blocked pending verdict
|
|
80
|
+
tv=$(field tipVerify "$line")
|
|
81
|
+
done=$(bucket done "$line"); failed=$(bucket failed "$line")
|
|
82
|
+
human=$(bucket human "$line"); blocked=$(bucket blocked "$line")
|
|
83
|
+
pending=$(bucket pending "$line")
|
|
84
|
+
# GREEN IS A CONJUNCTION, AND THE SHORT FORM OF IT IS WRONG. "run-end plus tip verify" passes a run
|
|
85
|
+
# that ended `done=[T1,T3,T4] human=[T2]` — three delivered, one PARKED — and calling that green is
|
|
86
|
+
# how a park becomes invisible. Grade every clause here so the reader never has to remember to.
|
|
87
|
+
if [ "$tv" != "failed" ] && [ -z "$failed$human$blocked$pending" ]; then
|
|
88
|
+
verdict="GREEN"
|
|
89
|
+
else
|
|
90
|
+
verdict="NOT GREEN"
|
|
91
|
+
fi
|
|
92
|
+
echo "RUN_END $run — $verdict (tipVerify=${tv:-unknown})"
|
|
93
|
+
echo " done=[${done}] failed=[${failed}] human=[${human}] blocked=[${blocked}] pending=[${pending}]"
|
|
94
|
+
[ "$verdict" = "GREEN" ] \
|
|
95
|
+
&& echo " all four buckets empty and tip verify is not failed — this run is green" \
|
|
96
|
+
|| echo " a non-empty bucket above is the reason; name it, never report this run as green"
|
|
97
|
+
;;
|
|
98
|
+
task-human)
|
|
99
|
+
echo "TASK_HUMAN $(field taskId "$line") — $run"
|
|
100
|
+
echo " $(printf '%s' "$line" | sed -n 's/.*"reason":"\([^"]\{0,160\}\).*/\1/p')"
|
|
101
|
+
echo " a park waits for a DECISION; read the gate evidence, then \`tickmarkr approve $run $(field taskId "$line")\` or re-scope"
|
|
102
|
+
;;
|
|
103
|
+
task-failed)
|
|
104
|
+
echo "TASK_FAILED $(field taskId "$line") — $run"
|
|
105
|
+
echo " $(printf '%s' "$line" | sed -n 's/.*"error":"\([^"]\{0,160\}\).*/\1/p')"
|
|
106
|
+
echo " the run may continue on independent tasks; this task did not deliver"
|
|
107
|
+
;;
|
|
108
|
+
consult-verdict)
|
|
109
|
+
echo "CONSULT_VERDICT $(field taskId "$line") action=$(field action "$line") — $run"
|
|
110
|
+
echo " $(printf '%s' "$line" | sed -n 's/.*"notes":"\([^"]\{0,160\}\).*/\1/p')"
|
|
111
|
+
;;
|
|
112
|
+
*)
|
|
113
|
+
echo "JOURNAL_EVENT ${ev:-unparseable} — $run"
|
|
114
|
+
echo " ${line:0:200}"
|
|
115
|
+
;;
|
|
116
|
+
esac
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
elapsed=0
|
|
120
|
+
while [ "$elapsed" -lt "$CAP" ]; do
|
|
121
|
+
sleep "$POLL"
|
|
122
|
+
elapsed=$((elapsed + POLL))
|
|
123
|
+
|
|
124
|
+
J=$(newest_journal)
|
|
125
|
+
[ -z "${J:-}" ] && continue
|
|
126
|
+
|
|
127
|
+
# A new run resets the baseline — its whole journal is unseen by definition.
|
|
128
|
+
if [ "$J" != "$seen_run" ]; then seen_run="$J"; base=0; fi
|
|
129
|
+
|
|
130
|
+
hit=$(since_arm "$J" | grep -E "$PAT" | head -1)
|
|
131
|
+
if [ -n "$hit" ]; then
|
|
132
|
+
report "$hit" "$(basename "$(dirname "$J")")"
|
|
133
|
+
exit 0
|
|
134
|
+
fi
|
|
135
|
+
done
|
|
136
|
+
|
|
137
|
+
echo "WATCH_CAP_REACHED — no ${EVENTS} in ${CAP}s (newest: $(basename "$(dirname "${J:-none/none}")"))"
|