tickmarkr 1.87.0 → 1.89.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/catalog.d.ts +18 -1
- package/dist/adapters/catalog.js +44 -1
- package/dist/adapters/fake.d.ts +2 -1
- package/dist/adapters/fake.js +7 -0
- package/dist/adapters/grok.js +11 -0
- package/dist/adapters/kimi.d.ts +2 -1
- package/dist/adapters/kimi.js +36 -0
- package/dist/adapters/opencode.js +17 -0
- package/dist/adapters/pi.js +11 -0
- package/dist/adapters/prompt.js +8 -1
- package/dist/adapters/registry.js +76 -57
- package/dist/adapters/types.d.ts +34 -3
- package/dist/adapters/types.js +99 -1
- package/dist/cli/commands/approve.d.ts +2 -0
- package/dist/cli/commands/approve.js +104 -84
- package/dist/cli/commands/compile.d.ts +1 -1
- package/dist/cli/commands/compile.js +29 -12
- package/dist/cli/commands/init.js +1 -1
- package/dist/cli/commands/plan.d.ts +1 -1
- package/dist/cli/commands/plan.js +10 -1
- package/dist/cli/commands/report.js +49 -0
- package/dist/cli/commands/status.js +298 -96
- package/dist/cli/harness.d.ts +13 -0
- package/dist/cli/harness.js +50 -0
- package/dist/compile/collateral.js +4 -4
- package/dist/compile/index.d.ts +14 -3
- package/dist/compile/index.js +36 -10
- package/dist/compile/native.js +101 -25
- package/dist/drivers/subprocess.d.ts +6 -1
- package/dist/drivers/subprocess.js +9 -4
- package/dist/gates/acceptance.d.ts +21 -1
- package/dist/gates/acceptance.js +67 -22
- package/dist/gates/artifact-manifest.d.ts +119 -0
- package/dist/gates/artifact-manifest.js +357 -0
- package/dist/gates/baseline.d.ts +6 -0
- package/dist/gates/baseline.js +52 -7
- package/dist/gates/llm.js +37 -26
- package/dist/gates/review.d.ts +16 -11
- package/dist/gates/review.js +44 -150
- package/dist/gates/run-gates.js +124 -7
- package/dist/graph/schema.d.ts +3 -1
- package/dist/graph/schema.js +4 -1
- package/dist/run/daemon.d.ts +42 -0
- package/dist/run/daemon.js +2321 -1967
- package/dist/run/git.d.ts +50 -0
- package/dist/run/git.js +113 -2
- package/dist/run/interactive-seed.d.ts +6 -2
- package/dist/run/interactive-seed.js +72 -5
- package/dist/run/journal.d.ts +9 -1
- package/dist/run/journal.js +99 -9
- package/dist/run/lock.d.ts +11 -0
- package/dist/run/lock.js +97 -6
- package/dist/run/outcome.d.ts +50 -0
- package/dist/run/outcome.js +152 -0
- package/dist/run/protocol.d.ts +460 -0
- package/dist/run/protocol.js +433 -0
- package/dist/run/supervision.d.ts +29 -0
- package/dist/run/supervision.js +189 -0
- package/fixtures/wrapped-acceptance.native.md +29 -0
- package/package.json +1 -1
- package/schema/rungraph.schema.json +21 -2
- package/skills/tickmarkr-overseer/SKILL.md +257 -5
- package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +79 -8
- package/skills/tickmarkr-overseer/scripts/watch-contamination.sh +77 -0
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +86 -0
- package/skills/tickmarkr-overseer/scripts/watch-parks.sh +96 -0
- package/skills/tickmarkr-overseer/scripts/watch-pending-input.sh +183 -0
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
#!/bin/bash
|
|
2
|
+
# watch-context.sh — wake (or act) when a seat's CONTEXT is running out, and only clear when it is SAFE.
|
|
3
|
+
#
|
|
4
|
+
# Context is the one resource a seat cannot observe about itself reliably and cannot recover from once
|
|
5
|
+
# spent. This project's law is `/clear` plus a fresh brief, NEVER `/compact` — a compaction is a lossy
|
|
6
|
+
# summary nobody trusts, while a clean session re-oriented from disk-verifiable state is reliable.
|
|
7
|
+
#
|
|
8
|
+
# The measurement everyone misses: a seat's context percentage is RENDERED IN ITS OWN STATUSLINE. It does
|
|
9
|
+
# not need to be asked, and asking is unreliable — a seat estimating its own usage is guessing.
|
|
10
|
+
#
|
|
11
|
+
# WHAT MAKES A CLEAR SAFE, and this is the whole point of the script:
|
|
12
|
+
# A clear is safe exactly when NOTHING THE SEAT IS HOLDING EXISTS ONLY IN ITS HEAD.
|
|
13
|
+
# Operationally: a handoff artifact exists AND is newer than the seat's last significant action. If the
|
|
14
|
+
# handoff is stale, the seat is holding state that a clear would destroy, and the correct move is to wake
|
|
15
|
+
# a supervisor — never to clear and hope.
|
|
16
|
+
#
|
|
17
|
+
# usage: watch-context.sh <agent|pane> <warn-pct> <act-pct> [handoff-file] [poll-s] [cap-s]
|
|
18
|
+
# TKR_AUTO_CLEAR=1 at act-pct WITH a fresh handoff, send /clear and re-brief instead of waking.
|
|
19
|
+
# TKR_REBRIEF=<path> the file the re-briefed seat is told to read (defaults to the handoff).
|
|
20
|
+
# TKR_HANDOFF_MAX_AGE_S how fresh "fresh" is (default 900).
|
|
21
|
+
|
|
22
|
+
set -u
|
|
23
|
+
TARGET="${1:?agent name or pane id required}"
|
|
24
|
+
WARN="${2:-60}"
|
|
25
|
+
ACT="${3:-75}"
|
|
26
|
+
HANDOFF="${4:-}"
|
|
27
|
+
POLL="${5:-120}"
|
|
28
|
+
CAP="${6:-28800}"
|
|
29
|
+
MAXAGE="${TKR_HANDOFF_MAX_AGE_S:-900}"
|
|
30
|
+
REBRIEF="${TKR_REBRIEF:-$HANDOFF}"
|
|
31
|
+
|
|
32
|
+
# The seat's own rendered truth. Anchor on the model marker so a percentage elsewhere on screen — a
|
|
33
|
+
# progress figure, a coverage number — cannot be mistaken for the context gauge.
|
|
34
|
+
context_pct() {
|
|
35
|
+
herdr agent read "$TARGET" --source visible --lines 8 2>/dev/null \
|
|
36
|
+
| grep '✳' | tail -1 | grep -oE '[0-9]+%' | tail -1 | tr -d '%'
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
handoff_fresh() {
|
|
40
|
+
[ -n "$HANDOFF" ] || return 1
|
|
41
|
+
[ -f "$HANDOFF" ] || return 1
|
|
42
|
+
local age now mt
|
|
43
|
+
now=$(date +%s)
|
|
44
|
+
mt=$(stat -f %m "$HANDOFF" 2>/dev/null || stat -c %Y "$HANDOFF" 2>/dev/null) || return 1
|
|
45
|
+
age=$((now - mt))
|
|
46
|
+
[ "$age" -le "$MAXAGE" ]
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
warned=0
|
|
50
|
+
elapsed=0
|
|
51
|
+
while [ "$elapsed" -lt "$CAP" ]; do
|
|
52
|
+
P=$(context_pct)
|
|
53
|
+
if [ -z "$P" ]; then
|
|
54
|
+
sleep "$POLL"; elapsed=$((elapsed + POLL)); continue
|
|
55
|
+
fi
|
|
56
|
+
|
|
57
|
+
if [ "$P" -ge "$ACT" ] 2>/dev/null; then
|
|
58
|
+
if handoff_fresh; then
|
|
59
|
+
if [ "${TKR_AUTO_CLEAR:-0}" = "1" ]; then
|
|
60
|
+
herdr agent prompt "$TARGET" "/clear" >/dev/null 2>&1
|
|
61
|
+
sleep 6
|
|
62
|
+
herdr agent prompt "$TARGET" "Read ${REBRIEF} and continue exactly where it says. Your context was cleared at ${P}% against that handoff; it is current as of $(date '+%H:%M'). Do not reconstruct from memory — everything you need is on disk." >/dev/null 2>&1
|
|
63
|
+
echo "CONTEXT_CLEARED $TARGET at ${P}% — handoff fresh, re-briefed from ${REBRIEF}"
|
|
64
|
+
exit 0
|
|
65
|
+
fi
|
|
66
|
+
echo "CONTEXT_ACT $TARGET ${P}% (>= ${ACT}) — handoff is FRESH, a clear is SAFE now"
|
|
67
|
+
echo " herdr agent prompt $TARGET \"/clear\" then re-brief from ${REBRIEF}"
|
|
68
|
+
exit 0
|
|
69
|
+
fi
|
|
70
|
+
echo "CONTEXT_ACT_UNSAFE $TARGET ${P}% (>= ${ACT}) — NO FRESH HANDOFF (${HANDOFF:-none})"
|
|
71
|
+
echo " the seat is holding state that exists only in its head; a clear would destroy it"
|
|
72
|
+
echo " make it write the handoff FIRST, then clear"
|
|
73
|
+
exit 0
|
|
74
|
+
fi
|
|
75
|
+
|
|
76
|
+
if [ "$P" -ge "$WARN" ] 2>/dev/null && [ "$warned" -eq 0 ]; then
|
|
77
|
+
warned=1
|
|
78
|
+
echo "CONTEXT_WARN $TARGET ${P}% (>= ${WARN}) — write the handoff NOW, while judgement is still good"
|
|
79
|
+
echo " a handoff written at ${ACT}% is written by a seat already degraded; that is the wrong time"
|
|
80
|
+
exit 0
|
|
81
|
+
fi
|
|
82
|
+
|
|
83
|
+
sleep "$POLL"; elapsed=$((elapsed + POLL))
|
|
84
|
+
done
|
|
85
|
+
|
|
86
|
+
echo "WATCH_CAP_REACHED $TARGET context=$(context_pct)% — no threshold crossed in ${CAP}s"
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
#!/bin/bash
|
|
2
|
+
# watch-parks.sh — wake the AUTHORITY seat on the one event that cannot proceed without it.
|
|
3
|
+
#
|
|
4
|
+
# A `task-human` park is a decision, and a decision is this seat's column. Every other tier can keep
|
|
5
|
+
# working around a park; the park itself waits for a ruling and nothing else releases it.
|
|
6
|
+
#
|
|
7
|
+
# WHY THIS EXISTS, 2026-08-07: this seat shipped an auto-supersede so a stalled orchestrator would stop
|
|
8
|
+
# needing it. That worked — ten interventions, pipeline moving, no wakes. But its only remaining wake was
|
|
9
|
+
# an artifact watcher pointed at a run-end file THAT HAD ALREADY BEEN WRITTEN, and a watcher whose trigger
|
|
10
|
+
# has passed is not coverage, it is a process that will never fire. A park was then found only because the
|
|
11
|
+
# seat happened to look — NO watcher covered parks at all.
|
|
12
|
+
#
|
|
13
|
+
# **Automating an unblock removes your NOTIFICATION without removing your RESPONSIBILITY.** Every time you
|
|
14
|
+
# make a tier need you less, re-ask what still needs you and arm for that.
|
|
15
|
+
#
|
|
16
|
+
# ⚠ The first version of this comment claimed the park sat "THREE HOURS AND THIRTEEN MINUTES". It sat TEN.
|
|
17
|
+
# The author compared UTC journal timestamps against a local wall clock (+03) and shipped the error inside
|
|
18
|
+
# a self-criticism — the one sentence nobody audits, including its writer. The gap is real and the watcher
|
|
19
|
+
# is justified by the ABSENCE OF COVERAGE, not by that number. See OBS-435.
|
|
20
|
+
#
|
|
21
|
+
# Prints ONE wake reason and exits. Re-arm after every wake.
|
|
22
|
+
#
|
|
23
|
+
# usage: watch-parks.sh <runs-dir> [poll-s] [cap-s]
|
|
24
|
+
# Tracks the NEWEST run directory, so it follows a resume or a fresh run without being re-aimed.
|
|
25
|
+
|
|
26
|
+
set -u
|
|
27
|
+
RUNS="${1:?runs dir required (e.g. .tickmarkr/runs)}"
|
|
28
|
+
POLL="${2:-45}"
|
|
29
|
+
CAP="${3:-28800}"
|
|
30
|
+
|
|
31
|
+
# `grep -c` EXITS 1 WHEN THE COUNT IS ZERO, while still printing `0`. So the idiom
|
|
32
|
+
# `n=$(grep -c PAT f || echo 0)` yields the two-line string "0\n0", and every later `[ "$n" -gt … ]`
|
|
33
|
+
# dies with `integer expression expected` and evaluates FALSE. Found 2026-08-07 by a positive control,
|
|
34
|
+
# not by use: this watcher fired correctly all day because it was always armed on a journal that ALREADY
|
|
35
|
+
# held a park, which is the branch where grep exits 0. **Armed at the start of a run — zero parks — it
|
|
36
|
+
# could never report the first park, which is the one it exists for.** A guard whose failure is silence
|
|
37
|
+
# needs a control; this one had shipped without one.
|
|
38
|
+
count_parks() {
|
|
39
|
+
local c
|
|
40
|
+
c=$(grep -c '"event":"task-human"' "$1" 2>/dev/null || true)
|
|
41
|
+
printf '%s' "${c:-0}"
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
newest_journal() {
|
|
45
|
+
local d
|
|
46
|
+
d=$(ls -t "$RUNS" 2>/dev/null | grep '^run-' | head -1)
|
|
47
|
+
[ -n "$d" ] && [ -f "$RUNS/$d/journal.jsonl" ] && printf '%s' "$RUNS/$d/journal.jsonl"
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
# Seed on the CURRENT park count so we wake on the next one, not on history already ruled.
|
|
51
|
+
J=$(newest_journal)
|
|
52
|
+
seen=0
|
|
53
|
+
[ -n "${J:-}" ] && seen=$(count_parks "$J")
|
|
54
|
+
seen_run="${J:-}"
|
|
55
|
+
|
|
56
|
+
# ...and seed a LINE POSITION, because the park counter is not enough (evidence rule 30).
|
|
57
|
+
# `run-end` is a HISTORICAL RECORD once written. A whole-file grep for it finds the PREVIOUS run-end
|
|
58
|
+
# on every resume, and on any re-arm after a run has ended — so the watcher exits in its first poll,
|
|
59
|
+
# a supervisor re-execs it into the same instant exit, and the process table shows coverage that does
|
|
60
|
+
# not exist. Measured 2026-08-07: re-arming this watcher after run 264's run-end would have done
|
|
61
|
+
# exactly that. Only lines appended AFTER arming are evidence about now.
|
|
62
|
+
base=0
|
|
63
|
+
[ -n "${J:-}" ] && base=$(wc -l < "$J" 2>/dev/null || echo 0)
|
|
64
|
+
since_arm() { tail -n +$((base + 1)) "$1" 2>/dev/null; }
|
|
65
|
+
|
|
66
|
+
elapsed=0
|
|
67
|
+
while [ "$elapsed" -lt "$CAP" ]; do
|
|
68
|
+
sleep "$POLL"
|
|
69
|
+
elapsed=$((elapsed + POLL))
|
|
70
|
+
|
|
71
|
+
J=$(newest_journal)
|
|
72
|
+
[ -z "${J:-}" ] && continue
|
|
73
|
+
|
|
74
|
+
# A new run resets the baseline — its parks AND its whole journal are unseen by definition.
|
|
75
|
+
if [ "$J" != "$seen_run" ]; then seen=0; seen_run="$J"; base=0; fi
|
|
76
|
+
|
|
77
|
+
now=$(count_parks "$J")
|
|
78
|
+
if [ "$now" -gt "$seen" ]; then
|
|
79
|
+
echo "PARK $((now - seen)) new — $(basename "$(dirname "$J")")"
|
|
80
|
+
grep '"event":"task-human"' "$J" 2>/dev/null | tail -n "$((now - seen))" \
|
|
81
|
+
| sed -n 's/.*"taskId":"\([^"]*\)".*"reason":"\([^"]\{0,160\}\).*/ \1: \2/p'
|
|
82
|
+
echo " a park waits for a RULING and nothing else releases it — read the gate evidence, then rule"
|
|
83
|
+
exit 0
|
|
84
|
+
fi
|
|
85
|
+
|
|
86
|
+
# A run that ended is also this seat's business: the milestone verdict is a decision.
|
|
87
|
+
# Scoped to lines appended since arming — see the `base` comment above.
|
|
88
|
+
if since_arm "$J" | grep -q '"event":"run-end"'; then
|
|
89
|
+
tv=$(since_arm "$J" | grep '"event":"run-end"' | tail -1 | sed -n 's/.*"tipVerify":"\([a-z]*\)".*/\1/p')
|
|
90
|
+
echo "RUN_END $(basename "$(dirname "$J")") tipVerify=${tv:-unknown}"
|
|
91
|
+
echo " read tipVerify as a FIELD; a failing verify is a DIFFERENT event with no pass field"
|
|
92
|
+
exit 0
|
|
93
|
+
fi
|
|
94
|
+
done
|
|
95
|
+
|
|
96
|
+
echo "WATCH_CAP_REACHED — no new park or run-end in ${CAP}s (newest: $(basename "$(dirname "${J:-none/none}")"))"
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
#!/bin/bash
|
|
2
|
+
# watch-pending-input.sh — wake when a seat is IDLE with UNSUBMITTED TEXT in its input box.
|
|
3
|
+
#
|
|
4
|
+
# The third quiet state, and the two standard watchers are both blind to it:
|
|
5
|
+
# blocked-state (`herdr agent wait --until blocked`) — the seat is NOT blocked, it is idle
|
|
6
|
+
# artifact (`watch-artifacts.sh`) — no file appears, and none ever will
|
|
7
|
+
# A seat in this state reports `done`, holds live work in its own prompt, and does nothing forever.
|
|
8
|
+
#
|
|
9
|
+
# Measured 2026-08-07, twice in one hour on one orchestrator: `❯ classify worker-dead-held, then author
|
|
10
|
+
# the fresh-run spec` and `❯ dry-compile ships too — add it as T13`, each sitting unsubmitted while the
|
|
11
|
+
# seat read `done`. Both were found only because a supervising seat happened to read the pane. An Enter
|
|
12
|
+
# swallowed by bracketed paste produces exactly this, and so does a draft the seat never sent.
|
|
13
|
+
#
|
|
14
|
+
# Prints ONE wake reason and exits. Re-arm after every wake.
|
|
15
|
+
#
|
|
16
|
+
# usage: watch-pending-input.sh <agent-name-or-pane> [poll-s] [cap-s] [confirm-polls]
|
|
17
|
+
|
|
18
|
+
set -u
|
|
19
|
+
TARGET="${1:?agent name or pane id required}"
|
|
20
|
+
POLL="${2:-30}"
|
|
21
|
+
CAP="${3:-14400}"
|
|
22
|
+
CONFIRM="${4:-2}" # consecutive polls before waking — a draft mid-typing is not a stall
|
|
23
|
+
|
|
24
|
+
status_of() {
|
|
25
|
+
herdr agent get "$TARGET" 2>/dev/null \
|
|
26
|
+
| sed -n 's/.*"agent_status":"\([a-z_]*\)".*/\1/p' | head -1
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
# The rendered prompt line is the condition itself. `agent_status` is the harness's OPINION about the
|
|
30
|
+
# seat and fails in both directions (a wedged worker has reported `idle`, a working one `done`), so the
|
|
31
|
+
# text in the box is the stronger signal — it is what will not run.
|
|
32
|
+
pending_text() {
|
|
33
|
+
herdr agent read "$TARGET" --source visible --lines 14 2>/dev/null \
|
|
34
|
+
| sed -n 's/^[[:space:]]*❯[[:space:]]*//p' | head -1 \
|
|
35
|
+
| sed 's/[[:space:]]*$//'
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
# Auto-supersede is OPT-IN and BUDGETED. It resubmits the seat's own draft and keeps watching instead of
|
|
39
|
+
# waking anyone. The budget exists because a seat stalling forever is a different defect from a seat
|
|
40
|
+
# stalling ten times, and silently papering over the first would hide it: when the budget runs out the
|
|
41
|
+
# watcher exits loudly with the history.
|
|
42
|
+
AUTO_MAX="${TKR_AUTO_SUPERSEDE_MAX:-20}"
|
|
43
|
+
AUTO_LOG="${TKR_AUTO_SUPERSEDE_LOG:-.tickmarkr/overseer/auto-supersede.log}"
|
|
44
|
+
auto=0
|
|
45
|
+
|
|
46
|
+
# ⚠ AUTO-SUPERSEDE SUBMITS A DRAFT THE SUPERVISING TIER NEVER READ. That is fine for "carry it to
|
|
47
|
+
# run-end" and catastrophic for "start the run", and NOTHING IN IT LOOKED AT THE CONTENT.
|
|
48
|
+
#
|
|
49
|
+
# Measured 2026-08-07. An orchestrator sat idle holding `❯ run doctor to refresh, then start the run`
|
|
50
|
+
# while the authority seat had not authorised any run — the seat's own last report said, correctly,
|
|
51
|
+
# *"run is yours to authorise."* Auto-supersede would have submitted it. **The only reason it did not is
|
|
52
|
+
# that the budget had run out at #20 one minute earlier.** Safety came from a LIMIT, not from a CHECK,
|
|
53
|
+
# and a limit is not a safety property — it is a coincidence with a counter.
|
|
54
|
+
#
|
|
55
|
+
# So: a draft naming a GATED or IRREVERSIBLE act is never auto-submitted. It wakes the supervisor, which
|
|
56
|
+
# is the one case where making the supervisor the bottleneck is the entire point.
|
|
57
|
+
# Bare verbs, not just `tickmarkr <verb>`: the draft that slipped through the first version of this
|
|
58
|
+
# matcher was the REAL one from 2026-08-07 10:26 — `approve --uphold T6, carry 2b to run-end` — which
|
|
59
|
+
# auto-supersede duly submitted. That is a GATE DECISION executed without the seat that owns gate
|
|
60
|
+
# decisions. **The cost is asymmetric: a false GATED merely wakes the supervisor, a false AUTO submits
|
|
61
|
+
# an authorisation nobody gave.** So bias toward GATED and accept the false wakes.
|
|
62
|
+
GATED_RE='tickmarkr (run|resume|approve)|(^|[^[:alnum:]])(approve|resume|uphold)([^[:alnum:]]|$)|npm publish|git (tag|push)|start the run|authoris?z?e the run'
|
|
63
|
+
|
|
64
|
+
# ⚠ THE DENYLIST ABOVE IS NOT THE SAFETY PROPERTY — THE ALLOWLIST BELOW IS.
|
|
65
|
+
#
|
|
66
|
+
# A denylist must enumerate every PHRASING of every dangerous act. It failed on 2026-08-07 at 20:32:39,
|
|
67
|
+
# submitting `run authorised — arm the four tiers and go` and STARTING A RUN NO SEAT HAD AUTHORISED.
|
|
68
|
+
# That text matches nothing above: not `start the run` and not `authoris?z?e the run` (word order), and
|
|
69
|
+
# `authorised` contains no standalone approve/resume/uphold. The incoming authority seat found it only
|
|
70
|
+
# by reading this log. The comment two paragraphs up had already named this exact catastrophe and the
|
|
71
|
+
# regex still did not implement it — **the policy was right and the matcher was a list of guesses.**
|
|
72
|
+
#
|
|
73
|
+
# So the direction is inverted: auto-supersede now requires a POSITIVE match on a known-benign shape.
|
|
74
|
+
# The two failure modes are not symmetric. An unrecognised draft that wakes the supervisor costs one
|
|
75
|
+
# wake; an unrecognised draft that is submitted costs whatever it said. Every entry in the 2026-08-07
|
|
76
|
+
# log that was genuinely safe to auto-submit is one of these shapes — the rest were instructions, and
|
|
77
|
+
# instructions are the supervisor's column by definition.
|
|
78
|
+
# ⚠ NARROWED 2026-08-07, ~40 minutes after the allowlist was written, by reading it adversarially
|
|
79
|
+
# instead of admiringly. The first version allowed `send `, `clear`, `read `, `report `, `status`.
|
|
80
|
+
# Two of those are not inert:
|
|
81
|
+
# `send ` — auto-submits `send the release to npm`, which GATED_RE does not catch (it looks for
|
|
82
|
+
# the literal `npm publish`). An allowlist entry that admits a publish is worse than
|
|
83
|
+
# the denylist it replaced, because it reads as the considered fix.
|
|
84
|
+
# `clear` — wipes the receiving seat's working context mid-run. Destructive, and irreversible in
|
|
85
|
+
# the only sense that matters: the context is gone.
|
|
86
|
+
#
|
|
87
|
+
# The line that survives scrutiny: auto-supersede may only ever cause a NOTIFICATION. Its safety then does
|
|
88
|
+
# not depend on knowing who wrote the draft — which matters here, because authorship of the first
|
|
89
|
+
# draft in each chain is UNRESOLVED (CORRECTION to OBS-448) and may stay that way. A guard whose
|
|
90
|
+
# correctness is conditional on an open question is not a guard.
|
|
91
|
+
#
|
|
92
|
+
# Everything else wakes the supervisor. That is more wakes; it is not more risk.
|
|
93
|
+
AUTO_ALLOW_RE='^(ping me|notify me|wake me)\b'
|
|
94
|
+
|
|
95
|
+
# ...and never supersede our OWN injection. The marker text lands in the seat's box, is read back as
|
|
96
|
+
# "pending input" on the next poll, and gets superseded again — nesting the message inside itself. Four
|
|
97
|
+
# of the twenty budget entries on 2026-08-07 were this, one of them doubly nested (#15). The mechanism
|
|
98
|
+
# was eating its own budget and would have exhausted it on nothing.
|
|
99
|
+
#
|
|
100
|
+
# ⚠ THIS GUARD WAS DEFEATED BY THE PANE'S WRAP BOUNDARY, and the trap is one this project has
|
|
101
|
+
# documented six times (OBS-396). `pending_text()` takes `head -1` — the FIRST rendered line of the
|
|
102
|
+
# box. A long injection wraps, so line 1 is a PREFIX of the marker: `<-- disregard everything before
|
|
103
|
+
# this`, with `arrow` on line 2. A fixed-string match on the whole sentence therefore returns FALSE on
|
|
104
|
+
# the watcher's own injection, and it re-wraps its own text and submits it again — a self-feeding loop.
|
|
105
|
+
# Measured 2026-08-07: entries #12, #14, #15, #17 in auto-supersede.log each carry the watcher's own
|
|
106
|
+
# prefix inside `$T`, and #15 shows the truncation directly. Found by the ORCHESTRATOR, not by this
|
|
107
|
+
# seat, and only because it was asked to enumerate rather than recall.
|
|
108
|
+
#
|
|
109
|
+
# So match the SHORTEST DISTINCTIVE TOKEN that cannot straddle a wrap — the same rule this project
|
|
110
|
+
# already applies to read-back probes. `<--` is three characters at position 1 of the injection.
|
|
111
|
+
# A false SELF match costs one skipped poll, which is the safe direction.
|
|
112
|
+
SELF_RE='<--|disregard'
|
|
113
|
+
|
|
114
|
+
streak=0
|
|
115
|
+
elapsed=0
|
|
116
|
+
while [ "$elapsed" -lt "$CAP" ]; do
|
|
117
|
+
sleep "$POLL"
|
|
118
|
+
elapsed=$((elapsed + POLL))
|
|
119
|
+
|
|
120
|
+
S=$(status_of)
|
|
121
|
+
if [ -z "$S" ]; then
|
|
122
|
+
echo "TARGET_GONE $TARGET — agent get returned nothing (pane closed, or the name now resolves nowhere)"
|
|
123
|
+
exit 0
|
|
124
|
+
fi
|
|
125
|
+
|
|
126
|
+
case "$S" in
|
|
127
|
+
idle|done)
|
|
128
|
+
T=$(pending_text)
|
|
129
|
+
if [ -n "$T" ]; then
|
|
130
|
+
streak=$((streak + 1))
|
|
131
|
+
if [ "$streak" -ge "$CONFIRM" ]; then
|
|
132
|
+
# TKR_AUTO_SUPERSEDE: unblock without waking the supervisor. Earned after TEN consecutive
|
|
133
|
+
# benign occurrences on one seat in one night, every draft being that seat's own correct next
|
|
134
|
+
# step. Waking a supervising tier to retype a seat's own instruction makes the supervisor the
|
|
135
|
+
# bottleneck in a loop it adds no judgement to.
|
|
136
|
+
# A gated act is the supervisor's decision, budget or no budget. Wake, never submit.
|
|
137
|
+
if printf '%s' "$T" | grep -qEi -- "$GATED_RE"; then
|
|
138
|
+
echo "PENDING_INPUT_GATED $TARGET status=$S sustained=$((streak * POLL))s"
|
|
139
|
+
echo " unsubmitted: $T"
|
|
140
|
+
echo " ⛔ this draft names a GATED or IRREVERSIBLE act, so auto-supersede REFUSED to submit it."
|
|
141
|
+
echo " It is one Enter from running. Decide it yourself, then supersede with YOUR decision:"
|
|
142
|
+
echo " herdr pane run <pane> \" <-- DISREGARD the line above (self-drafted, unauthorised). ACTUAL: …\""
|
|
143
|
+
exit 0
|
|
144
|
+
fi
|
|
145
|
+
# Never re-supersede our own injected marker — it nests the message inside itself.
|
|
146
|
+
if printf '%s' "$T" | grep -qE -- "$SELF_RE"; then
|
|
147
|
+
streak=0
|
|
148
|
+
continue
|
|
149
|
+
fi
|
|
150
|
+
if [ "${TKR_AUTO_SUPERSEDE:-0}" = "1" ] && [ "$auto" -lt "$AUTO_MAX" ] \
|
|
151
|
+
&& printf '%s' "$T" | grep -qEi -- "$AUTO_ALLOW_RE"; then
|
|
152
|
+
auto=$((auto + 1))
|
|
153
|
+
herdr agent prompt "$TARGET" \
|
|
154
|
+
" <-- disregard everything before this arrow (stale unsubmitted draft, auto-detected). ACTUAL: proceed exactly as that draft said: ${T}" \
|
|
155
|
+
>/dev/null 2>&1
|
|
156
|
+
echo "$(date '+%H:%M:%S') auto-superseded #${auto}: ${T}" >> "${AUTO_LOG}"
|
|
157
|
+
streak=0
|
|
158
|
+
continue
|
|
159
|
+
fi
|
|
160
|
+
echo "PENDING_INPUT $TARGET status=$S sustained=$((streak * POLL))s"
|
|
161
|
+
echo " unsubmitted: $T"
|
|
162
|
+
echo " the seat is idle and holding live work in its prompt — supersede it, do not re-send:"
|
|
163
|
+
echo " herdr agent prompt $TARGET \" <-- disregard everything before this arrow (stale draft). ACTUAL: …\""
|
|
164
|
+
# Name the ACTUAL reason we fell through. Saying "budget exhausted" when the cause was the
|
|
165
|
+
# allowlist misattributes it, and a misattributed cause reads exactly like a verified one.
|
|
166
|
+
if [ "${TKR_AUTO_SUPERSEDE:-0}" = "1" ]; then
|
|
167
|
+
if [ "$auto" -ge "$AUTO_MAX" ]; then
|
|
168
|
+
echo " (auto-supersede budget of $AUTO_MAX exhausted — $AUTO_LOG has the history)"
|
|
169
|
+
else
|
|
170
|
+
echo " (auto-supersede declined: draft is not on the benign allowlist — $auto/$AUTO_MAX used)"
|
|
171
|
+
fi
|
|
172
|
+
fi
|
|
173
|
+
exit 0
|
|
174
|
+
fi
|
|
175
|
+
else
|
|
176
|
+
streak=0
|
|
177
|
+
fi
|
|
178
|
+
;;
|
|
179
|
+
*) streak=0 ;;
|
|
180
|
+
esac
|
|
181
|
+
done
|
|
182
|
+
|
|
183
|
+
echo "WATCH_CAP_REACHED $TARGET status=$(status_of) — no sustained pending input in ${CAP}s"
|