tickmarkr 2.6.1 → 2.6.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (85) hide show
  1. package/README.md +14 -3
  2. package/dist/adapters/catalog-remote.js +89 -47
  3. package/dist/adapters/claude-code.js +9 -6
  4. package/dist/adapters/codex.js +7 -4
  5. package/dist/adapters/prompt.d.ts +1 -0
  6. package/dist/adapters/prompt.js +14 -6
  7. package/dist/adapters/registry.js +3 -3
  8. package/dist/adapters/types.d.ts +12 -4
  9. package/dist/adapters/types.js +6 -0
  10. package/dist/cli/commands/approve.d.ts +5 -1
  11. package/dist/cli/commands/approve.js +66 -23
  12. package/dist/cli/commands/compile.js +13 -3
  13. package/dist/cli/commands/doctor.d.ts +2 -0
  14. package/dist/cli/commands/doctor.js +11 -3
  15. package/dist/cli/commands/fleet.js +45 -7
  16. package/dist/cli/commands/report.js +18 -2
  17. package/dist/cli/commands/resume.js +4 -2
  18. package/dist/cli/commands/status.js +24 -19
  19. package/dist/cli/help.d.ts +2 -0
  20. package/dist/cli/help.js +9 -2
  21. package/dist/config/config.d.ts +20 -0
  22. package/dist/config/config.js +47 -8
  23. package/dist/config/fleet-overlay.d.ts +1 -0
  24. package/dist/config/fleet-overlay.js +56 -0
  25. package/dist/drivers/orca.d.ts +26 -1
  26. package/dist/drivers/orca.js +193 -60
  27. package/dist/eval/canary.d.ts +2 -1
  28. package/dist/eval/canary.js +2 -2
  29. package/dist/eval/dispatch.js +1 -0
  30. package/dist/gates/acceptance.d.ts +2 -1
  31. package/dist/gates/acceptance.js +7 -2
  32. package/dist/gates/baseline.d.ts +12 -1
  33. package/dist/gates/baseline.js +11 -4
  34. package/dist/gates/llm.d.ts +5 -4
  35. package/dist/gates/llm.js +13 -13
  36. package/dist/gates/review.d.ts +8 -0
  37. package/dist/gates/review.js +40 -4
  38. package/dist/gates/run-gates.d.ts +2 -1
  39. package/dist/gates/run-gates.js +27 -13
  40. package/dist/gates/test-manifest.d.ts +3 -1
  41. package/dist/gates/test-manifest.js +9 -2
  42. package/dist/graph/schema.d.ts +2 -0
  43. package/dist/graph/schema.js +2 -0
  44. package/dist/plan/scope.js +2 -2
  45. package/dist/route/preference.d.ts +20 -2
  46. package/dist/route/preference.js +48 -13
  47. package/dist/route/router.js +30 -15
  48. package/dist/run/consult.d.ts +13 -1
  49. package/dist/run/consult.js +14 -5
  50. package/dist/run/daemon.d.ts +37 -2
  51. package/dist/run/daemon.js +579 -141
  52. package/dist/run/git.d.ts +8 -0
  53. package/dist/run/git.js +14 -0
  54. package/dist/run/journal.d.ts +126 -3
  55. package/dist/run/journal.js +410 -37
  56. package/dist/run/merge.d.ts +3 -1
  57. package/dist/run/merge.js +3 -2
  58. package/dist/run/operator-summary.d.ts +3 -0
  59. package/dist/run/operator-summary.js +3 -1
  60. package/dist/run/protocol.d.ts +31 -1
  61. package/dist/run/protocol.js +3 -1
  62. package/dist/run/supervision.d.ts +7 -1
  63. package/dist/run/supervision.js +5 -2
  64. package/dist/tui/cockpit/board.js +3 -3
  65. package/dist/tui/cockpit/decision-actions.d.ts +8 -5
  66. package/dist/tui/cockpit/decision-actions.js +55 -32
  67. package/dist/tui/cockpit/derive.js +13 -2
  68. package/dist/tui/cockpit/live-runtime.d.ts +10 -0
  69. package/dist/tui/cockpit/live-runtime.js +50 -3
  70. package/dist/tui/cockpit/run-cockpit.d.ts +3 -0
  71. package/dist/tui/cockpit/run-cockpit.js +26 -1
  72. package/dist/tui/cockpit/run-view.d.ts +9 -3
  73. package/dist/tui/cockpit/run-view.js +60 -7
  74. package/dist/tui/cockpit/setup-cockpit.d.ts +4 -0
  75. package/dist/tui/cockpit/setup-cockpit.js +6 -3
  76. package/dist/tui/ink/fleet-app.d.ts +15 -3
  77. package/dist/tui/ink/fleet-app.js +91 -22
  78. package/package.json +2 -1
  79. package/schema/config.schema.json +818 -0
  80. package/skills/tickmarkr-loop/SKILL.md +8 -2
  81. package/skills/tickmarkr-overseer/SKILL.md +42 -0
  82. package/skills/tickmarkr-overseer/scripts/classify-vitest-log.sh +88 -0
  83. package/skills/tickmarkr-overseer/scripts/context-statusline.sh +81 -0
  84. package/skills/tickmarkr-overseer/scripts/grade-ci.sh +36 -34
  85. package/skills/tickmarkr-overseer/scripts/watch-journal.sh +6 -4
@@ -112,8 +112,14 @@ enacts at its next task boundary; if a different live run owns the repository lo
112
112
  for that run to end before resuming this one.
113
113
 
114
114
  For a non-TTY decision, the same command is
115
- `tickmarkr approve <runId> T2 --by operator --reason 'ready to proceed'`; check its receipt
116
- and explicitly resume. Preserve resume refusals and repair the named source/config issue
115
+ `tickmarkr approve <runId> T2 --park <line>@<ts> --by operator --reason 'ready to proceed'`,
116
+ naming the park token `tickmarkr status <runId>` prints; check its receipt and explicitly
117
+ resume. Every decision binds to one park (a waive also to its failed gate:
118
+ `tickmarkr approve <runId> T2 --waive --park <line>@<ts> --gate review`); once a newer park
119
+ opens, a stale token, unbound release or mismatched gate refuses — the daemon journals
120
+ `approval-refused` and never waives the newer gate. A failed task keeps its recheck with its
121
+ own bound failure token: status prints `failed — T3 — failure <line>@<ts>`, then
122
+ `tickmarkr approve <runId> T3 --recheck --park <line>@<ts>` re-gates its landed commits. Preserve resume refusals and repair the named source/config issue
117
123
  (including deny/prefer conflicts); never edit the compiled graph to force a result. Resume
118
124
  makes CURRENT TIP PENDING; historical GATES RAN does not prove completion. The completed
119
125
  case has 3/3 recorded merges, a latest run-end, a nonfailed known tip result and empty
@@ -18,6 +18,14 @@ Fleet/Bootstrap and Plan/Health remain follow-ons with existing CLI entries. A r
18
18
  1/3 merged, human T2, blocked T3 run is PARTIAL despite tip pass. The orchestrator owns
19
19
  the Run confirmation and read-back of the appended `task-approved` receipt after a ruling;
20
20
  permission is not dispatch, and a closed run needs explicit `tickmarkr resume <runId>`.
21
+ A ruling is bound to the park it decides (OBS-1178): relay it with that park's token, e.g.
22
+ `tickmarkr approve <runId> T2 --waive --park <line>@<ts> --gate review`, copied from `tickmarkr
23
+ status <runId>` or the park notice. A standing order or queued ruling written
24
+ against one park is refused once a newer park opens (the daemon journals `approval-refused`) — rule
25
+ again on the new token; never re-issue the old one (D-470: an unbound approve waived a RED test
26
+ gate). A failed task keeps its recheck with its bound failure token: status prints
27
+ `failed — T3 — failure <line>@<ts>`, and `tickmarkr approve <runId> T3 --recheck --park <line>@<ts>`
28
+ re-gates its landed commits with no worker.
21
29
  Manual UI retains the receipt; the daemon-owned board gracefully stands down only its own
22
30
  presence before closing its owned pane. Non-TTY supervision keeps `status`, `report` and
23
31
  default-watch line output (`--watch --plain` is also available on a TTY). Keep the canonical
@@ -744,6 +752,40 @@ number — an unmeasured budget is not a small budget.
744
752
 
745
753
  Write the handoff and announce it with `orca terminal send --terminal <handle> --text "Read <handoff> and <brief>" --enter --wait-submit 15 --json`. Require observed submission (`result.send.prompt.stages.includes("turn_started")`) and inspect `orca terminal read --terminal <handle> --screen --json` before and after clearing. Arm file/journal and context evidence watchers for both seats.
746
754
 
755
+ #### Opt-in Claude context sidecar — a Claude seat's `%` from disk (OBS-1122)
756
+
757
+ A Claude seat's fill can also be read from a file instead of off its screen. It is opt-in, per seat,
758
+ Claude only, and fully reversible. Claude has ONE `statusLine` slot and it belongs to the operator, so
759
+ `scripts/context-statusline.sh` never takes it over: it **chains the operator's existing status
760
+ command** — runs it with the same stdin and returns its stdout and exit status unchanged — and only as a
761
+ side effect writes the payload's `context_window.used_percentage` (nothing else) to
762
+ `.tickmarkr/overseer/context/<seat>.json` under the launch-supplied root, atomically (temp file +
763
+ rename).
764
+
765
+ 1. **Record the operator's current `statusLine.command` verbatim in the brief** — it is the reversal.
766
+ 2. **The operator wraps it; this skill never edits their home configuration.** The wrapped command is
767
+ `<repo>/.claude/skills/tickmarkr-overseer/scripts/context-statusline.sh '<existing statusLine command>'`,
768
+ set in the same settings file, or in that one seat's launch `--settings` JSON so it ends with the seat.
769
+ With no existing command, omit the argument; the collector then prints nothing.
770
+ 3. **The launch recipe names the seat and its root:** start the Claude seat with
771
+ `TKR_CONTEXT_SEAT=<seat>` and `TKR_CONTEXT_ROOT=<absolute repo path>` in its environment. Only a safe
772
+ basename (`[A-Za-z0-9][A-Za-z0-9._-]{0,63}`) and an absolute root are accepted; anything else writes
773
+ nothing — never a sanitized name that could land in another seat's file — and a seat launched without
774
+ either collects nothing while its statusline still chains. The root never comes from the payload or
775
+ the seat's cwd: both can differ from the last valid payload's directory, which would let a malformed
776
+ payload's invalidation land in another tree while the stale number stayed readable.
777
+ 4. **Read:** `context-statusline.sh --read <seat> [<root>]` prints the percentage, or `unknown` when the
778
+ file is absent, malformed, attributed to another seat, older than 120 seconds, or the payload carried
779
+ no percentage. A malformed payload overwrites the seat's last number with no percentage, so a stale
780
+ good reading never outlives it. **`unknown` is never 0 %** — it sends you back to the screen read,
781
+ never to headroom.
782
+ 5. **Reverse:** restore the recorded command in the settings it came from, then remove
783
+ `.tickmarkr/overseer/context/`.
784
+
785
+ **Codex seats stay screen-read.** The collector reads Claude's statusLine payload only, so it measures
786
+ nothing about a Codex seat — its `--read` stays `unknown` — and a Codex seat's fill is read off its screen
787
+ exactly as above.
788
+
747
789
  ### A GO has a deadline — arm the launch observer in the same act as the GO
748
790
 
749
791
  #### On herdr (`HERDR_ENV=1`) — launch watcher
@@ -0,0 +1,88 @@
1
+ #!/bin/bash
2
+ # classify-vitest-log.sh <log> — the ONE strict Vitest log classifier (OBS-1184). The public CI
3
+ # wrapper (scripts/run-ci-vitest.sh) and grade-ci.sh both call it, so the badge and the grade cannot
4
+ # disagree. Reads a raw Vitest log or a gh job log (gh's "job<TAB>step<TAB>timestamp " prefix and ANSI
5
+ # styling — the raw ESC byte and gh's `^[` rendering of it — are stripped) and prints ONE line:
6
+ # VITEST_LOG verdict=<CLEAN|RPC_ONLY|RED> reason=<why> summaries=S passed=P skipped=K failed=F
7
+ # timedout=T unhandled=U runner_rpc_timeouts=R coverage_misses=C signals=G terminal_errors=E
8
+ # unhandled counts the unhandled errors that are NOT runner RPC timeouts. Exit 0 CLEAN or RPC_ONLY, 1 RED.
9
+ #
10
+ # RED unless every Vitest summary is complete (Test Files, Tests and Duration lines, and every
11
+ # "Unhandled Errors" block closed by the summary that follows it — a cut-off log is RED), with no
12
+ # failed file or test, no timed-out test or hook, no coverage-threshold ERROR, no npm signal line,
13
+ # no unhandled error other than a vitest-worker RPC timeout, and no terminal error: a structural
14
+ # "Unhandled Error", "Startup Error" or "Collect Error" banner outside Vitest's counted block — how
15
+ # Vitest reports an exception thrown after the summary, such as a coverage-reporting failure, before it
16
+ # exits 1 (terminal_errors counts them; an RPC timeout never offsets one). RPC_ONLY means unhandled
17
+ # errors were present and EVERY one is such a timeout (OBS-1058: the starved 2-core runner talking,
18
+ # not a test).
19
+ #
20
+ # Which timeouts count: only one that is ITSELF an unhandled entry — the first payload line under a
21
+ # per-error header, nothing later. The header is Vitest's own structural line — a run of ⎯ on each side
22
+ # of "Unhandled Error", nothing else on the line — so prose that merely contains the words never opens
23
+ # a window, and the payload must be Vitest's whole line (`Error: [vitest-worker]: Timeout calling
24
+ # "<method>"`), so an assertion quoting it stays RED. Per-error headers are honoured only inside
25
+ # Vitest's own block — after its structural "Unhandled Errors" line and "Vitest caught N" tally, until
26
+ # the file summary — and each block's count never exceeds its N, so a test that prints a header and a
27
+ # timeout to stdout forges nothing. The tallies must also sum to the summaries' own "Errors N" lines
28
+ # (Vitest prints both from one list), so a forged or a missing block is RED as well.
29
+ set -u
30
+
31
+ log=${1:?log path required}
32
+ fields='summaries=0 passed=0 skipped=0 failed=0 timedout=0 unhandled=0 runner_rpc_timeouts=0 coverage_misses=0 signals=0 terminal_errors=0'
33
+ if [ ! -s "$log" ] || [ ! -r "$log" ]; then
34
+ echo "VITEST_LOG verdict=RED reason=unreadable $fields"
35
+ exit 1
36
+ fi
37
+
38
+ esc=$(printf '\033')
39
+ awk -v esc="$esc" '
40
+ function num(s, word) { return match(s, "[0-9]+ " word) ? substr(s, RSTART, RLENGTH) + 0 : 0 }
41
+ function close_block() { rpc += (found < cap ? found : cap); armed = 0; pending = 0 }
42
+ {
43
+ line = $0
44
+ gsub(/\^\[\[[0-9;]*m/, "", line); gsub(esc "\\[[0-9;]*m", "", line)
45
+ sub(/^[^\t]*\t[^\t]*\t[0-9T:.-]+Z[ ]?/, "", line)
46
+ }
47
+ line ~ /Test timed out|Error: Hook timed out/ { timedout++ }
48
+ line ~ /ERROR: Coverage for .* does not meet .*threshold/ { coverage++ }
49
+ line ~ /^npm (error|ERR!) signal / { signals++ }
50
+ line ~ /^(⎯)+ Unhandled Errors (⎯)+[ \t]*$/ { if (armed) close_block(); opening = 1; next }
51
+ opening && line !~ /[^ \t]/ { next }
52
+ opening {
53
+ opening = 0
54
+ if (line ~ /^Vitest caught [0-9]+ unhandled error/) { armed = 1; found = 0; cap = num(line, "unhandled"); tally += cap }
55
+ else untallied++
56
+ }
57
+ line ~ /^[ \t]*Test Files[ \t]/ {
58
+ if (armed) close_block()
59
+ summaries++; in_summary = 1; saw_tests = 0
60
+ failed += num(line, "failed"); passed += num(line, "passed"); skipped += num(line, "skipped")
61
+ next
62
+ }
63
+ in_summary && line ~ /^[ \t]*Tests[ \t]/ { saw_tests = 1; failed_tests += num(line, "failed"); next }
64
+ in_summary && line ~ /^[ \t]*Errors[ \t]+[0-9]+ error/ { summary_errors += num(line, "error"); next }
65
+ in_summary && line ~ /^[ \t]*Duration[ \t]/ { if (saw_tests) complete++; in_summary = 0; next }
66
+ armed && line ~ /^(⎯)+ Unhandled Error (⎯)+[ \t]*$/ { pending = 1; next }
67
+ line ~ /^(⎯)+ (Unhandled|Startup|Collect) Error (⎯)+[ \t]*$/ { terminal++; next }
68
+ pending && line ~ /[^ \t]/ { pending = 0; if (line ~ /^Error: \[vitest-worker\]: Timeout calling "[A-Za-z]+"[ \t]*$/) found++ }
69
+ END {
70
+ unhandled = tally + untallied
71
+ if (summary_errors > unhandled) unhandled = summary_errors
72
+ other = unhandled - rpc
73
+ if (summaries == 0) reason = "no-summary"
74
+ else if (complete != summaries || opening || armed) reason = "truncated"
75
+ else if (failed || failed_tests) reason = "failed"
76
+ else if (timedout) reason = "timed-out"
77
+ else if (coverage) reason = "coverage-threshold"
78
+ else if (signals) reason = "signal"
79
+ else if (terminal) reason = "terminal-error"
80
+ else if (other > 0) reason = "unhandled-error"
81
+ else if (tally + untallied != summary_errors) reason = "unhandled-mismatch"
82
+ verdict = reason != "" ? "RED" : (unhandled > 0 ? "RPC_ONLY" : "CLEAN")
83
+ if (reason == "") reason = "none"
84
+ printf "VITEST_LOG verdict=%s reason=%s summaries=%d passed=%d skipped=%d failed=%d timedout=%d unhandled=%d runner_rpc_timeouts=%d coverage_misses=%d signals=%d terminal_errors=%d\n", \
85
+ verdict, reason, summaries, passed, skipped, failed, timedout, other, rpc, coverage, signals, terminal
86
+ exit (verdict == "RED")
87
+ }
88
+ ' "$log"
@@ -0,0 +1,81 @@
1
+ #!/bin/bash
2
+ # context-statusline.sh — opt-in Claude statusLine sidecar: a seat's context % on disk (OBS-1122).
3
+ #
4
+ # Claude has ONE statusLine slot and it is the operator's. This collector never takes it over: it CHAINS
5
+ # the operator's existing command — same stdin, its stdout and exit status returned unchanged — and only
6
+ # as a side effect writes the payload's `context_window.used_percentage` to
7
+ # `$TKR_CONTEXT_ROOT/.tickmarkr/overseer/context/<seat>.json`, atomically (temp file + rename).
8
+ # Nothing else from the payload is kept. A malformed payload still writes the seat's record — with no
9
+ # percentage — so the last good number is invalidated, never left standing. The destination never
10
+ # depends on the payload or the process cwd: a malformed payload names no directory, and a cwd can drift,
11
+ # so either would let the invalidation land in another tree while the stale number stays readable.
12
+ # Every collector failure is swallowed: the operator's statusline must never break because the sidecar did.
13
+ #
14
+ # Seat AND root come ONLY from the seat's launch recipe: TKR_CONTEXT_SEAT must be a safe basename
15
+ # ([A-Za-z0-9][A-Za-z0-9._-]{0,63}) and TKR_CONTEXT_ROOT an absolute path. Either unset or unsafe means
16
+ # nothing is written — never a sanitized name, which could land in another seat's file.
17
+ #
18
+ # Reading is fail-closed to `unknown`, NEVER 0: absent, malformed, attributed to another seat, older than
19
+ # 120 s (or dated in the future), or a payload that carried no percentage all read `unknown`. Claude only:
20
+ # a Codex seat has no such payload and stays screen-read.
21
+ #
22
+ # usage:
23
+ # statusLine command: TKR_CONTEXT_SEAT=<seat> TKR_CONTEXT_ROOT=<abs repo> in the seat's env, then
24
+ # context-statusline.sh '<existing statusLine command>' (no argument: prints nothing)
25
+ # read: context-statusline.sh --read <seat> [<project-dir>] → <pct> | unknown
26
+ set -u
27
+
28
+ PY='
29
+ import json, os, re, sys, tempfile, time
30
+ MAX_AGE_S = 120
31
+ mode, seat, root = sys.argv[1], sys.argv[2], sys.argv[3]
32
+ if not re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9._-]{0,63}", seat) or not os.path.isabs(root):
33
+ sys.exit(1)
34
+ ctx = os.path.join(root, ".tickmarkr", "overseer", "context")
35
+ def pct_of(value):
36
+ ok = isinstance(value, (int, float)) and not isinstance(value, bool) and 0 <= value <= 100
37
+ return value if ok else None
38
+ def field(obj, *keys):
39
+ for key in keys:
40
+ obj = obj.get(key) if isinstance(obj, dict) else None
41
+ return obj
42
+ if mode == "collect":
43
+ try:
44
+ payload = json.load(sys.stdin)
45
+ except Exception:
46
+ payload = None # malformed: still recorded below, as no percentage
47
+ pct = pct_of(field(payload, "context_window", "used_percentage"))
48
+ os.makedirs(ctx, exist_ok=True)
49
+ fd, tmp = tempfile.mkstemp(dir=ctx, prefix="." + seat + ".", suffix=".tmp")
50
+ try:
51
+ with os.fdopen(fd, "w") as f:
52
+ json.dump({"seat": seat, "pct": pct, "ts": time.time()}, f)
53
+ os.replace(tmp, os.path.join(ctx, seat + ".json"))
54
+ except BaseException:
55
+ os.unlink(tmp)
56
+ raise
57
+ else:
58
+ with open(os.path.join(ctx, seat + ".json")) as f:
59
+ record = json.load(f)
60
+ ts, pct = record.get("ts"), pct_of(record.get("pct"))
61
+ age = time.time() - ts if isinstance(ts, (int, float)) and not isinstance(ts, bool) else None
62
+ if record.get("seat") != seat or pct is None or age is None or not -5 <= age <= MAX_AGE_S:
63
+ sys.exit(1)
64
+ print(int(pct) if pct == int(pct) else pct)
65
+ '
66
+
67
+ if [ "${1:-}" = "--read" ]; then
68
+ python3 -c "$PY" read "${2:-}" "${3:-$PWD}" 2>/dev/null || echo unknown
69
+ exit 0
70
+ fi
71
+
72
+ # $(...) strips trailing newlines; the sentinel keeps stdin byte-for-byte
73
+ # ponytail: bash drops NUL bytes, which a JSON statusLine payload never carries
74
+ payload=$(cat; printf .)
75
+ payload=${payload%.}
76
+ if [ -n "${TKR_CONTEXT_SEAT:-}" ] && [ -n "${TKR_CONTEXT_ROOT:-}" ]; then
77
+ printf '%s' "$payload" | python3 -c "$PY" collect "$TKR_CONTEXT_SEAT" "$TKR_CONTEXT_ROOT" >/dev/null 2>&1
78
+ fi
79
+ [ $# -eq 0 ] && exit 0
80
+ # the operator's command runs exactly as Claude ran it (a shell string), so its output and status are its own
81
+ printf '%s' "$payload" | /bin/sh -c "$*"
@@ -7,6 +7,7 @@ set -u
7
7
  run=${1:?run id required}
8
8
  expected=${2:?expected tracked test-file count required}
9
9
  tag=${3:-$run}
10
+ here=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)
10
11
  repo=${TKR_GRADE_CI_REPO:-alzahrani-khalid/tickmarkr}
11
12
  out_dir=${TKR_GRADE_CI_DIR:-${TKR_STATE_DIR:-.tickmarkr}/overseer/diag}
12
13
  mkdir -p "$out_dir" || { echo "UNREADABLE: cannot create log directory $out_dir"; exit 2; }
@@ -14,6 +15,7 @@ mkdir -p "$out_dir" || { echo "UNREADABLE: cannot create log directory $out_dir"
14
15
  verdict=0
15
16
  mark_unreadable() { verdict=2; }
16
17
  mark_red() { [ "$verdict" -eq 0 ] && verdict=1; }
18
+ field() { printf '%s\n' "$classified" | sed -n "s/^VITEST_LOG.* $1=\([^ ]*\).*/\1/p"; }
17
19
 
18
20
  jobs=$(gh run view "$run" --repo "$repo" --json jobs \
19
21
  --jq '.jobs[] | [.databaseId, .name, .status, (.conclusion // "")] | @tsv') \
@@ -46,45 +48,45 @@ while IFS=$'\t' read -r id name status conclusion; do
46
48
 
47
49
  oracle=$(grep -oE 'COUNT_ORACLE [A-Z]+ expected=[0-9A-Z]+ actual=[0-9A-Z]+' "$log" | tail -1)
48
50
  files=$(grep -oE 'Test Files .*' "$log" | sed 's/[[:space:]]*$//')
49
- passed=$(printf '%s\n' "$files" | grep -oE '[0-9]+ passed' | awk '{s+=$1} END{print s+0}')
50
- skipped=$(printf '%s\n' "$files" | grep -oE '[0-9]+ skipped' | awk '{s+=$1} END{print s+0}')
51
- failed=$(printf '%s\n' "$files" | grep -oE '[0-9]+ failed' | awk '{s+=$1} END{print s+0}')
52
- timedout=$(grep -cE 'Test timed out|Error: Hook timed out' "$log" || true)
53
- # Vitest reports errors thrown outside any test (unhandled rejections, worker crashes) in an
54
- # "Unhandled Errors" block that leaves the file summary green and the count oracle satisfied;
55
- # its own tally line names how many. Any unhandled error is RED, never a green suite.
56
- unhandled=$(grep -oE 'Vitest caught [0-9]+ unhandled error' "$log" | grep -oE '[0-9]+' | awk '{s+=$1} END{print s+0}')
57
- if [ "$unhandled" -eq 0 ]; then
58
- unhandled=$(grep -cE 'Unhandled Errors' "$log" || true)
59
- fi
60
- # OBS-1058 (RULING-231-17 precedent, v2.5.5 close R115 add.7): a vitest-worker RPC timeout
61
- # ("Timeout calling \"onTaskUpdate\"") is the starved runner talking, not a test — every green
62
- # release run has carried one. Only a timeout that is ITSELF an unhandled entry counts: the first
63
- # payload line under a per-error header, nothing later. The header is Vitest's own structural line —
64
- # a run of ⎯ on each side of "Unhandled Error", ANSI stripped (both the raw ESC byte and gh's `^[` rendering of it), nothing else on the line — so prose that
65
- # merely contains the words never opens a window — and the payload must be Vitest's whole line
66
- # (`Error: [vitest-worker]: Timeout calling "<method>"`), so an assertion quoting it stays RED. Per-error headers are honoured only inside Vitest's own block —
67
- # after its structural "Unhandled Errors" line and "Vitest caught N" tally, until the file summary — and
68
- # the count never exceeds N, so a test that prints a header and a timeout to stdout forges nothing. A stray diagnostic elsewhere in the log never
69
- # offsets a real unhandled error; anything else under a header stays RED.
70
- esc=$(printf '\033')
71
- rpc=$(awk -v esc="$esc" '{ line = $0; gsub(/\^\[\[[0-9;]*m/, "", line); gsub(esc "\\[[0-9;]*m", "", line); sub(/^[^\t]*\t[^\t]*\t[0-9T:.-]+Z[ ]?/, "", line) }
72
- line ~ /^(⎯)+ Unhandled Errors (⎯)+[ \t]*$/ { opening = 1; next }
73
- opening && line !~ /[^ \t]/ { next }
74
- opening { opening = 0; if (line ~ /^Vitest caught [0-9]+ unhandled error/) { armed = 1; cap = line; sub(/^Vitest caught /, "", cap); sub(/ .*$/, "", cap) } }
75
- armed && line ~ /^[ \t]*Test Files / { armed = 0 }
76
- armed && line ~ /^(⎯)+ Unhandled Error (⎯)+[ \t]*$/ { pending = 1; next }
77
- pending && line ~ /[^ \t]/ { pending = 0; if (line ~ /^Error: \[vitest-worker\]: Timeout calling "[A-Za-z]+"[ \t]*$/) n++ }
78
- END { if (n > cap + 0) n = cap + 0; print n + 0 }' "$log")
79
- if [ "$rpc" -gt 0 ] && [ "$unhandled" -ge "$rpc" ]; then unhandled=$((unhandled - rpc)); fi
51
+ # OBS-1184: the log verdict comes from the ONE classifier the public CI wrapper also uses
52
+ # (scripts/run-ci-vitest.sh), so the badge and this grade cannot disagree. Its rules — complete
53
+ # summaries, failed/timed-out tests, coverage-threshold misses, and the RPC-timeout-only exception
54
+ # to "any unhandled error is RED" (OBS-1058) — live in classify-vitest-log.sh, stated once there.
55
+ classified=$(bash "$here/classify-vitest-log.sh" "$log" 2>&1)
56
+ log_verdict=$(field verdict)
57
+ passed=$(field passed); skipped=$(field skipped); failed=$(field failed); timedout=$(field timedout)
58
+ unhandled=$(field unhandled); rpc=$(field runner_rpc_timeouts); coverage=$(field coverage_misses)
80
59
  errors=$(grep -oE '##\[error\].*' "$log" | sort | uniq -c | sed 's/^ *//' | tr '\n' ';')
60
+ # The commands' own outcomes stand, as in the wrapper: GREEN needs a success conclusion with no step
61
+ # exit annotation, or a failure explained ONLY by steps that exited exactly 1 on a log of their own
62
+ # (gh's step column) that the classifier calls RPC_ONLY — the one exception run-ci-vitest.sh grants
63
+ # (a pre-wrapper run concludes failure on it). Any other code (2, a signal's 137), an unexplained
64
+ # failure, or a cancelled, timed-out or missing conclusion is RED.
65
+ exits=$(grep -E '##\[error\]Process completed with exit code [0-9]+' "$log")
66
+ case "$conclusion" in
67
+ success) outcome=ok ;;
68
+ failure) if [ -n "$exits" ]; then outcome=ok; else outcome=unexplained-failure; fi ;;
69
+ *) outcome="conclusion-${conclusion:-none}" ;;
70
+ esac
71
+ if [ -n "$exits" ]; then
72
+ while IFS= read -r exit_line; do
73
+ code=$(printf '%s\n' "$exit_line" | sed -E 's/.*Process completed with exit code ([0-9]+).*/\1/')
74
+ if [ "$code" != 1 ]; then outcome="exit-$code"; break; fi
75
+ step=$(printf '%s\n' "$exit_line" | awk -F'\t' 'NF >= 3 { print $2 }')
76
+ awk -F'\t' -v s="$step" 'NF >= 3 && $2 == s' "$log" > "$log.step"
77
+ case $(bash "$here/classify-vitest-log.sh" "$log.step" 2>&1) in
78
+ "VITEST_LOG verdict=RPC_ONLY "*) ;;
79
+ *) outcome="exit-1-not-rpc-only"; break ;;
80
+ esac
81
+ done <<< "$exits"
82
+ fi
81
83
 
82
- echo "$name: oracle=[${oracle:-MISSING}] files=[$(printf '%s' "$files" | tr '\n' '|')] passed=$passed skipped=$skipped failed=$failed timedout=$timedout unhandled=$unhandled runner_rpc_timeouts=${rpc:-0} errors=[$errors]"
83
- if [ -z "$oracle" ] || [ -z "$files" ]; then
84
+ echo "$name: oracle=[${oracle:-MISSING}] files=[$(printf '%s' "$files" | tr '\n' '|')] passed=$passed skipped=$skipped failed=$failed timedout=$timedout unhandled=$unhandled runner_rpc_timeouts=$rpc coverage_misses=$coverage log=[$(field reason)] outcome=[$outcome] errors=[$errors]"
85
+ if [ -z "$oracle" ] || [ -z "$files" ] || [ -z "$log_verdict" ]; then
84
86
  echo "$name: UNREADABLE"
85
87
  mark_unreadable
86
88
  elif [ "$oracle" = "COUNT_ORACLE GREEN expected=$expected actual=$expected" ] \
87
- && [ "$failed" -eq 0 ] && [ "$timedout" -eq 0 ] && [ "$unhandled" -eq 0 ] \
89
+ && { [ "$log_verdict" = CLEAN ] || [ "$log_verdict" = RPC_ONLY ]; } && [ "$outcome" = ok ] \
88
90
  && [ $((passed + skipped)) -eq "$expected" ]; then
89
91
  echo "$name: GREEN"
90
92
  else
@@ -72,7 +72,8 @@ field() { printf '%s' "$2" | sed -n "s/.*\"$1\":\"\([^\"]*\)\".*/\1/p" | head -1
72
72
  bucket() { printf '%s' "$2" | sed -n "s/.*\"$1\":\[\([^]]*\)\].*/\1/p" | head -1; }
73
73
 
74
74
  report() {
75
- local line="$1" run="$2" ev
75
+ # $3 is the row's physical journal line: with its ts it is the token a decision binds to (OBS-1178)
76
+ local line="$1" run="$2" at="${3:-}" ev
76
77
  ev=$(field event "$line")
77
78
  case "$ev" in
78
79
  run-end)
@@ -98,12 +99,13 @@ report() {
98
99
  task-human)
99
100
  echo "TASK_HUMAN $(field taskId "$line") — $run"
100
101
  echo " $(printf '%s' "$line" | sed -n 's/.*"reason":"\([^"]\{0,160\}\).*/\1/p')"
101
- echo " a park waits for a DECISION; read the gate evidence, then \`tickmarkr approve $run $(field taskId "$line")\` or re-scope"
102
+ echo " a park waits for a DECISION; read the gate evidence, then \`tickmarkr approve $run $(field taskId "$line") --park $at@$(field ts "$line")\` (bound to THIS park; refused once a newer one opens) or re-scope"
102
103
  ;;
103
104
  task-failed)
104
105
  echo "TASK_FAILED $(field taskId "$line") — $run"
105
106
  echo " $(printf '%s' "$line" | sed -n 's/.*"error":"\([^"]\{0,160\}\).*/\1/p')"
106
107
  echo " the run may continue on independent tasks; this task did not deliver"
108
+ echo " landed commits re-gate with \`tickmarkr approve $run $(field taskId "$line") --recheck --park $at@$(field ts "$line")\` (its bound failure token)"
107
109
  ;;
108
110
  consult-verdict)
109
111
  echo "CONSULT_VERDICT $(field taskId "$line") action=$(field action "$line") — $run"
@@ -127,9 +129,9 @@ while [ "$elapsed" -lt "$CAP" ]; do
127
129
  # A new run resets the baseline — its whole journal is unseen by definition.
128
130
  if [ "$J" != "$seen_run" ]; then seen_run="$J"; base=0; fi
129
131
 
130
- hit=$(since_arm "$J" | grep -E "$PAT" | head -1)
132
+ hit=$(since_arm "$J" | grep -n -E "$PAT" | head -1)
131
133
  if [ -n "$hit" ]; then
132
- report "$hit" "$(basename "$(dirname "$J")")"
134
+ report "${hit#*:}" "$(basename "$(dirname "$J")")" "$((base + ${hit%%:*}))"
133
135
  exit 0
134
136
  fi
135
137
  done