tickmarkr 1.93.0 → 1.97.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/model-lints.d.ts +1 -1
- package/dist/adapters/model-lints.js +116 -52
- package/dist/brand.d.ts +3 -0
- package/dist/brand.js +3 -1
- package/dist/cli/commands/beat.d.ts +1 -0
- package/dist/cli/commands/beat.js +50 -0
- package/dist/cli/commands/status.js +325 -127
- package/dist/cli/commands/verify.js +52 -26
- package/dist/cli/index.d.ts +1 -1
- package/dist/cli/index.js +3 -1
- package/dist/compile/gsd.js +34 -1
- package/dist/compile/native.js +26 -1
- package/dist/drivers/herdr.d.ts +26 -1
- package/dist/drivers/herdr.js +77 -29
- package/dist/gates/acceptance.js +26 -4
- package/dist/gates/baseline.d.ts +10 -1
- package/dist/gates/baseline.js +35 -6
- package/dist/run/daemon.d.ts +5 -0
- package/dist/run/daemon.js +25 -4
- package/dist/run/merge.d.ts +6 -1
- package/dist/run/merge.js +46 -11
- package/fixtures/gsd-sample/07-live-check/07-01-PLAN.md +1 -1
- package/fixtures/gsd-sample/PROJECT.md +19 -0
- package/fixtures/payload-shape/p99-shaped.spec.md +68 -0
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +42 -18
- package/skills/tickmarkr-overseer/scripts/seat-send.sh +98 -5
package/dist/run/daemon.js
CHANGED
|
@@ -117,6 +117,11 @@ export function formatSummary(s) {
|
|
|
117
117
|
: "";
|
|
118
118
|
return `done: ${s.done.length}, failed: ${s.failed.length}, human: ${s.human.length}, blocked: ${s.blocked.length}, pending: ${s.pending.length}\nintegration branch: ${s.branch}${tip}${outstanding}`;
|
|
119
119
|
}
|
|
120
|
+
/** The narrator's command, bound to THIS run. `status` takes exactly one positional and it is the
|
|
121
|
+
* run id (cli/commands/status.ts positionalRunId), so naming it here is what stops the board from
|
|
122
|
+
* following the newest journal in a repo that already carries a second, newer run — a board showing
|
|
123
|
+
* the wrong run is a recorded incident (skills/tickmarkr-overseer/SKILL.md). */
|
|
124
|
+
export const watchCommand = (runId) => `tickmarkr status --watch ${runId}`;
|
|
120
125
|
const MAX_ATTEMPTS = 10; // ponytail: hard cap so a pathological ladder can never loop forever
|
|
121
126
|
// v1.85 T3 (retry economics): two repairs per engagement, then the fresh ladder. A repair re-uses the
|
|
122
127
|
// findings and the landed diff instead of re-buying onboarding; when two of them have not closed the
|
|
@@ -1157,13 +1162,14 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1157
1162
|
// T6: open the narrator AFTER run-start/run-resume is journaled so the watch surface has a run to
|
|
1158
1163
|
// show. driver.narrator is undefined on subprocess → no-op (subprocess spawns nothing). Swallowed:
|
|
1159
1164
|
// a failed-to-open or later-dead watch pane never affects the run.
|
|
1160
|
-
// OBS-103: hold the returned slot — narrator()
|
|
1161
|
-
//
|
|
1162
|
-
//
|
|
1165
|
+
// OBS-103: hold the returned slot — narrator() returns the run's board under its canonical owned
|
|
1166
|
+
// name whichever way it got there (the herdr driver retires a survivor it finds and re-splits, so
|
|
1167
|
+
// the live command is run-bound), and the run-end sweep below retires it by that name regardless
|
|
1168
|
+
// of which daemon instance split the pane.
|
|
1163
1169
|
const watchName = formatOwnedName({ role: "watch", taskId: "run", attempt: 0, runId });
|
|
1164
1170
|
let watchSlot;
|
|
1165
1171
|
try {
|
|
1166
|
-
watchSlot = await driver.narrator?.(repoRoot,
|
|
1172
|
+
watchSlot = await driver.narrator?.(repoRoot, watchCommand(runId), runId);
|
|
1167
1173
|
}
|
|
1168
1174
|
catch {
|
|
1169
1175
|
/* cosmetic-only — the run proceeds without a live surface */
|
|
@@ -1350,6 +1356,12 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1350
1356
|
// be able to tell "the suite found a defect" from "the runner never ran", and the merge
|
|
1351
1357
|
// predicate's reason for refusing has to be legible in the ledger it refused from.
|
|
1352
1358
|
...(g.meta?.infra === true ? { infra: true } : {}),
|
|
1359
|
+
// OBS-540: preserve terminal-vs-retryable infra exactly. normalizeGateOutcome deliberately
|
|
1360
|
+
// defaults a legacy infra row to retryable, so dropping an explicit false here reverses the
|
|
1361
|
+
// oracle's recorded verdict when any journal-backed reader reconstructs it.
|
|
1362
|
+
...(g.meta?.infra === true && typeof g.meta.retryable === "boolean"
|
|
1363
|
+
? { retryable: g.meta.retryable }
|
|
1364
|
+
: {}),
|
|
1353
1365
|
// R3 (OBS-186): a declined review is journal truth, not an absence. `skipped: true` alone
|
|
1354
1366
|
// says a gate did not run; these say WHICH policy declined it and WHY, so a reader of the
|
|
1355
1367
|
// ledger never has to infer participation from a details string. The green-skip branch that
|
|
@@ -2941,6 +2953,15 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2941
2953
|
}
|
|
2942
2954
|
break gateLoop;
|
|
2943
2955
|
}
|
|
2956
|
+
// OBS-540: infra is a fail-closed NON-verdict, not quality degradation. Park with the blocker
|
|
2957
|
+
// already journaled by onGate, before gateFails and before any ladder selection can fund an
|
|
2958
|
+
// identical retry in the same environment. Parsed judge refusals never carry infra and keep
|
|
2959
|
+
// flowing through the chargeable quality path below.
|
|
2960
|
+
const infraFailure = results.find((g) => gateFailed(g) && g.meta?.infra === true);
|
|
2961
|
+
if (infraFailure) {
|
|
2962
|
+
await park(t, `${infraFailure.gate}: ${infraFailure.details}`, "infra", assignment, attempt + 1, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
2963
|
+
return;
|
|
2964
|
+
}
|
|
2944
2965
|
gateFails++; // this attempt's gates failed — the one place quality degradation is verified (never inferred from attempts)
|
|
2945
2966
|
// v1.53 T3: prefer the CLI's own session id captured from this attempt's output (kimi's resume
|
|
2946
2967
|
// trailer) over the harness slot name; absent hook or no capture keeps today's slot-name id.
|
package/dist/run/merge.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { TickmarkrConfig } from "../config/config.js";
|
|
2
|
-
import { type Baseline } from "../gates/baseline.js";
|
|
2
|
+
import { type Baseline, type FailureClassification } from "../gates/baseline.js";
|
|
3
3
|
export interface TipVerifyResult {
|
|
4
4
|
gate: string;
|
|
5
5
|
cmd: string;
|
|
@@ -10,6 +10,11 @@ export interface TipVerifyResult {
|
|
|
10
10
|
artifact?: string;
|
|
11
11
|
/** Q121s: nonzero exit whose failures are ALL baseline-recorded — forgiven exactly as the battery forgives. */
|
|
12
12
|
forgiven?: boolean;
|
|
13
|
+
/**
|
|
14
|
+
* OBS-534: what a nonzero exit is evidence OF, taken from the battery's own readers — `ceilingKillResult`
|
|
15
|
+
* for a kill, `classifyFailureOutput` for everything else. `infra` means nothing was verified.
|
|
16
|
+
*/
|
|
17
|
+
cause?: FailureClassification;
|
|
13
18
|
}
|
|
14
19
|
export declare function integrationBranch(cfg: TickmarkrConfig, runId: string): string;
|
|
15
20
|
export declare function ensureIntegration(repo: string, branch: string, baseRef: string): Promise<string>;
|
package/dist/run/merge.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { existsSync, writeFileSync } from "node:fs";
|
|
2
2
|
import { join } from "node:path";
|
|
3
3
|
import { shq } from "../adapters/types.js";
|
|
4
|
-
import { classifyFailureOutput, fingerprint, freshFailures } from "../gates/baseline.js";
|
|
4
|
+
import { ceilingKillResult, classifyFailureOutput, effectiveCeilingMs, fingerprint, freshFailures, } from "../gates/baseline.js";
|
|
5
5
|
import { tickmarkrDir } from "../graph/graph.js";
|
|
6
6
|
import { gitHead, linkNodeModules, resolveIntegrationBranch, sh, shGit, shGitOk, WORKTREES_DIR } from "./git.js";
|
|
7
7
|
export function integrationBranch(cfg, runId) {
|
|
@@ -53,21 +53,55 @@ export async function mergeTask(intWt, taskBranch, message, gatedCommit) {
|
|
|
53
53
|
export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
54
54
|
const results = [];
|
|
55
55
|
for (const [gate, cmd] of Object.entries(commands)) {
|
|
56
|
-
const
|
|
56
|
+
const entry = baseline?.commands[gate];
|
|
57
|
+
// OBS-534: the ceiling is the BATTERY's, derived by effectiveCeilingMs from the same baseline entry
|
|
58
|
+
// this loop already reads for forgiveness two lines down — never the flat DEFAULT_SHELL_TIMEOUT_MS
|
|
59
|
+
// `sh` defaults to. A suite whose capture measured 600007ms carries a recorded 1800021ms ceiling;
|
|
60
|
+
// running it under 600000ms here SIGKILLed a green tip three times while every per-task gate passed.
|
|
61
|
+
const ceilingMs = effectiveCeilingMs(entry);
|
|
62
|
+
const r = await sh(cmd, intWt, ceilingMs);
|
|
57
63
|
const raw = r.stdout + "\n" + r.stderr;
|
|
58
64
|
const stripped = raw.split(intWt).join("");
|
|
59
|
-
const
|
|
65
|
+
const artifact = join(runDir, `tip-verify-${gate}.log`);
|
|
66
|
+
// Battery parity on the ceiling too (baseline.ts Q24): the kill is read BEFORE the exit code is
|
|
67
|
+
// interpreted at all. A SIGKILLed battery never returned a verdict, so no line of its partial
|
|
68
|
+
// output is one — fingerprinting it is what produced the `<unrecognized failure output>` an
|
|
69
|
+
// operator cannot act on. The kill reader's own text (ceiling + elapsed) and cause replace it.
|
|
70
|
+
const killed = ceilingKillResult(gate, r, ceilingMs);
|
|
71
|
+
if (killed) {
|
|
72
|
+
writeFileSync(artifact, raw);
|
|
73
|
+
results.push({
|
|
74
|
+
gate,
|
|
75
|
+
cmd,
|
|
76
|
+
pass: false,
|
|
77
|
+
exitCode: r.code,
|
|
78
|
+
fingerprints: [],
|
|
79
|
+
details: killed.details,
|
|
80
|
+
cause: killed.meta?.classification,
|
|
81
|
+
artifact,
|
|
82
|
+
});
|
|
83
|
+
continue;
|
|
84
|
+
}
|
|
60
85
|
const { failing, unreadable } = freshFailures(entry, stripped);
|
|
86
|
+
const cause = r.code === 0 ? undefined : classifyFailureOutput(stripped);
|
|
61
87
|
// `?? 1` is the battery's own default (baseline.ts compareToBaseline): an exitCode-less legacy
|
|
62
88
|
// entry reads as red-at-baseline there, so it must read the same here or old baselines silently
|
|
63
|
-
// lose forgiveness.
|
|
64
|
-
//
|
|
65
|
-
//
|
|
66
|
-
|
|
67
|
-
|
|
89
|
+
// lose forgiveness. OBS-534 (T2): a capture killed at its ceiling now records a CAUSE and no
|
|
90
|
+
// verdict, and that default must not launder the missing exit code back into red-at-baseline.
|
|
91
|
+
// Only a recorded verdict is forgivable, so this reads the same predicate the battery does
|
|
92
|
+
// (baseline.ts `baselineRed`) — `infra` first, the legacy default only after it. freshFailures
|
|
93
|
+
// already drops the killed capture's flushed fingerprints, but it cannot close this alone: an
|
|
94
|
+
// output whose only shape is a diagnostic HEADING (vitest's "Unhandled Errors" banner) is
|
|
95
|
+
// fingerprintable yet OBS-42-exempt from rejecting, so `failing` comes back empty, `unreadable`
|
|
96
|
+
// false and the cause reads "regression" — every other guard satisfied, and a real red forgiven
|
|
97
|
+
// against a capture that never finished asking the question.
|
|
98
|
+
const baselineRed = entry !== undefined && entry.infra !== true && (entry.exitCode ?? 1) !== 0;
|
|
99
|
+
// Battery parity on the infra rule too (T9): infrastructure-only output means the runner never
|
|
100
|
+
// completed a suite — nothing was verified, so nothing is forgivable, however familiar its
|
|
101
|
+
// fingerprints. Stricter-than-battery edge kept: unreadable output never forgives.
|
|
102
|
+
const forgiven = r.code !== 0 && baselineRed && failing.length === 0 && !unreadable && cause !== "infra";
|
|
68
103
|
const pass = r.code === 0 || forgiven;
|
|
69
|
-
|
|
70
|
-
if (artifact)
|
|
104
|
+
if (!pass)
|
|
71
105
|
writeFileSync(artifact, raw);
|
|
72
106
|
results.push({
|
|
73
107
|
gate,
|
|
@@ -79,7 +113,8 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
79
113
|
: forgiven ? `exit ${r.code} but only baseline-recorded failures (forgiven vs baseline)`
|
|
80
114
|
: `exit ${r.code}`,
|
|
81
115
|
...(forgiven ? { forgiven: true } : {}),
|
|
82
|
-
...(
|
|
116
|
+
...(cause ? { cause } : {}),
|
|
117
|
+
...(pass ? {} : { artifact }),
|
|
83
118
|
});
|
|
84
119
|
}
|
|
85
120
|
return results;
|
|
@@ -20,7 +20,7 @@ Implement the first objective sentence. Additional prose that is not the title.
|
|
|
20
20
|
</objective>
|
|
21
21
|
|
|
22
22
|
<context>
|
|
23
|
-
|
|
23
|
+
@fixtures/gsd-sample/PROJECT.md
|
|
24
24
|
@$HOME/.claude/get-shit-done/workflows/execute-plan.md
|
|
25
25
|
@~/somewhere/outside.md
|
|
26
26
|
</context>
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
# Sample project — vendored GSD fixture input
|
|
2
|
+
|
|
3
|
+
This file exists so the sibling phase's `07-01-PLAN.md` can cite a repo-relative `<context>` path that
|
|
4
|
+
is **part of the fixture itself**. It previously cited `.planning/PROJECT.md`, which resolved only
|
|
5
|
+
inside the private development checkout: the public export strips `.planning/` at any depth, so the
|
|
6
|
+
compiler's context-reachability refusal (v1.96 T3) correctly failed eight tests on the exported tree
|
|
7
|
+
and the release ritual's pre-tag proof caught it. A vendored fixture must carry its own inputs.
|
|
8
|
+
|
|
9
|
+
Nothing here is read by an assertion — the fixture only needs the path to resolve. The prose stands in
|
|
10
|
+
for the project brief a real GSD plan would point a worker at.
|
|
11
|
+
|
|
12
|
+
## Objective
|
|
13
|
+
|
|
14
|
+
Ship a small feature end to end, with each plan in the phase owning one objective sentence.
|
|
15
|
+
|
|
16
|
+
## Constraints
|
|
17
|
+
|
|
18
|
+
- One plan, one objective.
|
|
19
|
+
- A plan's `<context>` block promises the worker can read every repo-relative path it lists.
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
<!-- tickmarkr:spec -->
|
|
2
|
+
<!-- provenance: the shape of .planning/phases/99-arabic-coverage as compiled on 2026-08-18 into
|
|
3
|
+
run-20260818-185710-0000000000000011 (41 tasks). OBS-535's fix — files[] is write scope, not
|
|
4
|
+
payload — was measured against that graph (41 unreadable-payload lints to 0; 31 overflow lints
|
|
5
|
+
when the window is forced to 50k) and against nothing in this repo, because no fixture carried
|
|
6
|
+
its shape. This one does, at five tasks instead of forty-one.
|
|
7
|
+
|
|
8
|
+
The four shape features that made every task on that graph skip its context-window comparison,
|
|
9
|
+
each present below:
|
|
10
|
+
1. a brace-glob write scope (`scripts/{a,b}`) — a set, not a document; unmeasurable as one file
|
|
11
|
+
2. an output declared in files[] (`*-SUMMARY.md`) — absent from the base tree BY CONSTRUCTION
|
|
12
|
+
3. a wave-1 artifact consumed by a TRANSITIVE dependent (T3 reads what T1 writes, via T2)
|
|
13
|
+
4. the same artifact cited by a task with NO producer upstream (T4) — the actionable class,
|
|
14
|
+
which on the real graph was 17 tasks citing a self-gitignored tree (RULING-P99-14)
|
|
15
|
+
T5 adds the class no commit can ever satisfy: a ref under a gitignored directory.
|
|
16
|
+
|
|
17
|
+
Compile this fixture from a NON-REPO directory. compileNative's context reachability check
|
|
18
|
+
(native.ts:684) fails open when git cannot answer, and T3/T4/T5 deliberately cite paths absent
|
|
19
|
+
from any base tree — that absence is the fixture's whole subject. -->
|
|
20
|
+
|
|
21
|
+
## T1: Land the instruments the later waves measure with
|
|
22
|
+
- goal: Write the two audit instruments and this task's own summary, so a later wave has something to read
|
|
23
|
+
- shape: implement
|
|
24
|
+
- files: scripts/{audit-strict.mjs,census.mjs}, .planning/payload-shape/T1-SUMMARY.md
|
|
25
|
+
- context: docs/payload-shape/PLAN.md
|
|
26
|
+
- complexity: 3
|
|
27
|
+
- acceptance:
|
|
28
|
+
- command: node -e "process.exit(0)"
|
|
29
|
+
- judge: both instruments exist and the summary records what they measured
|
|
30
|
+
|
|
31
|
+
## T2: Rewrite the localisation sources the instruments flag
|
|
32
|
+
- goal: Apply the instrument's findings across the localisation sources named in the write scope
|
|
33
|
+
- shape: implement
|
|
34
|
+
- deps: T1
|
|
35
|
+
- files: src/i18n/{ar/common.json,en/common.json,ar/dossier.json,en/dossier.json,ar/intake.json,en/intake.json,ar/tasks.json,en/tasks.json,ar/settings.json,en/settings.json}, .planning/payload-shape/T2-SUMMARY.md
|
|
36
|
+
- context: docs/payload-shape/PLAN.md, scripts/audit-strict.mjs
|
|
37
|
+
- complexity: 5
|
|
38
|
+
- acceptance:
|
|
39
|
+
- judge: every key src/i18n/ar/common.json shares with its en counterpart carries a non-empty Arabic value, and .planning/payload-shape/T2-SUMMARY.md records the count it rewrote
|
|
40
|
+
|
|
41
|
+
## T3: Close the census the second instrument opens
|
|
42
|
+
- goal: Read the census instrument written two waves back and close its remaining senses
|
|
43
|
+
- shape: implement
|
|
44
|
+
- deps: T2
|
|
45
|
+
- files: src/i18n/{ar/common.json,en/common.json}, .planning/payload-shape/T3-SUMMARY.md
|
|
46
|
+
- context: docs/payload-shape/PLAN.md, scripts/census.mjs
|
|
47
|
+
- complexity: 4
|
|
48
|
+
- acceptance:
|
|
49
|
+
- judge: the census reports no unresolved sense for any rewritten key
|
|
50
|
+
|
|
51
|
+
## T4: Report on the audit without depending on the task that writes it
|
|
52
|
+
- goal: Summarise the audit's findings for the operator record
|
|
53
|
+
- shape: chore
|
|
54
|
+
- files: .planning/payload-shape/T4-SUMMARY.md
|
|
55
|
+
- context: docs/payload-shape/PLAN.md, scripts/audit-strict.mjs
|
|
56
|
+
- complexity: 2
|
|
57
|
+
- acceptance:
|
|
58
|
+
- judge: .planning/payload-shape/T4-SUMMARY.md states both the flagged-site count and the total key count it was measured against
|
|
59
|
+
|
|
60
|
+
## T5: Apply the standing ruling to the rewritten sources
|
|
61
|
+
- goal: Enforce the ruling's terminology decisions across the sources T2 rewrote
|
|
62
|
+
- shape: chore
|
|
63
|
+
- deps: T2, T3
|
|
64
|
+
- files: src/i18n/{ar/common.json,en/common.json}
|
|
65
|
+
- context: docs/payload-shape/PLAN.md, .state/RULING-TERMINOLOGY.md
|
|
66
|
+
- complexity: 2
|
|
67
|
+
- acceptance:
|
|
68
|
+
- judge: no file under src/i18n/ar contains the string "Dossier" or "Engagement" in Latin script
|
package/package.json
CHANGED
|
@@ -63,15 +63,14 @@ through brief lineage. **An executor choice nobody made is still an executor cho
|
|
|
63
63
|
tab OVERSEER; create ONE tab ORCHESTRATOR.
|
|
64
64
|
**FIVE-TAB CANON (standing operator layout — corrected three times on 2026-07-27, layout approved
|
|
65
65
|
2026-07-29, re-earned 2026-08-17):**
|
|
66
|
-
- `OVERSEER` — you
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
2026-08-17: a planning seat split beside the orchestrator, and the operator caught it, again.
|
|
66
|
+
- `OVERSEER` — you. Do not add a second live run surface: the daemon self-places the shipped board
|
|
67
|
+
beside the supervising seat that invokes the run.
|
|
68
|
+
- `ORCH` — the orchestrator and the daemon-placed, run-id-pinned shipped board beside it. **Look for
|
|
69
|
+
that `role: "watch"` pane; never hand-place or hand-roll a live run surface. Nothing else, ever: a
|
|
70
|
+
work seat NEVER splits into the ORCH tab.** Operator verbatim: *"in orch tab should be the orch and
|
|
71
|
+
the watcher only."* Re-earned 2026-08-17: a planning seat split beside the orchestrator, and the
|
|
72
|
+
operator caught it, again. The daemon owns the side placement and board-first width allocation;
|
|
73
|
+
neither the worker-pane halving floor nor an overseer split command places this pane.
|
|
75
74
|
- Worker/seat tabs — tickmarkr opens ONE TAB PER TASK itself; GSD-leg seats get the same treatment
|
|
76
75
|
(own tab, or a shared WORKERS tab), never the ORCH tab.
|
|
77
76
|
- `CONSULT · <topic>` — ONE shared tab for ALL consultants of a round, side-by-side splits; never one
|
|
@@ -102,7 +101,7 @@ through brief lineage. **An executor choice nobody made is still an executor cho
|
|
|
102
101
|
and only then delete. A guard's home must outlive the mission that earned it. The project ledger does
|
|
103
102
|
NOT count as that home — `CLAUDE.md` itself says planning records are read-only archives and current
|
|
104
103
|
guidance belongs in the memory file or the shipped docs.
|
|
105
|
-
4. Arm the watcher (Supervision). Report the hierarchy map (pane ids + names) to the user.
|
|
104
|
+
4. Arm the watcher and your own supervision beat (Supervision). Report the hierarchy map (pane ids + names) to the user.
|
|
106
105
|
|
|
107
106
|
## Supervising tickmarkr as the executor — WHO DOES WHAT
|
|
108
107
|
|
|
@@ -114,7 +113,7 @@ and the first thing to get right is that **almost none of it is yours**.
|
|
|
114
113
|
| | ORCHESTRATOR | OVERSEER |
|
|
115
114
|
|---|---|---|
|
|
116
115
|
| `compile` · `plan` · `run` · `resume` | **owns** | never |
|
|
117
|
-
| journal watchers
|
|
116
|
+
| journal and dialog watchers; verifying the daemon's live surface | **owns** | watches the ORCHESTRATOR, not the run |
|
|
118
117
|
| orphan sweeps, worker pane hygiene | **owns** | — |
|
|
119
118
|
| reading a gate failure and assembling its evidence | **owns** | reads the file it writes |
|
|
120
119
|
| **deciding** a gate, spend, or ship | never | **owns** |
|
|
@@ -134,10 +133,11 @@ journal tail to decide what happens next, or sweeping orphans — you have taken
|
|
|
134
133
|
|
|
135
134
|
### What the ORCHESTRATOR does, and what you require of it
|
|
136
135
|
|
|
137
|
-
- **
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
136
|
+
- **The live surface arrives with the run.** `tickmarkr run` is stdout-silent until run-end by design;
|
|
137
|
+
its daemon self-places one shipped `role: "watch"` board beside the supervising seat and pins that
|
|
138
|
+
board to the daemon's run id. **Look for the matching daemon-placed pane.** If it is absent, treat that
|
|
139
|
+
as a daemon/run liveness fault and use the normal recovery path; never hand-place, hand-roll, or launch
|
|
140
|
+
a replacement live surface.
|
|
141
141
|
- **The journal is the source of truth**, not panes. Watchers go on `run-end` / `task-human` /
|
|
142
142
|
`task-failed` / `consult-verdict`; never sleep-poll inside an agent turn. **Never key a watcher on an
|
|
143
143
|
agent's `done`** — that is turn end and fires the moment a seat finishes acknowledging you.
|
|
@@ -377,6 +377,26 @@ they are left implicit:
|
|
|
377
377
|
|
|
378
378
|
## Supervision watcher
|
|
379
379
|
|
|
380
|
+
**Arm your OWN tier first, in the same call chain that arms everything else.** `status` derives each
|
|
381
|
+
tier's state from a beat file the tier itself writes, so a seat that never beats reads `ABSENT` — and
|
|
382
|
+
`ABSENT` means *never armed*, which is a lie about a seat that is working the run. Measured on the P99
|
|
383
|
+
run: `orchestrator ARMED / overseer ABSENT / watch ABSENT` for the whole milestone, with a live overseer
|
|
384
|
+
watching it. Two thirds of that line were constants, not measurements.
|
|
385
|
+
|
|
386
|
+
The beat is one shipped command and the loop is yours, run from the repo root as its own
|
|
387
|
+
`run_in_background` Bash call:
|
|
388
|
+
|
|
389
|
+
```bash
|
|
390
|
+
cd <repo> && while :; do tickmarkr beat overseer; sleep 10; done # 10s = SUPERVISION_BEAT_MS
|
|
391
|
+
tickmarkr beat overseer --stand-down # at stand-down, in the same act
|
|
392
|
+
```
|
|
393
|
+
|
|
394
|
+
One beat per invocation, deliberately: the loop is what proves the seat is alive, so a command that
|
|
395
|
+
kept beating on its own would keep reporting a dead seat as healthy. Stop the loop — or die — and the
|
|
396
|
+
tier ages to `STALE` (never `ABSENT`) within six beats, which is the state that says *armed, then lost*.
|
|
397
|
+
Stand down explicitly when you hand off, or a deliberate exit reads as a death. Same rule as rule 29
|
|
398
|
+
below, now with a conventional path the other tier already reads: `tickmarkr status` shows it.
|
|
399
|
+
|
|
380
400
|
Arm the bundled watcher as its OWN Bash call with `run_in_background` — chaining it after other commands
|
|
381
401
|
with `&` orphans it from the wake chain. It prints one wake reason and exits; re-arm after every wake.
|
|
382
402
|
|
|
@@ -448,9 +468,9 @@ herdr agent wait <name> --until blocked --timeout <ms> # run_in_background
|
|
|
448
468
|
⚠ **Its exit status is not evidence.** That command exits **0 on timeout** and **0 when the pane is gone**,
|
|
449
469
|
exactly as it does on a real block — so confirm every wake by READING the pane before acting on it.
|
|
450
470
|
|
|
451
|
-
**And note what no watcher can cover:** the
|
|
452
|
-
|
|
453
|
-
|
|
471
|
+
**And note what no watcher can cover:** the supervision beat above proves this seat is ARMED and nothing
|
|
472
|
+
more — it is a liveness claim, not a wake signal, so the seat still improvises the wakes, and an improvised
|
|
473
|
+
set is where a whole failure class hides. A **host** permission
|
|
454
474
|
modal is invisible to tickmarkr entirely, so no `src/**` change closes that one — it is covered here or
|
|
455
475
|
nowhere.
|
|
456
476
|
|
|
@@ -531,6 +551,10 @@ orchestrator turn boundary.
|
|
|
531
551
|
safe 53 → floor 108"*), and it splits right only while `paneWidth/2 ≥ 108 + 2` (`herdr.ts:494`),
|
|
532
552
|
otherwise **down**. Apply the same test by hand: `herdr pane layout --pane <id>`, halve the width,
|
|
533
553
|
and if the halves fall under the floor, split `--direction down`.
|
|
554
|
+
- **The daemon-placed ORCH board is outside this manual split rule.** The halving bound protects
|
|
555
|
+
worker-pane trailers; the daemon, not the overseer, places the run-id-pinned shipped board beside
|
|
556
|
+
the supervising seat and owns its board-first width allocation. Look for that pane and do not split,
|
|
557
|
+
place, or recreate it.
|
|
534
558
|
- **Binary splits cannot produce an even 3-column row at any width.** 220 goes to 110/55/55 whichever
|
|
535
559
|
pane you split. **At a 220-col terminal the width-derived cap is TWO side-by-side panes**; a third
|
|
536
560
|
seat goes below one of them, or into its own tab. "Three panes" is a *height* heuristic
|
|
@@ -25,7 +25,10 @@
|
|
|
25
25
|
#
|
|
26
26
|
# First output line is the machine-greppable verdict:
|
|
27
27
|
# DELIVERED_SUBMITTED | QUEUED_BEHIND_TURN | SEND_UNSUBMITTED | INTERRUPT_FAILED | TARGET_GONE
|
|
28
|
-
# | REFUSED_SIZE | REFUSED_BOX_OCCUPIED
|
|
28
|
+
# | REFUSED_SIZE | REFUSED_BOX_OCCUPIED | SEND_UNVERIFIED | SEND_REFUSED | SEND_FAILED
|
|
29
|
+
#
|
|
30
|
+
# A success verdict names the surface that delivered when it was not the default one, so a fallback
|
|
31
|
+
# is visible in the transcript rather than inferred (OBS-552).
|
|
29
32
|
#
|
|
30
33
|
# This verifies SUBMISSION — the text left the input box — never that the seat understood or acted.
|
|
31
34
|
# Read the seat's ARTIFACT for that; a send receipt is not a response.
|
|
@@ -46,6 +49,14 @@ status_of() {
|
|
|
46
49
|
| sed -n 's/.*"agent_status":"\([a-z_]*\)".*/\1/p' | head -1
|
|
47
50
|
}
|
|
48
51
|
|
|
52
|
+
# The submit-fallback needs a PANE id, and $TARGET may be an agent NAME. `agent get` answers for a
|
|
53
|
+
# detected-but-undrivable seat — it is only `agent prompt` that refuses one (OBS-552) — so this
|
|
54
|
+
# resolves either target form. The first `pane_id` in the response is the agent's own.
|
|
55
|
+
pane_of() {
|
|
56
|
+
herdr agent get "$TARGET" 2>/dev/null \
|
|
57
|
+
| sed -n 's/.*"pane_id":"\([^"]*\)".*/\1/p' | head -1
|
|
58
|
+
}
|
|
59
|
+
|
|
49
60
|
# The rendered `❯` line — the only state that discriminates submitted from sitting. head -1 returns
|
|
50
61
|
# the first rendered line, i.e. a PREFIX of a wrapped draft, so callers compare prefixes, never
|
|
51
62
|
# whole sentences (OBS-396: a full-sentence match returns false on a message that arrived intact).
|
|
@@ -54,6 +65,24 @@ prompt_line() {
|
|
|
54
65
|
| sed -n 's/^[[:space:]]*❯[[:space:]]*//p' | head -1 | sed 's/[[:space:]]*$//'
|
|
55
66
|
}
|
|
56
67
|
|
|
68
|
+
# STAGED-QUEUE DISCRIMINATOR (OBS-552 addendum, captured 2026-08-19 on a live kimi seat). After its
|
|
69
|
+
# first turn ends, kimi stages every later message into a SECOND `❯` region —
|
|
70
|
+
# `❯ <text> ↑ to edit · ctrl-s to steer immediately` — and EMPTIES the ordinary box when it does. So
|
|
71
|
+
# `prompt_line`'s emptiness is not submission on a kimi seat: it is the queue swallowing the
|
|
72
|
+
# directive. Six consecutive paths were tried against that state and the pane revision never moved
|
|
73
|
+
# off 3; `seat-send.sh` called the first one DELIVERED_SUBMITTED. This reads the staging AFFORDANCE
|
|
74
|
+
# rather than the vendor name, so any TUI that grows the same queue is covered without a matcher list.
|
|
75
|
+
# Returns the staged TEXT only: the `❯` marker and the trailing affordance are chrome, and leaving the
|
|
76
|
+
# affordance on would break `is_ours`, whose contract is that the rendered text is a PREFIX of $MSG
|
|
77
|
+
# (OBS-396). A wrapped staged entry still yields a prefix, which is all the comparison needs.
|
|
78
|
+
staged_line() {
|
|
79
|
+
herdr agent read "$TARGET" --source visible --lines 14 2>/dev/null \
|
|
80
|
+
| sed -n '/↑ to edit/{/steer/p;}' | head -1 \
|
|
81
|
+
| sed 's/^[[:space:]]*❯[[:space:]]*//' \
|
|
82
|
+
| sed 's/[[:space:]]*↑ to edit.*$//' \
|
|
83
|
+
| sed 's/[[:space:]]*$//'
|
|
84
|
+
}
|
|
85
|
+
|
|
57
86
|
# Is $1 a rendered prefix of our message? (first line of a wrapped draft = prefix of MSG)
|
|
58
87
|
is_ours() {
|
|
59
88
|
case "$MSG" in "$1"*) return 0 ;; *) return 1 ;; esac
|
|
@@ -107,9 +136,73 @@ fi
|
|
|
107
136
|
|
|
108
137
|
# Atomic text+Enter honouring the pane's live bracketed-paste mode — never send-text + separate
|
|
109
138
|
# Enter, which is the "sitting unsubmitted" failure by construction.
|
|
110
|
-
|
|
139
|
+
#
|
|
140
|
+
# OBS-552: the outcome of this verb was thrown away — `>/dev/null 2>&1` discarded BOTH the exit code
|
|
141
|
+
# and the reason. A refusal that wrote NOTHING then fell through to the prompt-line checks below with
|
|
142
|
+
# an empty line, which is byte-identical to a clean submit, and this script exited 0 announcing
|
|
143
|
+
# DELIVERED_SUBMITTED — or QUEUED_BEHIND_TURN when the seat read `working`, which also EXPLAINS AWAY
|
|
144
|
+
# the silence that follows. Measured 2026-08-19 against a live seat: `herdr agent prompt` refuses a
|
|
145
|
+
# detected-but-undrivable occupant with `agent_not_ready` and never touches the input box, while
|
|
146
|
+
# `pane run` delivers to that same pane. A send receipt that cannot fail is not a receipt.
|
|
147
|
+
PROMPT_ERR=$(herdr agent prompt "$TARGET" "$MSG" 2>&1 >/dev/null)
|
|
148
|
+
PROMPT_RC=$?
|
|
149
|
+
VIA=""
|
|
150
|
+
if [ "$PROMPT_RC" -ne 0 ]; then
|
|
151
|
+
case "$PROMPT_ERR" in
|
|
152
|
+
*agent_not_ready*)
|
|
153
|
+
# UNEQUIVOCALLY PRE-WRITE: herdr rejected the target before typing anything, so a second
|
|
154
|
+
# delivery cannot duplicate. This is the ONLY class that earns a fallback.
|
|
155
|
+
PANE=$(pane_of)
|
|
156
|
+
if [ -z "$PANE" ]; then
|
|
157
|
+
echo "SEND_REFUSED $TARGET — agent prompt refused pre-write and no pane id resolved: $PROMPT_ERR"
|
|
158
|
+
exit 1
|
|
159
|
+
fi
|
|
160
|
+
PANE_ERR=$(herdr pane run "$PANE" "$MSG" 2>&1 >/dev/null)
|
|
161
|
+
PANE_RC=$?
|
|
162
|
+
if [ "$PANE_RC" -ne 0 ]; then
|
|
163
|
+
echo "SEND_FAILED $TARGET — both delivery surfaces refused; nothing was typed."
|
|
164
|
+
echo " agent prompt: $PROMPT_ERR"
|
|
165
|
+
echo " pane run ($PANE): $PANE_ERR"
|
|
166
|
+
exit 1
|
|
167
|
+
fi
|
|
168
|
+
VIA=" (via pane run $PANE — agent prompt refused pre-write: $PROMPT_ERR)"
|
|
169
|
+
;;
|
|
170
|
+
*)
|
|
171
|
+
# AMBIGUOUS: this class cannot prove the write did not land, and a second delivery APPENDS to
|
|
172
|
+
# whatever did. So it never retries — it surfaces the cause and stops, which is the one outcome
|
|
173
|
+
# the discarded stderr made impossible.
|
|
174
|
+
echo "SEND_UNVERIFIED $TARGET — agent prompt exited $PROMPT_RC with an ambiguous outcome: $PROMPT_ERR"
|
|
175
|
+
echo " NOT retrying (a second send appends to anything that landed). Read the pane, then decide."
|
|
176
|
+
exit 1
|
|
177
|
+
;;
|
|
178
|
+
esac
|
|
179
|
+
fi
|
|
111
180
|
|
|
112
181
|
sleep 2
|
|
182
|
+
|
|
183
|
+
# BEFORE any emptiness verdict: a non-empty staged queue means this seat has accepted text and is not
|
|
184
|
+
# running it, so an empty ordinary box proves nothing. Fail closed — there is no delivery remedy to
|
|
185
|
+
# offer, and every one was tried against the captured instance: bare `enter` on the empty box no-ops,
|
|
186
|
+
# `up`+`enter` moves the entry to the box and puts it straight BACK in the queue, `herdr agent
|
|
187
|
+
# send-keys` cannot type kimi's own `ctrl-s` affordance (no ctrl chords), and `pane run` is a
|
|
188
|
+
# shell-pane path with no effect on a TUI. Recovery is the operator's: capture provenance
|
|
189
|
+
# (pid/ppid/pgid/lstart), TERM the CLI, then `herdr agent start … -- --auto -c` in the SAME pane —
|
|
190
|
+
# session continuation preserved 136k of context and the queue drained on restart. NEVER retry here:
|
|
191
|
+
# the entry may drain at any moment and a second send would then run twice.
|
|
192
|
+
STAGED=$(staged_line)
|
|
193
|
+
if [ -n "$STAGED" ]; then
|
|
194
|
+
if is_ours "$STAGED"; then
|
|
195
|
+
echo "SEND_UNVERIFIED $TARGET — the directive is STAGED in the seat's queue, not running: $STAGED"
|
|
196
|
+
else
|
|
197
|
+
echo "SEND_UNVERIFIED $TARGET — this seat's queue holds an entry that is not draining, so nothing sent now is running: $STAGED"
|
|
198
|
+
fi
|
|
199
|
+
echo " An empty prompt line does NOT mean submitted on a staging TUI — the box is empty BECAUSE the"
|
|
200
|
+
echo " text went to the queue. Do NOT re-send: the queue may drain at any moment and run it twice."
|
|
201
|
+
echo " No input path drains it (enter no-ops, up+enter re-queues, send-keys has no ctrl chords,"
|
|
202
|
+
echo " pane run is shell-only). Capture provenance, TERM the CLI, restart it in the SAME pane with"
|
|
203
|
+
echo " session continuation, then verify the pane REVISION moves before trusting any receipt."
|
|
204
|
+
exit 1
|
|
205
|
+
fi
|
|
113
206
|
PL=$(prompt_line)
|
|
114
207
|
if [ -n "$PL" ] && is_ours "$PL"; then
|
|
115
208
|
# The text is SITTING, not sent — the swallowed-Enter shape. Submitting the existing draft is not a
|
|
@@ -127,14 +220,14 @@ fi
|
|
|
127
220
|
|
|
128
221
|
if [ -n "$PL" ]; then
|
|
129
222
|
# Not ours: either autosuggest ghost (dim SGR — benign) or a draft that appeared under us.
|
|
130
|
-
echo "DELIVERED_SUBMITTED $TARGET — prompt line clear of our text; NOTE it now holds: $PL"
|
|
223
|
+
echo "DELIVERED_SUBMITTED $TARGET$VIA — prompt line clear of our text; NOTE it now holds: $PL"
|
|
131
224
|
echo " ANSI-verify (dim = ghost). If typed, treat per D-206: discriminate before anyone clears it."
|
|
132
225
|
exit 0
|
|
133
226
|
fi
|
|
134
227
|
|
|
135
228
|
if [ -n "$QUEUED" ]; then
|
|
136
|
-
echo "QUEUED_BEHIND_TURN $TARGET — submitted into a working seat's queue; it is READ at turn end. A freeze-class or superseding directive belongs behind TKR_INTERRUPT=1."
|
|
229
|
+
echo "QUEUED_BEHIND_TURN $TARGET$VIA — submitted into a working seat's queue; it is READ at turn end. A freeze-class or superseding directive belongs behind TKR_INTERRUPT=1."
|
|
137
230
|
else
|
|
138
|
-
echo "DELIVERED_SUBMITTED $TARGET"
|
|
231
|
+
echo "DELIVERED_SUBMITTED $TARGET$VIA"
|
|
139
232
|
fi
|
|
140
233
|
exit 0
|