tickmarkr 2.4.2 → 2.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +110 -24
- package/dist/cli/commands/approve.d.ts +42 -0
- package/dist/cli/commands/approve.js +80 -13
- package/dist/cli/commands/doctor.d.ts +234 -0
- package/dist/cli/commands/doctor.js +139 -5
- package/dist/cli/commands/report.js +15 -44
- package/dist/cli/commands/scope.js +36 -6
- package/dist/cli/commands/stats.d.ts +2 -0
- package/dist/cli/commands/stats.js +42 -23
- package/dist/cli/commands/status.js +20 -4
- package/dist/cli/commands/ui.js +42 -53
- package/dist/cli/commands/unlock.d.ts +1 -1
- package/dist/cli/commands/unlock.js +59 -9
- package/dist/cli/help.d.ts +219 -0
- package/dist/cli/help.js +212 -0
- package/dist/cli/index.d.ts +42 -2
- package/dist/cli/index.js +23 -8
- package/dist/drivers/herdr.d.ts +7 -2
- package/dist/drivers/herdr.js +78 -48
- package/dist/drivers/orca.d.ts +3 -2
- package/dist/drivers/orca.js +43 -4
- package/dist/drivers/subprocess.d.ts +1 -0
- package/dist/drivers/subprocess.js +3 -0
- package/dist/drivers/types.d.ts +14 -0
- package/dist/gates/artifact-manifest.d.ts +50 -0
- package/dist/gates/artifact-manifest.js +23 -0
- package/dist/plan/scope.d.ts +25 -0
- package/dist/plan/scope.js +92 -12
- package/dist/report/operator-record.d.ts +49 -0
- package/dist/report/operator-record.js +137 -0
- package/dist/run/daemon.d.ts +3 -4
- package/dist/run/daemon.js +18 -10
- package/dist/run/lock.d.ts +59 -2
- package/dist/run/lock.js +184 -26
- package/dist/run/operator-state.d.ts +86 -0
- package/dist/run/operator-state.js +165 -0
- package/dist/run/supervision.d.ts +32 -0
- package/dist/run/supervision.js +138 -17
- package/dist/tui/cockpit/capture.d.ts +19 -0
- package/dist/tui/cockpit/capture.js +89 -1
- package/dist/tui/cockpit/components.d.ts +15 -1
- package/dist/tui/cockpit/components.js +79 -9
- package/dist/tui/cockpit/decision-actions.d.ts +147 -0
- package/dist/tui/cockpit/decision-actions.js +315 -0
- package/dist/tui/cockpit/derive.d.ts +1 -1
- package/dist/tui/cockpit/derive.js +2 -0
- package/dist/tui/cockpit/evidence-view.d.ts +119 -0
- package/dist/tui/cockpit/evidence-view.js +210 -0
- package/dist/tui/cockpit/home-view.d.ts +88 -0
- package/dist/tui/cockpit/home-view.js +240 -0
- package/dist/tui/cockpit/keys.d.ts +125 -0
- package/dist/tui/cockpit/keys.js +31 -0
- package/dist/tui/cockpit/layout.d.ts +14 -0
- package/dist/tui/cockpit/layout.js +15 -0
- package/dist/tui/cockpit/live-runtime.d.ts +46 -0
- package/dist/tui/cockpit/live-runtime.js +683 -0
- package/dist/tui/cockpit/live-store.d.ts +289 -0
- package/dist/tui/cockpit/live-store.js +308 -0
- package/dist/tui/cockpit/live.d.ts +21 -1
- package/dist/tui/cockpit/live.js +12 -1
- package/dist/tui/cockpit/run-view.d.ts +100 -0
- package/dist/tui/cockpit/run-view.js +202 -0
- package/dist/tui/cockpit/shell.d.ts +50 -0
- package/dist/tui/cockpit/shell.js +74 -0
- package/dist/tui/cockpit/theme.d.ts +27 -0
- package/dist/tui/cockpit/theme.js +21 -0
- package/package.json +1 -1
- package/skills/tickmarkr-auto/SKILL.md +10 -1
- package/skills/tickmarkr-loop/SKILL.md +72 -1
- package/skills/tickmarkr-overseer/SKILL.md +12 -0
package/README.md
CHANGED
|
@@ -79,7 +79,7 @@ tickmarkr init # guided setup + doctor; scaffolds config and spe
|
|
|
79
79
|
tickmarkr compile tickmarkr.spec.md # spec → task graph (fails without acceptance criteria)
|
|
80
80
|
tickmarkr plan # dry-run routing decisions + cost estimate
|
|
81
81
|
tickmarkr run # execute, route to best CLI, gate every result (--concurrency N)
|
|
82
|
-
tickmarkr report <runId> --md #
|
|
82
|
+
tickmarkr report <runId> --md # Markdown on stdout; redirect to save beside the spec
|
|
83
83
|
```
|
|
84
84
|
|
|
85
85
|
That's the flow: `init` scaffolds config, you write tasks with `acceptance[]` criteria, `compile`
|
|
@@ -113,16 +113,51 @@ Consent rules — every write is additive, never destructive:
|
|
|
113
113
|
|
|
114
114
|
## Monitor and supervise
|
|
115
115
|
|
|
116
|
+
`tickmarkr ui [runId]` opens one cockpit with **1 Home, 4 Run, 5 Evidence**. Without a
|
|
117
|
+
run ID it selects the latest journal; an empty repository opens Home. Use
|
|
118
|
+
`tickmarkr ui <runId> --view run` or `--view evidence` to open a delivered view directly.
|
|
119
|
+
`tickmarkr ui --setup <runId>` opens Run Parks and preserves the requested run identity.
|
|
120
|
+
Fleet/Bootstrap and Plan/Health are explicit follow-ons: keys 2/3/6 are not installed.
|
|
121
|
+
Use the existing `tickmarkr fleet`, `tickmarkr init`, `tickmarkr plan` and `tickmarkr doctor`
|
|
122
|
+
commands for those workflows.
|
|
123
|
+
|
|
124
|
+
`?` opens the shortcut sheet; Tab/Shift-Tab move focus through visible regions, Enter opens
|
|
125
|
+
selection, and Esc closes the deepest overlay. `q` quits; text-entry mode keeps `q1?`
|
|
126
|
+
literal. Run shows every task in the matching graph, its recorded attempt/path/pane/alarm,
|
|
127
|
+
and current-attempt gate evidence. `o` requests focus of the recorded owned pane when the
|
|
128
|
+
driver supports it; unavailable panes leave an evidence diagnostic. Evidence keeps original
|
|
129
|
+
journal `#L` identities, full verdict paging, and a stable selection with Follow off.
|
|
130
|
+
|
|
116
131
|
```bash
|
|
117
|
-
tickmarkr status
|
|
132
|
+
tickmarkr status <runId> # preserved printed engagement state
|
|
133
|
+
tickmarkr status <runId> --oneline # compact snapshot, then exit
|
|
134
|
+
tickmarkr status <runId> --watch # TTY: Run cockpit; non-TTY: line output
|
|
135
|
+
tickmarkr status <runId> --watch --plain # preserved line/ANSI fallback, including on a TTY
|
|
118
136
|
tickmarkr resume <runId> # continue an engagement from the local execution log
|
|
119
|
-
tickmarkr approve <runId> <taskId> #
|
|
137
|
+
tickmarkr approve <runId> <taskId> # append permission for a non-gate park; see below
|
|
120
138
|
tickmarkr report <runId> # cost/quality report
|
|
139
|
+
tickmarkr report <runId> --md > feature.record.md # explicit file write beside your spec
|
|
121
140
|
tickmarkr profile # show the learned routing profile
|
|
122
141
|
tickmarkr profile --explain <shape> <channel> # why a channel ranks where it does for a shape
|
|
123
142
|
```
|
|
124
143
|
|
|
125
|
-
|
|
144
|
+
For machines, `tickmarkr status <runId> --watch --events` replays and follows projected
|
|
145
|
+
decision events as one JSON document per stdout line. `--jsonl` and `--decision-events`
|
|
146
|
+
are aliases; keep stderr keepalives separate (never `2>&1`). This projection is distinct
|
|
147
|
+
from raw `journal.jsonl`. Webhook delivery remains opt-in via `--webhook <url>`.
|
|
148
|
+
Preserved printed twins also include separate `fleet --print` and `fleet --why` outputs,
|
|
149
|
+
`plan`, `doctor`, `report --compare <baseline-runId>`, `report --bundle <path>` (an explicit
|
|
150
|
+
proof-bundle write), and `stats` (all runs, no run ID). Report retains its learning preview
|
|
151
|
+
and comparison warnings; absent metering stays “not measurable,” never an invented $0.
|
|
152
|
+
|
|
153
|
+
A manual cockpit keeps its final receipt at run-end and follows a later resume of that run.
|
|
154
|
+
By default, the daemon-owned board requests graceful shutdown at run-end and closes only its owned pane
|
|
155
|
+
after checking its watch presence stood down; unconfirmed cleanup is reported. The existing
|
|
156
|
+
`visibility.keepPanes: forever` debug override preserves panes. Herdr keeps
|
|
157
|
+
task/gate grouping, short titles, and the board beside its caller without taking focus.
|
|
158
|
+
Quitting or orderly signals restore raw mode, pointer tracking, title and alternate screen,
|
|
159
|
+
and release only that observer's presence. Green tasks land on `tickmarkr/<runId>`;
|
|
160
|
+
merge to your mainline is always your call.
|
|
126
161
|
|
|
127
162
|
### Escalation and consults
|
|
128
163
|
|
|
@@ -133,15 +168,57 @@ away from a failing adapter will never retry it in subsequent `tickmarkr resume`
|
|
|
133
168
|
|
|
134
169
|
### Approving tasks
|
|
135
170
|
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
171
|
+
The recorded partial-human-park case has **1/3 merged**, human T2, blocked T3, and a
|
|
172
|
+
passed tip verify. It is **PARTIAL**, not green: T2's `humanGate: true` parks it **before
|
|
173
|
+
dispatch**. In Run (`4`, or `tickmarkr ui --setup <runId>`), select T2, press `a` for
|
|
174
|
+
Actions, choose Approve with Enter, and review the confirmation: run/task, original park
|
|
175
|
+
`#L`, actor/reason, exact `tickmarkr approve` argv, append-only consequence and enactor.
|
|
176
|
+
Only `y` confirms; `n`/Esc cancel, and Enter never confirms.
|
|
177
|
+
|
|
178
|
+
Read the receipt's newly appended `task-approved` line, actor/reason and disposition back
|
|
179
|
+
from the journal. Approval records permission; it does not dispatch work, pass a gate or
|
|
180
|
+
mark the task done. With no live owner the receipt says **approved; resume required**.
|
|
181
|
+
Exit the observer and run `tickmarkr resume <runId>` explicitly. A matching live daemon
|
|
182
|
+
can enact the release at its next task boundary; a different live run must end before this
|
|
183
|
+
one resumes. Keep any resume refusal and its remediation visible, including a deny/prefer
|
|
184
|
+
config conflict; repair the source/config as directed, never edit the compiled graph to
|
|
185
|
+
force success. After resume, CURRENT TIP is PENDING until fresh evidence arrives. Completion
|
|
186
|
+
requires the latest run-end, a nonfailed known tip result, and empty `failed`, `human`,
|
|
187
|
+
`blocked` and `pending` buckets; the completed case has 3/3 recorded merges. An unrelated
|
|
188
|
+
graph says “not comparable” and supplies no borrowed denominator. Historical GATES RAN
|
|
189
|
+
does not establish current completion.
|
|
190
|
+
|
|
191
|
+
The CLI twin for the same decision is
|
|
192
|
+
`tickmarkr approve <runId> T2 --by operator --reason 'ready to proceed'`, followed by the
|
|
193
|
+
receipt check and explicit resume above. Other parks have different permitted decisions:
|
|
194
|
+
|
|
195
|
+
| Park | Decision and effect |
|
|
196
|
+
|---|---|
|
|
197
|
+
| Human gate / other non-gate park | Plain approve records permission to dispatch. |
|
|
198
|
+
| Attempt cap | Plain approve grants a fresh attempt budget; prior routing exclusions remain. |
|
|
199
|
+
| Infrastructure | Plain approve or `--recheck`; recheck reruns the declared battery and satisfies no gate. |
|
|
200
|
+
| Failed review gate | `--waive` satisfies only that identified gate; `--uphold` funds one fixed attempt carrying findings; `--recheck` reruns the battery. |
|
|
201
|
+
| Other failed gate | `--waive` or `--recheck`; plain approve refuses. |
|
|
202
|
+
| Tombstone / gate failure without identifying evidence | Diagnostic only; no invented decision. |
|
|
203
|
+
|
|
204
|
+
Decisions are append-only and cannot be undone. Unknown tasks, duplicate decisions and
|
|
205
|
+
changed parks refuse; no success receipt is claimed without reading back the append.
|
|
206
|
+
|
|
207
|
+
### Help and recovery
|
|
208
|
+
|
|
209
|
+
`tickmarkr ui --help`, `tickmarkr eval --help`, `tickmarkr unlock --help` and
|
|
210
|
+
`tickmarkr profile reset --help` print guidance without opening a UI, seeding fixtures,
|
|
211
|
+
unlocking or resetting history. Help flags are recognized before `--`; arguments after
|
|
212
|
+
it are literal data. Help examples describe operations; printing them never executes them.
|
|
213
|
+
|
|
214
|
+
Recovery remains explicit: `unlock <runId>` targets a matching provably dead lock;
|
|
215
|
+
`unlock --garbage` handles malformed lock bytes without inventing a run ID. Both require
|
|
216
|
+
confirmation (`--yes` for non-TTY) and refuse live, inaccessible or changed holders.
|
|
217
|
+
`doctor --cached` reads cached diagnostics; `doctor --probe-preflight` discloses probe
|
|
218
|
+
counts/files; `doctor --fix-only` repairs locally without model probes or catalog refresh.
|
|
219
|
+
Default doctor and `--fix` still probe. `doctor --refresh-catalog` refreshes only the catalog.
|
|
220
|
+
`scope <intent-file> --preview` is a local, non-writing preview; authoring requires TTY
|
|
221
|
+
confirmation or `--yes` and can make model calls.
|
|
145
222
|
|
|
146
223
|
## Choosing your fleet: `tickmarkr fleet`
|
|
147
224
|
|
|
@@ -155,7 +232,7 @@ tickmarkr plan # lint the resolved routing table against your spec
|
|
|
155
232
|
tickmarkr run # dispatch with the fleet you confirmed
|
|
156
233
|
```
|
|
157
234
|
|
|
158
|
-
`tickmarkr doctor`
|
|
235
|
+
`tickmarkr doctor` probes and records health; `tickmarkr fleet` edits routing config. The browser is one
|
|
159
236
|
two-pane surface: the left rail lists views (**All models**, **Shapes**, **Steering**) and every
|
|
160
237
|
installed agent CLI with its auth state and model count; the right pane is a searchable model list
|
|
161
238
|
with tier, context, price, and probe-latency columns. `Space` allows/denies, `Enter` classifies an
|
|
@@ -283,7 +360,7 @@ the frontier-model consult is the *National Office*. The terms below use that vo
|
|
|
283
360
|
- (bare token) — member is running normally
|
|
284
361
|
- **cleanup · <taskId>**: overflow/teardown generation tabs. When a new generation starts (on retry escalation), a new cleanup tab
|
|
285
362
|
opens labeled with the newest live member's task ID; it auto-closes when the generation completes
|
|
286
|
-
- **watch**: a single pane running `tickmarkr
|
|
363
|
+
- **watch**: a single owned pane running `tickmarkr ui <runId> --view run` — the same Run cockpit opened by TTY `status --watch`
|
|
287
364
|
|
|
288
365
|
### Pane naming (when visibility.llm = pane)
|
|
289
366
|
|
|
@@ -311,13 +388,17 @@ This keeps the Partner focused on decisions that require attention, not noise.
|
|
|
311
388
|
|
|
312
389
|
tickmarkr closes exactly what it owns and no longer needs, no matter how any process died.
|
|
313
390
|
|
|
314
|
-
|
|
391
|
+
Tabs use short human labels (at most 20 characters); panes carry durable ownership names:
|
|
315
392
|
- `<taskId>` — the task's tab, holding its worker and its judge/review/consult panes
|
|
316
393
|
- `cleanup · <taskId>` — teardown generation tab for overflow attempts
|
|
317
|
-
- `watch
|
|
318
|
-
-
|
|
394
|
+
- `tickmarkr:watch:run:0:<runId>` — daemon-owned Run board
|
|
395
|
+
- `tickmarkr:<role>:<taskId>:<attempt>:<runId>` — durable judge, review, consult and worker pane names; display titles may be shorter
|
|
319
396
|
|
|
320
|
-
|
|
397
|
+
Reconciliation stays within the run's workspace. It can retire another run's panes only
|
|
398
|
+
when this repository's journal evidence proves that run ended; live or unknown runs and
|
|
399
|
+
other workspaces remain protected. Pane names outside the ownership contract are **foreign**.
|
|
400
|
+
Board replacement additionally checks repository/run ownership and its acknowledged watch
|
|
401
|
+
presence; matching a short tab label is never sufficient.
|
|
321
402
|
|
|
322
403
|
**Desired-state reconciliation**: A pure function computes the exact set of panes that should exist from the local journal at any moment:
|
|
323
404
|
- Worker panes for all in-flight task attempts
|
|
@@ -326,12 +407,14 @@ tickmarkr creates all owned panes only within the run's workspace; any tickmarkr
|
|
|
326
407
|
- Empty set (after engagement end)
|
|
327
408
|
|
|
328
409
|
The daemon reconciles at every safe point:
|
|
329
|
-
1. **Run start** —
|
|
410
|
+
1. **Run start** — reconcile this run and older runs proven ended in this repository
|
|
330
411
|
2. **Resume** — reconcile the restarted journal state and close panes for superseded attempts
|
|
331
412
|
3. **After terminal events** (task done, failed, human gate) — close the corresponding worker/gate pane and its emptied tab
|
|
332
|
-
4. **At engagement end** — close
|
|
413
|
+
4. **At engagement end** — close remaining owned panes and tabs, with graceful board shutdown
|
|
333
414
|
|
|
334
|
-
|
|
415
|
+
`visibility.keepPanes: forever` disables this sweep. Reconciliation failures (herdr
|
|
416
|
+
unavailable, a pane vanished mid-sweep) do not replace gate verdicts; unconfirmed board
|
|
417
|
+
cleanup is journaled and reported.
|
|
335
418
|
|
|
336
419
|
### Workspace trust
|
|
337
420
|
|
|
@@ -372,10 +455,13 @@ them; worker-declared deviations are recorded as notes, not authority.
|
|
|
372
455
|
|
|
373
456
|
If you clone this repo and use Claude Code, project skills are installed in `.claude/skills/`:
|
|
374
457
|
|
|
375
|
-
-
|
|
376
|
-
-
|
|
458
|
+
- **[/tickmarkr-loop](skills/tickmarkr-loop/SKILL.md)** — compile a spec, review the routing plan, run the engagement, and commit the Markdown record
|
|
459
|
+
- **[/tickmarkr-auto](skills/tickmarkr-auto/SKILL.md)** — autonomous multi-phase runs (GSD milestones, etc.)
|
|
377
460
|
|
|
378
461
|
These are optional — the CLI works standalone. Skills are repo-scoped and ship in the npm tarball for agents working in projects that have run `tickmarkr init --agent`.
|
|
462
|
+
The canonical sources live in `skills/`; this repository's installed `.claude/skills/` links
|
|
463
|
+
resolve there. The [overseer skill](skills/tickmarkr-overseer/SKILL.md) links to the same
|
|
464
|
+
loop walkthrough for cockpit and decision guidance; skill names remain unchanged.
|
|
379
465
|
|
|
380
466
|
## Contributing
|
|
381
467
|
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { Journal, type JournalEvent } from "../../run/journal.js";
|
|
1
2
|
export declare const APPROVAL_DISPOSITIONS: readonly ["dispatch", "waive-gate", "re-dispatch", "fund-fixed-attempt", "fresh-budget"];
|
|
2
3
|
export type ApprovalDisposition = (typeof APPROVAL_DISPOSITIONS)[number];
|
|
3
4
|
/**
|
|
@@ -8,6 +9,47 @@ export type ApprovalDisposition = (typeof APPROVAL_DISPOSITIONS)[number];
|
|
|
8
9
|
*/
|
|
9
10
|
export declare const APPROVAL_ENACTS: Record<ApprovalDisposition, string>;
|
|
10
11
|
export declare function approvalDispositionForRelease(release: unknown): ApprovalDisposition;
|
|
12
|
+
/** The closed verb set every decision surface may name. Nothing outside it reaches this command. */
|
|
13
|
+
export declare const DECISION_VERBS: readonly ["approve", "waive", "uphold", "recheck"];
|
|
14
|
+
export type DecisionVerb = (typeof DECISION_VERBS)[number];
|
|
15
|
+
/** The newest park a decision binds to, read the one way this command reads it. */
|
|
16
|
+
export interface NewestPark {
|
|
17
|
+
/** Index into the journal's event array; `line` is the physical 1-based journal line. */
|
|
18
|
+
index: number;
|
|
19
|
+
line: number;
|
|
20
|
+
ts: string | undefined;
|
|
21
|
+
/** The daemon-recorded kind (task-human data.kind), never inferred from prose. */
|
|
22
|
+
kind: string | undefined;
|
|
23
|
+
reason: string | undefined;
|
|
24
|
+
/** The newest failed gate before the park — the gate a waive would satisfy. */
|
|
25
|
+
failedGate: string | undefined;
|
|
26
|
+
/** A pre-dispatch human gate whose reason marks it permanent by design (see isTombstonePark). */
|
|
27
|
+
tombstone: boolean;
|
|
28
|
+
}
|
|
29
|
+
/**
|
|
30
|
+
* There is no closed park kind for a declaration-shaped retirement, so the evidence is the one the
|
|
31
|
+
* daemon recorded: a pre-dispatch human-gate park whose reason carries the task's own title, where the
|
|
32
|
+
* spec declares the tombstone. Read narrowly on purpose — every other kind is actionable regardless of prose.
|
|
33
|
+
*/
|
|
34
|
+
export declare function isTombstonePark(kind: string | undefined, reason: string | undefined): boolean;
|
|
35
|
+
export declare function newestPark(events: readonly JournalEvent[], taskId: string,
|
|
36
|
+
/** Physical zero-based source indexes corresponding one-for-one with `events`. */
|
|
37
|
+
sourceIndexes?: readonly number[]): NewestPark | undefined;
|
|
38
|
+
/** Parsed events paired with their immutable physical JSONL identities. */
|
|
39
|
+
export declare function readJournalEvents(journal: Journal): {
|
|
40
|
+
events: JournalEvent[];
|
|
41
|
+
sourceIndexes: number[];
|
|
42
|
+
};
|
|
43
|
+
/**
|
|
44
|
+
* FINAL §3.3's decision menu as data: human-gate/attempt-cap/other non-gate parks → approve; infra →
|
|
45
|
+
* approve or recheck; review gate-fail → waive/uphold/recheck; other gate-fail → waive/recheck; a
|
|
46
|
+
* gate-fail park with no failed-gate evidence, or a tombstone → nothing (a diagnostic, never a
|
|
47
|
+
* fabricated verb). The refusals in `approve` below enforce the same table; this is the one place a
|
|
48
|
+
* surface may read it from, so what a menu offers and what the command accepts cannot drift.
|
|
49
|
+
*/
|
|
50
|
+
export declare function permittedDecisionVerbs(park: Pick<NewestPark, "kind" | "failedGate" | "tombstone"> | undefined): readonly DecisionVerb[];
|
|
51
|
+
/** The release marker this command appends for a verb on a park — the fact a read-back must match. */
|
|
52
|
+
export declare function releaseForDecision(verb: DecisionVerb, park: Pick<NewestPark, "kind" | "failedGate">): string | undefined;
|
|
11
53
|
export type ApprovalStatus = "deferred-live" | "recorded-no-owner";
|
|
12
54
|
/** The requested run plus the different live run currently blocking its repository, when present. */
|
|
13
55
|
export interface ApprovalRunOwner {
|
|
@@ -27,6 +27,73 @@ export function approvalDispositionForRelease(release) {
|
|
|
27
27
|
return "dispatch";
|
|
28
28
|
}
|
|
29
29
|
import { acquireApprovalSerialization, runLockOwner } from "../../run/lock.js";
|
|
30
|
+
/** The closed verb set every decision surface may name. Nothing outside it reaches this command. */
|
|
31
|
+
export const DECISION_VERBS = ["approve", "waive", "uphold", "recheck"];
|
|
32
|
+
/**
|
|
33
|
+
* There is no closed park kind for a declaration-shaped retirement, so the evidence is the one the
|
|
34
|
+
* daemon recorded: a pre-dispatch human-gate park whose reason carries the task's own title, where the
|
|
35
|
+
* spec declares the tombstone. Read narrowly on purpose — every other kind is actionable regardless of prose.
|
|
36
|
+
*/
|
|
37
|
+
export function isTombstonePark(kind, reason) {
|
|
38
|
+
// Declaration retirements use an explicit title marker: either an em-dash-delimited
|
|
39
|
+
// `— tombstone` suffix or the canonical `tombstone, never dispatched` phrase. A task title
|
|
40
|
+
// that merely discusses tombstones is still an ordinary, actionable human gate.
|
|
41
|
+
return kind === "human-gate"
|
|
42
|
+
&& /(?:\s—\s+tombstone\b|\btombstone,\s*never dispatched\b)/iu.test(reason ?? "");
|
|
43
|
+
}
|
|
44
|
+
export function newestPark(events, taskId,
|
|
45
|
+
/** Physical zero-based source indexes corresponding one-for-one with `events`. */
|
|
46
|
+
sourceIndexes) {
|
|
47
|
+
for (let i = events.length - 1; i >= 0; i--) {
|
|
48
|
+
const e = events[i];
|
|
49
|
+
if (e.event !== "task-human" || e.taskId !== taskId)
|
|
50
|
+
continue;
|
|
51
|
+
const kind = typeof e.data.kind === "string" ? e.data.kind : undefined;
|
|
52
|
+
const reason = typeof e.data.reason === "string" ? e.data.reason : undefined;
|
|
53
|
+
return {
|
|
54
|
+
index: i, line: (sourceIndexes?.[i] ?? i) + 1, ts: typeof e.ts === "string" ? e.ts : undefined, kind, reason,
|
|
55
|
+
failedGate: failedGateForNewestPark(events, taskId, i), tombstone: isTombstonePark(kind, reason),
|
|
56
|
+
};
|
|
57
|
+
}
|
|
58
|
+
return undefined;
|
|
59
|
+
}
|
|
60
|
+
/** Parsed events paired with their immutable physical JSONL identities. */
|
|
61
|
+
export function readJournalEvents(journal) {
|
|
62
|
+
const tracked = journal.readTracked();
|
|
63
|
+
return {
|
|
64
|
+
events: tracked.map((row) => row.raw),
|
|
65
|
+
sourceIndexes: tracked.map((row) => row.sourceIndex),
|
|
66
|
+
};
|
|
67
|
+
}
|
|
68
|
+
/**
|
|
69
|
+
* FINAL §3.3's decision menu as data: human-gate/attempt-cap/other non-gate parks → approve; infra →
|
|
70
|
+
* approve or recheck; review gate-fail → waive/uphold/recheck; other gate-fail → waive/recheck; a
|
|
71
|
+
* gate-fail park with no failed-gate evidence, or a tombstone → nothing (a diagnostic, never a
|
|
72
|
+
* fabricated verb). The refusals in `approve` below enforce the same table; this is the one place a
|
|
73
|
+
* surface may read it from, so what a menu offers and what the command accepts cannot drift.
|
|
74
|
+
*/
|
|
75
|
+
export function permittedDecisionVerbs(park) {
|
|
76
|
+
if (!park || park.tombstone)
|
|
77
|
+
return [];
|
|
78
|
+
if (park.kind === "gate-fail") {
|
|
79
|
+
if (park.failedGate === undefined)
|
|
80
|
+
return [];
|
|
81
|
+
return park.failedGate === "review" ? ["waive", "uphold", "recheck"] : ["waive", "recheck"];
|
|
82
|
+
}
|
|
83
|
+
if (park.kind === "infra")
|
|
84
|
+
return ["approve", "recheck"];
|
|
85
|
+
return ["approve"];
|
|
86
|
+
}
|
|
87
|
+
/** The release marker this command appends for a verb on a park — the fact a read-back must match. */
|
|
88
|
+
export function releaseForDecision(verb, park) {
|
|
89
|
+
if (verb === "waive")
|
|
90
|
+
return GATE_SATISFIED_RELEASE;
|
|
91
|
+
if (verb === "uphold")
|
|
92
|
+
return REVIEW_UPHELD_RELEASE;
|
|
93
|
+
if (verb === "recheck")
|
|
94
|
+
return RECHECK_RELEASE;
|
|
95
|
+
return park.kind === ATTEMPT_CAP_RELEASE ? ATTEMPT_CAP_RELEASE : undefined;
|
|
96
|
+
}
|
|
30
97
|
// THE liveness rule, written once. The lock is REPOSITORY-wide, so a live owner of some OTHER run is
|
|
31
98
|
// not an owner of this one and sweeps none of its approvals — claiming otherwise is the same
|
|
32
99
|
// falsehood in a new shape. Liveness itself comes from lock.ts's runLockOwner (the same inspect() the
|
|
@@ -74,19 +141,13 @@ export async function approve(argv, cwd = process.cwd()) {
|
|
|
74
141
|
}
|
|
75
142
|
// OBS-18: only the most recent task-human for this task decides whether this approval grants a
|
|
76
143
|
// fresh attempt budget. The closed daemon-issued kind, never a human prose string, controls release.
|
|
77
|
-
const events = journal
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
}
|
|
85
|
-
const lastHuman = events[lastHumanIndex];
|
|
86
|
-
const capPark = lastHuman?.data.kind === ATTEMPT_CAP_RELEASE;
|
|
87
|
-
const gateFailPark = lastHuman?.data.kind === "gate-fail";
|
|
88
|
-
const infraPark = lastHuman?.data.kind === "infra";
|
|
89
|
-
const failedGate = gateFailPark ? failedGateForNewestPark(events, taskId, lastHumanIndex) : undefined;
|
|
144
|
+
const { events, sourceIndexes } = readJournalEvents(journal);
|
|
145
|
+
const park = newestPark(events, taskId, sourceIndexes);
|
|
146
|
+
const lastHuman = park === undefined ? undefined : events[park.index];
|
|
147
|
+
const capPark = park?.kind === ATTEMPT_CAP_RELEASE;
|
|
148
|
+
const gateFailPark = park?.kind === "gate-fail";
|
|
149
|
+
const infraPark = park?.kind === "infra";
|
|
150
|
+
const failedGate = gateFailPark ? park?.failedGate : undefined;
|
|
90
151
|
if (gateFailPark && !failedGate) {
|
|
91
152
|
throw new Error(`task ${taskId} is parked on gate-fail but has no failed gate result on the newest park — refusing to infer one`);
|
|
92
153
|
}
|
|
@@ -137,6 +198,12 @@ export async function approve(argv, cwd = process.cwd()) {
|
|
|
137
198
|
choices.push(`--uphold (disposition fund-fixed-attempt)`);
|
|
138
199
|
throw new Error(`task ${taskId} is parked on failed gate ${failedGate}; plain approve has disposition only for non-gate parks — pass ${choices.join(" or ")}`);
|
|
139
200
|
}
|
|
201
|
+
// A tombstone has no verb in permittedDecisionVerbs (the named decisions above already refused it
|
|
202
|
+
// as a non-gate park); the command enforces the same table for plain approve, so a surface that
|
|
203
|
+
// skips the helper still cannot release it.
|
|
204
|
+
if (park?.tombstone) {
|
|
205
|
+
throw new Error(`task ${taskId}'s newest park is a tombstone (${park.reason ?? "no reason"}) — permanent by design; no verb releases it`);
|
|
206
|
+
}
|
|
140
207
|
journal.append("task-approved", taskId, {
|
|
141
208
|
by,
|
|
142
209
|
...(reason ? { reason } : {}),
|
|
@@ -96,5 +96,239 @@ export declare function selfShadowFinding(ownVersion: string, cwd?: string, reso
|
|
|
96
96
|
* check) reads it and bumps the constant.
|
|
97
97
|
*/
|
|
98
98
|
export declare function liveBenchStalenessFinding(now: Date): string | undefined;
|
|
99
|
+
/**
|
|
100
|
+
* A no-call disclosure for the action that normal doctor performs. This is deliberately built
|
|
101
|
+
* from config alone: opening the disclosure must not turn a cache reader into a model purchase.
|
|
102
|
+
*/
|
|
103
|
+
export declare function doctorProbePreflight(cwd?: string, cfg?: {
|
|
104
|
+
concurrency: number;
|
|
105
|
+
driver: "auto" | "herdr" | "subprocess" | "orca";
|
|
106
|
+
integrationBranchPrefix: string;
|
|
107
|
+
taskTimeoutMinutes: number;
|
|
108
|
+
contextWarnTokens: number;
|
|
109
|
+
routing: {
|
|
110
|
+
map: Record<string, {
|
|
111
|
+
pin?: {
|
|
112
|
+
via: string;
|
|
113
|
+
model: string;
|
|
114
|
+
} | undefined;
|
|
115
|
+
pool?: {
|
|
116
|
+
mode: "any" | "ordered";
|
|
117
|
+
channels: string[];
|
|
118
|
+
} | undefined;
|
|
119
|
+
tier?: undefined;
|
|
120
|
+
prefer?: string[] | undefined;
|
|
121
|
+
escalate?: boolean | undefined;
|
|
122
|
+
}>;
|
|
123
|
+
floors: Record<string, "cheap" | "mid" | "frontier">;
|
|
124
|
+
learned: "on" | "off";
|
|
125
|
+
allowUnverifiedModels: boolean;
|
|
126
|
+
mode?: "partner-led" | "risk-based" | "staff-led" | undefined;
|
|
127
|
+
learnedTuning?: {
|
|
128
|
+
halfLifeRuns?: number | undefined;
|
|
129
|
+
availWeight?: number | undefined;
|
|
130
|
+
} | undefined;
|
|
131
|
+
explore?: {
|
|
132
|
+
mode?: "on" | "off" | undefined;
|
|
133
|
+
excludeShapes?: string[] | undefined;
|
|
134
|
+
excludeComplexityAtOrAbove?: number | null | undefined;
|
|
135
|
+
cap?: number | undefined;
|
|
136
|
+
} | undefined;
|
|
137
|
+
sla?: Record<string, number> | undefined;
|
|
138
|
+
allow?: {
|
|
139
|
+
adapters?: string[] | undefined;
|
|
140
|
+
models?: string[] | undefined;
|
|
141
|
+
} | undefined;
|
|
142
|
+
deny?: {
|
|
143
|
+
adapters?: string[] | undefined;
|
|
144
|
+
models?: string[] | undefined;
|
|
145
|
+
workers?: {
|
|
146
|
+
adapters?: string[] | undefined;
|
|
147
|
+
models?: string[] | undefined;
|
|
148
|
+
} | undefined;
|
|
149
|
+
} | undefined;
|
|
150
|
+
};
|
|
151
|
+
tiers: Record<string, {
|
|
152
|
+
vendor: string | null;
|
|
153
|
+
channel: "sub" | "api";
|
|
154
|
+
models: Record<string, "cheap" | "mid" | "frontier">;
|
|
155
|
+
modelOverrides?: Record<string, {
|
|
156
|
+
vendor?: string | undefined;
|
|
157
|
+
channel?: "sub" | "api" | undefined;
|
|
158
|
+
}> | undefined;
|
|
159
|
+
windows?: Record<string, number> | undefined;
|
|
160
|
+
}>;
|
|
161
|
+
pricing: Record<string, number>;
|
|
162
|
+
gates: {
|
|
163
|
+
build?: string | undefined;
|
|
164
|
+
test?: string | undefined;
|
|
165
|
+
lint?: string | undefined;
|
|
166
|
+
diffCap?: number | undefined;
|
|
167
|
+
byShape?: Partial<Record<"plan" | "spec" | "implement" | "tests" | "docs" | "migration" | "ui" | "refactor" | "chore", {
|
|
168
|
+
[x: string]: unknown;
|
|
169
|
+
acceptance?: boolean | undefined;
|
|
170
|
+
review?: boolean | undefined;
|
|
171
|
+
}>> | undefined;
|
|
172
|
+
};
|
|
173
|
+
judge: {
|
|
174
|
+
adapter: string;
|
|
175
|
+
model: string;
|
|
176
|
+
};
|
|
177
|
+
review: {
|
|
178
|
+
complexityThreshold: number;
|
|
179
|
+
timeoutMs: number;
|
|
180
|
+
required: boolean;
|
|
181
|
+
prefer?: string[] | undefined;
|
|
182
|
+
policy?: "full" | "judge-only" | undefined;
|
|
183
|
+
criticalPaths?: string[] | undefined;
|
|
184
|
+
};
|
|
185
|
+
consult: {
|
|
186
|
+
adapter: string;
|
|
187
|
+
model: string;
|
|
188
|
+
stallMinutes: number;
|
|
189
|
+
prefer?: string[] | undefined;
|
|
190
|
+
};
|
|
191
|
+
visibility: {
|
|
192
|
+
llm: "pane" | "headless";
|
|
193
|
+
keepPanes: "run" | "attempt" | "forever";
|
|
194
|
+
worker: "interactive" | "print";
|
|
195
|
+
workersPerTab: number;
|
|
196
|
+
};
|
|
197
|
+
setup?: string | undefined;
|
|
198
|
+
cost?: {
|
|
199
|
+
models?: Record<string, {
|
|
200
|
+
inPerMtok: number;
|
|
201
|
+
outPerMtok: number;
|
|
202
|
+
cacheReadPerMtok?: number | undefined;
|
|
203
|
+
rateDate?: string | undefined;
|
|
204
|
+
}> | undefined;
|
|
205
|
+
subs?: Record<string, {
|
|
206
|
+
planMonthly: number;
|
|
207
|
+
windowsPerMonthLow: number;
|
|
208
|
+
windowsPerMonthHigh: number;
|
|
209
|
+
}> | undefined;
|
|
210
|
+
} | undefined;
|
|
211
|
+
scope?: {
|
|
212
|
+
allowDeviations?: string[] | undefined;
|
|
213
|
+
} | undefined;
|
|
214
|
+
}, adapters?: WorkerAdapter[]): string;
|
|
215
|
+
/**
|
|
216
|
+
* Health surfaces use this cache-only reader. It intentionally receives adapters only to render
|
|
217
|
+
* their configured identities; it never invokes adapter.probe(), listModels(), or a model command.
|
|
218
|
+
*/
|
|
219
|
+
export declare function cachedDoctorDiagnostics(cwd?: string, adapters?: WorkerAdapter[], cfg?: {
|
|
220
|
+
concurrency: number;
|
|
221
|
+
driver: "auto" | "herdr" | "subprocess" | "orca";
|
|
222
|
+
integrationBranchPrefix: string;
|
|
223
|
+
taskTimeoutMinutes: number;
|
|
224
|
+
contextWarnTokens: number;
|
|
225
|
+
routing: {
|
|
226
|
+
map: Record<string, {
|
|
227
|
+
pin?: {
|
|
228
|
+
via: string;
|
|
229
|
+
model: string;
|
|
230
|
+
} | undefined;
|
|
231
|
+
pool?: {
|
|
232
|
+
mode: "any" | "ordered";
|
|
233
|
+
channels: string[];
|
|
234
|
+
} | undefined;
|
|
235
|
+
tier?: undefined;
|
|
236
|
+
prefer?: string[] | undefined;
|
|
237
|
+
escalate?: boolean | undefined;
|
|
238
|
+
}>;
|
|
239
|
+
floors: Record<string, "cheap" | "mid" | "frontier">;
|
|
240
|
+
learned: "on" | "off";
|
|
241
|
+
allowUnverifiedModels: boolean;
|
|
242
|
+
mode?: "partner-led" | "risk-based" | "staff-led" | undefined;
|
|
243
|
+
learnedTuning?: {
|
|
244
|
+
halfLifeRuns?: number | undefined;
|
|
245
|
+
availWeight?: number | undefined;
|
|
246
|
+
} | undefined;
|
|
247
|
+
explore?: {
|
|
248
|
+
mode?: "on" | "off" | undefined;
|
|
249
|
+
excludeShapes?: string[] | undefined;
|
|
250
|
+
excludeComplexityAtOrAbove?: number | null | undefined;
|
|
251
|
+
cap?: number | undefined;
|
|
252
|
+
} | undefined;
|
|
253
|
+
sla?: Record<string, number> | undefined;
|
|
254
|
+
allow?: {
|
|
255
|
+
adapters?: string[] | undefined;
|
|
256
|
+
models?: string[] | undefined;
|
|
257
|
+
} | undefined;
|
|
258
|
+
deny?: {
|
|
259
|
+
adapters?: string[] | undefined;
|
|
260
|
+
models?: string[] | undefined;
|
|
261
|
+
workers?: {
|
|
262
|
+
adapters?: string[] | undefined;
|
|
263
|
+
models?: string[] | undefined;
|
|
264
|
+
} | undefined;
|
|
265
|
+
} | undefined;
|
|
266
|
+
};
|
|
267
|
+
tiers: Record<string, {
|
|
268
|
+
vendor: string | null;
|
|
269
|
+
channel: "sub" | "api";
|
|
270
|
+
models: Record<string, "cheap" | "mid" | "frontier">;
|
|
271
|
+
modelOverrides?: Record<string, {
|
|
272
|
+
vendor?: string | undefined;
|
|
273
|
+
channel?: "sub" | "api" | undefined;
|
|
274
|
+
}> | undefined;
|
|
275
|
+
windows?: Record<string, number> | undefined;
|
|
276
|
+
}>;
|
|
277
|
+
pricing: Record<string, number>;
|
|
278
|
+
gates: {
|
|
279
|
+
build?: string | undefined;
|
|
280
|
+
test?: string | undefined;
|
|
281
|
+
lint?: string | undefined;
|
|
282
|
+
diffCap?: number | undefined;
|
|
283
|
+
byShape?: Partial<Record<"plan" | "spec" | "implement" | "tests" | "docs" | "migration" | "ui" | "refactor" | "chore", {
|
|
284
|
+
[x: string]: unknown;
|
|
285
|
+
acceptance?: boolean | undefined;
|
|
286
|
+
review?: boolean | undefined;
|
|
287
|
+
}>> | undefined;
|
|
288
|
+
};
|
|
289
|
+
judge: {
|
|
290
|
+
adapter: string;
|
|
291
|
+
model: string;
|
|
292
|
+
};
|
|
293
|
+
review: {
|
|
294
|
+
complexityThreshold: number;
|
|
295
|
+
timeoutMs: number;
|
|
296
|
+
required: boolean;
|
|
297
|
+
prefer?: string[] | undefined;
|
|
298
|
+
policy?: "full" | "judge-only" | undefined;
|
|
299
|
+
criticalPaths?: string[] | undefined;
|
|
300
|
+
};
|
|
301
|
+
consult: {
|
|
302
|
+
adapter: string;
|
|
303
|
+
model: string;
|
|
304
|
+
stallMinutes: number;
|
|
305
|
+
prefer?: string[] | undefined;
|
|
306
|
+
};
|
|
307
|
+
visibility: {
|
|
308
|
+
llm: "pane" | "headless";
|
|
309
|
+
keepPanes: "run" | "attempt" | "forever";
|
|
310
|
+
worker: "interactive" | "print";
|
|
311
|
+
workersPerTab: number;
|
|
312
|
+
};
|
|
313
|
+
setup?: string | undefined;
|
|
314
|
+
cost?: {
|
|
315
|
+
models?: Record<string, {
|
|
316
|
+
inPerMtok: number;
|
|
317
|
+
outPerMtok: number;
|
|
318
|
+
cacheReadPerMtok?: number | undefined;
|
|
319
|
+
rateDate?: string | undefined;
|
|
320
|
+
}> | undefined;
|
|
321
|
+
subs?: Record<string, {
|
|
322
|
+
planMonthly: number;
|
|
323
|
+
windowsPerMonthLow: number;
|
|
324
|
+
windowsPerMonthHigh: number;
|
|
325
|
+
}> | undefined;
|
|
326
|
+
} | undefined;
|
|
327
|
+
scope?: {
|
|
328
|
+
allowDeviations?: string[] | undefined;
|
|
329
|
+
} | undefined;
|
|
330
|
+
}): string;
|
|
331
|
+
/** Repair is an actuator, but never a reason to run the paid diagnostic sensor. */
|
|
332
|
+
export declare function doctorFixOnly(cwd?: string): string;
|
|
99
333
|
export declare function doctor(_argv: string[], cwd?: string, adapters?: WorkerAdapter[], opts?: DoctorOpts): Promise<string>;
|
|
100
334
|
export {};
|