tickmarkr 2.0.0 → 2.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -1
- package/dist/cli/commands/doctor.d.ts +19 -0
- package/dist/cli/commands/doctor.js +59 -0
- package/dist/cli/commands/init.js +5 -2
- package/dist/cli/commands/resume.js +17 -6
- package/dist/cli/commands/run.js +5 -2
- package/dist/cli/index.d.ts +1 -1
- package/dist/cli/index.js +1 -1
- package/dist/config/config.d.ts +1 -0
- package/dist/config/config.js +2 -2
- package/dist/drivers/herdr.js +3 -1
- package/dist/drivers/index.d.ts +5 -1
- package/dist/drivers/index.js +16 -1
- package/dist/drivers/orca.d.ts +189 -0
- package/dist/drivers/orca.js +879 -0
- package/dist/tui/ink/init-app.js +14 -4
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +7 -0
package/README.md
CHANGED
|
@@ -13,7 +13,8 @@ tickmarkr is a spec-driven orchestration harness for AI coding agent CLIs. You w
|
|
|
13
13
|
acceptance criteria; the engine routes tasks to the best installed agent CLI (claude-code, codex,
|
|
14
14
|
cursor-agent, opencode, grok, pi, kimi) by cost and capability, dispatches work in git worktrees for
|
|
15
15
|
change isolation — as interactive TUIs when running under [herdr](https://herdr.dev), headless
|
|
16
|
-
subprocesses otherwise
|
|
16
|
+
subprocesses otherwise, or in [Orca](https://onorca.dev) terminals when you name that driver
|
|
17
|
+
yourself — and independently verifies each committed result by checking for no new
|
|
17
18
|
baseline failures per task, then strictly verifying the integration tip. Green tasks consolidate onto a
|
|
18
19
|
`tickmarkr/<runId>` branch; merging to your mainline is always your call, never automated. Engage
|
|
19
20
|
with full visibility into routing decisions, worker progress, and gate verdicts — or run headless
|
|
@@ -250,6 +251,23 @@ and first-attempt success rate. Cost reporting follows strict honesty rules and
|
|
|
250
251
|
When running under [herdr](https://herdr.dev), tickmarkr creates a labeled pane-and-tab workspace
|
|
251
252
|
for real-time visibility (optional — omit `--driver herdr` or run headless if preferred).
|
|
252
253
|
|
|
254
|
+
### Orca: an explicit-selection execution surface
|
|
255
|
+
|
|
256
|
+
[Orca](https://onorca.dev) is the third execution surface, and the only one you must ask for by
|
|
257
|
+
name: `--driver orca` or `driver: orca` in config. `--driver auto` never selects it — auto picks
|
|
258
|
+
herdr when a herdr session is live and subprocess otherwise — so Orca is never inherited from an
|
|
259
|
+
ambient environment variable, and an Orca that is installed but unreachable is not silently
|
|
260
|
+
downgraded to a hidden subprocess worker either. Naming it is the whole gate; its runtime failures
|
|
261
|
+
stay Orca's, reported as failures.
|
|
262
|
+
|
|
263
|
+
What Orca supplies is terminals. What tickmarkr keeps is everything that decides whether work
|
|
264
|
+
ships: **it creates and owns the git worktree** for every task (Orca is told which checkout to bind
|
|
265
|
+
its terminal to, and never makes one), **it runs the full gate battery** — build, test, lint,
|
|
266
|
+
evidence, scope, acceptance, review — against the commits that land there, and **it holds merge
|
|
267
|
+
authority**, consolidating only green tasks onto the run's `tickmarkr/<runId>` integration branch.
|
|
268
|
+
Orca is given no say over any of the three. Merging that branch to your mainline remains your call,
|
|
269
|
+
exactly as with every other driver.
|
|
270
|
+
|
|
253
271
|
tickmarkr borrows audit-firm vocabulary for its roles: **you** are the *Partner* (final sign-off),
|
|
254
272
|
workers are the *field team*, the acceptance judge is the *EQR* (engagement quality reviewer), and
|
|
255
273
|
the frontier-model consult is the *National Office*. The terms below use that vocabulary:
|
|
@@ -2,6 +2,7 @@ import { type ClaudeAlias } from "../../adapters/claude-code.js";
|
|
|
2
2
|
import type { WorkerAdapter } from "../../adapters/types.js";
|
|
3
3
|
import { type KimiDoctorTurnResult } from "../../adapters/kimi.js";
|
|
4
4
|
import { type CatalogReadResult } from "../../adapters/catalog-remote.js";
|
|
5
|
+
import { type ShResult } from "../../run/git.js";
|
|
5
6
|
/** Where a newer `table_<date>.csv` is discovered — the deployed site builds filenames by
|
|
6
7
|
* concatenation and publishes no index, so the release listing is the only enumerable surface. */
|
|
7
8
|
export declare const LIVEBENCH_RELEASES_URL = "https://api.github.com/repos/LiveBench/livebench.github.io/contents/public";
|
|
@@ -15,7 +16,24 @@ export type DoctorOpts = {
|
|
|
15
16
|
/** init's between-acts surface: status rows only — the model matrix and inline drift stay
|
|
16
17
|
* behind `tickmarkr doctor` (files are still written; only the RETURNED string shrinks). */
|
|
17
18
|
compact?: boolean;
|
|
19
|
+
/** Test seam for the same `orca status --json` transport production invokes. */
|
|
20
|
+
orcaStatusProbe?: (cwd: string, binary: string) => Promise<ShResult>;
|
|
21
|
+
/** Test seam for shell-path discovery; absence remains a normal doctor row, never an exception. */
|
|
22
|
+
resolveOrcaBinary?: (cwd: string) => string | undefined;
|
|
18
23
|
};
|
|
24
|
+
type OrcaCapability = {
|
|
25
|
+
verdict: "pass" | "fail";
|
|
26
|
+
detail: string;
|
|
27
|
+
};
|
|
28
|
+
/**
|
|
29
|
+
* Orca's status body is deliberately interpreted by T1's one shared envelope parser. Doctor owns
|
|
30
|
+
* only capability presentation: it may classify an absent executable, but it never invents a second
|
|
31
|
+
* permissive JSON reader for a malformed or refused status response.
|
|
32
|
+
*
|
|
33
|
+
* Capability-row `detail` stays hermetic — never `OrcaError.message`, which embeds volatile CLI
|
|
34
|
+
* stderr (Electron timestamps), so the row is byte-stable across runs.
|
|
35
|
+
*/
|
|
36
|
+
export declare function probeOrcaCapability(cwd: string, opts?: Pick<DoctorOpts, "orcaStatusProbe" | "resolveOrcaBinary">): Promise<OrcaCapability>;
|
|
19
37
|
export declare function runnerIgnoreFinding(cwd: string): {
|
|
20
38
|
verdict: "pass" | "warn";
|
|
21
39
|
detail: string;
|
|
@@ -71,3 +89,4 @@ export declare function selfShadowFinding(ownVersion: string, cwd?: string, reso
|
|
|
71
89
|
*/
|
|
72
90
|
export declare function liveBenchStalenessFinding(now: Date): string | undefined;
|
|
73
91
|
export declare function doctor(_argv: string[], cwd?: string, adapters?: WorkerAdapter[], opts?: DoctorOpts): Promise<string>;
|
|
92
|
+
export {};
|
|
@@ -6,14 +6,17 @@ import { detectPackageManager, turboContinueFindings } from "../../gates/baselin
|
|
|
6
6
|
import { version } from "./version.js";
|
|
7
7
|
import { allAdapters, binaryShadowWarnings, detectCandidateClis, flagDriftWarnings, modelAliasExclusions, modelAliasLine, probeAll, probeModels, resolveShellBinary, servableExclusions, servabilityLine, writeDoctor } from "../../adapters/registry.js";
|
|
8
8
|
import { CLAUDE_ALIAS_IDENTITY_STAMPS, claudeCode, resolveClaudeAliasIdentity } from "../../adapters/claude-code.js";
|
|
9
|
+
import { shq } from "../../adapters/types.js";
|
|
9
10
|
import { BANNER, compactTokens, dim, fail, kvRow, legend, ok, rule, statusRow, title } from "../../brand.js";
|
|
10
11
|
import { tickmarkrDir, stateDirName } from "../../graph/graph.js";
|
|
11
12
|
import { catalogModelAdvisory, catalogTierRanking, declaredModelWindow, hasWindowsConfig, modelLints, suggestOverlay, ttyVisual } from "../../adapters/model-lints.js";
|
|
12
13
|
import { loadConfig, overlayPreferShapes } from "../../config/config.js";
|
|
13
14
|
import { HerdrDriver } from "../../drivers/herdr.js";
|
|
15
|
+
import { parseEnvelope } from "../../drivers/orca.js";
|
|
14
16
|
import { kimi, probeKimiDoctorTurn } from "../../adapters/kimi.js";
|
|
15
17
|
import { denyPreferCollisionLine, denyPreferCollisions, disallowedBy, excludedChannels, exclusionLine, preferRanks } from "../../route/preference.js";
|
|
16
18
|
import { LIVEBENCH_TABLE_DATE, readCachedCatalog, refreshCatalogCommand } from "../../adapters/catalog-remote.js";
|
|
19
|
+
import { sh } from "../../run/git.js";
|
|
17
20
|
/** Where a newer `table_<date>.csv` is discovered — the deployed site builds filenames by
|
|
18
21
|
* concatenation and publishes no index, so the release listing is the only enumerable surface. */
|
|
19
22
|
export const LIVEBENCH_RELEASES_URL = "https://api.github.com/repos/LiveBench/livebench.github.io/contents/public";
|
|
@@ -21,6 +24,57 @@ export const LIVEBENCH_TABLE_MAX_AGE_DAYS = 90;
|
|
|
21
24
|
const visual = () => process.stdout.isTTY === true && process.env.NO_COLOR === undefined;
|
|
22
25
|
const alignedStatusRow = (verdict, key, value) => ` ${statusRow(verdict, kvRow(key, value).slice(2))}`;
|
|
23
26
|
const attentionRow = (text) => ` ${statusRow("warn", text)}`;
|
|
27
|
+
/**
|
|
28
|
+
* Orca's status body is deliberately interpreted by T1's one shared envelope parser. Doctor owns
|
|
29
|
+
* only capability presentation: it may classify an absent executable, but it never invents a second
|
|
30
|
+
* permissive JSON reader for a malformed or refused status response.
|
|
31
|
+
*
|
|
32
|
+
* Capability-row `detail` stays hermetic — never `OrcaError.message`, which embeds volatile CLI
|
|
33
|
+
* stderr (Electron timestamps), so the row is byte-stable across runs.
|
|
34
|
+
*/
|
|
35
|
+
export async function probeOrcaCapability(cwd, opts = {}) {
|
|
36
|
+
const binary = opts.resolveOrcaBinary ? opts.resolveOrcaBinary(cwd) : resolveShellBinary("orca", cwd).resolved;
|
|
37
|
+
if (!binary)
|
|
38
|
+
return { verdict: "fail", detail: "CLI not installed" };
|
|
39
|
+
let response;
|
|
40
|
+
try {
|
|
41
|
+
response = await (opts.orcaStatusProbe ?? ((probeCwd, executable) => sh(`${shq(executable)} status --json`, probeCwd, 10_000)))(cwd, binary);
|
|
42
|
+
}
|
|
43
|
+
catch {
|
|
44
|
+
// The seam (or the shell) never reaching a verdict is a failed probe, never an exception out of
|
|
45
|
+
// doctor: an absent or sick runtime must still render its row.
|
|
46
|
+
return { verdict: "fail", detail: "CLI installed but runtime probe failed" };
|
|
47
|
+
}
|
|
48
|
+
try {
|
|
49
|
+
// Do this before accepting the process result: Orca's documented refused transport is rc=1
|
|
50
|
+
// with an ok:false envelope on stdout, and parseEnvelope preserves that refusal fail-closed.
|
|
51
|
+
const envelope = parseEnvelope("status", response.stdout);
|
|
52
|
+
if (response.code !== 0 || response.timedOut) {
|
|
53
|
+
return {
|
|
54
|
+
verdict: "fail",
|
|
55
|
+
detail: `CLI installed but runtime probe failed — orca status exited ${response.code}${response.timedOut ? " after timeout" : ""}`,
|
|
56
|
+
};
|
|
57
|
+
}
|
|
58
|
+
const runtime = envelope.result.runtime;
|
|
59
|
+
const reachable = typeof runtime === "object" && runtime !== null && !Array.isArray(runtime)
|
|
60
|
+
? runtime.reachable
|
|
61
|
+
: undefined;
|
|
62
|
+
if (reachable === true)
|
|
63
|
+
return { verdict: "pass", detail: `runtime reachable (${envelope.runtimeId})` };
|
|
64
|
+
if (reachable === false)
|
|
65
|
+
return { verdict: "fail", detail: "CLI installed but runtime unreachable" };
|
|
66
|
+
return { verdict: "fail", detail: "CLI installed but runtime probe failed — status carries no reachability proof" };
|
|
67
|
+
}
|
|
68
|
+
catch {
|
|
69
|
+
// Only the shared parser reads this body. Every shape it rejects — unparseable, non-object,
|
|
70
|
+
// ok:false refusal, no result, or absent/`none` `_meta.runtimeId` — is a FAILED probe, never
|
|
71
|
+
// an unreachable-runtime claim: a doctor that re-read those bytes with its own permissive
|
|
72
|
+
// reader would call malformed metadata "installed but unreachable". Unreachable is proven only
|
|
73
|
+
// by a well-formed envelope that says `reachable:false` (the shape a live orca CLI emits with
|
|
74
|
+
// its runtime down — tests/helpers/fake-orca.ts).
|
|
75
|
+
return { verdict: "fail", detail: "CLI installed but runtime probe failed" };
|
|
76
|
+
}
|
|
77
|
+
}
|
|
24
78
|
function detectRunner(cwd) {
|
|
25
79
|
const readIf = (p) => (existsSync(join(cwd, p)) ? readFileSync(join(cwd, p), "utf8") : "");
|
|
26
80
|
let pkg = {};
|
|
@@ -373,6 +427,11 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
|
|
|
373
427
|
}
|
|
374
428
|
if (trustNa.length)
|
|
375
429
|
rows.push(` ${dim("=")} ${dim(`n/a (${trustNa.length}): ${trustNa.join(", ")}`)}`);
|
|
430
|
+
// Orca is an explicit choice, so its health is capability information rather than an auto-routing
|
|
431
|
+
// input. A failed probe never changes pickDriver's auto ordering or substitutes subprocess.
|
|
432
|
+
const orca = await probeOrcaCapability(cwd, opts);
|
|
433
|
+
rows.push(legend("execution runtime:"));
|
|
434
|
+
rows.push(alignedStatusRow(orca.verdict, "orca", orca.detail));
|
|
376
435
|
for (const [role, sel] of [["judge", cfg.judge], ["consult", cfg.consult]]) {
|
|
377
436
|
if (!health[sel.adapter]?.installed) {
|
|
378
437
|
rows.push(attentionRow(`${role} runs on ${sel.adapter}:${sel.model} — NOT installed; that gate will fail closed until you install it or remap cfg.${role}`));
|
|
@@ -12,11 +12,14 @@ import { Journal } from "../../run/journal.js";
|
|
|
12
12
|
import { doctor } from "./doctor.js";
|
|
13
13
|
import { assembleFleetEditor } from "./fleet.js";
|
|
14
14
|
const SCAFFOLD_SPEC = "tickmarkr.spec.md";
|
|
15
|
-
// Operator-approved (2026-07-17) environments footer —
|
|
16
|
-
//
|
|
15
|
+
// Operator-approved (2026-07-17) environments footer — no npm install for herdr (npm package
|
|
16
|
+
// "herdr" is a reserved 0.0.0 placeholder as of that date). The orca row joins it with the v2.1
|
|
17
|
+
// driver: it is a THIRD execution surface an operator selects outright — `auto` still resolves
|
|
18
|
+
// herdr-else-subprocess and never picks it, so the footer names it beside the other two choices.
|
|
17
19
|
const ENVIRONMENTS_FOOTER = [
|
|
18
20
|
"environments:",
|
|
19
21
|
" herdr — the full cockpit — every worker, judge, and consult is a visible pane you can watch and unblock · https://herdr.dev",
|
|
22
|
+
" orca — visible terminals in the Orca app — an explicit driver choice: set driver: orca (auto never picks it) · https://onorca.dev",
|
|
20
23
|
" claude code — tickmarkr init --agent installs the /tkr skills + AGENTS.md so Claude Code (or any agent CLI) drives the loop natively",
|
|
21
24
|
" anywhere — no herdr? same fail-closed gates, headless subprocess driver",
|
|
22
25
|
].join("\n");
|
|
@@ -1,5 +1,6 @@
|
|
|
1
|
+
import { parseArgs } from "node:util";
|
|
1
2
|
import { loadConfig } from "../../config/config.js";
|
|
2
|
-
import { pickDriver } from "../../drivers/index.js";
|
|
3
|
+
import { parseDriverOverride, pickDriver } from "../../drivers/index.js";
|
|
3
4
|
import { loadGraph } from "../../graph/graph.js";
|
|
4
5
|
import { formatSummary, runDaemon } from "../../run/daemon.js";
|
|
5
6
|
import { denyPreferCollisionLine, denyPreferCollisions } from "../../route/preference.js";
|
|
@@ -7,14 +8,24 @@ import { narrationSink, bindNarration } from "./run.js";
|
|
|
7
8
|
const summaryGreen = (s) => s.failed.length === 0 && s.human.length === 0 && s.blocked.length === 0 && s.pending.length === 0
|
|
8
9
|
&& s.tipVerify !== "failed";
|
|
9
10
|
export async function resume(argv, cwd = process.cwd()) {
|
|
10
|
-
const
|
|
11
|
+
const { values, positionals } = parseArgs({
|
|
12
|
+
args: argv,
|
|
13
|
+
options: {
|
|
14
|
+
"graph-changed": { type: "boolean" },
|
|
15
|
+
"retry-failed": { type: "boolean" },
|
|
16
|
+
driver: { type: "string" },
|
|
17
|
+
},
|
|
18
|
+
allowPositionals: true,
|
|
19
|
+
});
|
|
20
|
+
const runId = positionals[0];
|
|
11
21
|
if (!runId)
|
|
12
|
-
throw new Error("usage: tickmarkr resume <run-id> [--graph-changed] [--retry-failed]");
|
|
22
|
+
throw new Error("usage: tickmarkr resume <run-id> [--graph-changed] [--retry-failed] [--driver <auto|herdr|subprocess|orca>]");
|
|
23
|
+
const driverOverride = parseDriverOverride(values.driver);
|
|
13
24
|
// T3: --graph-changed is the operator's audited release of the engagement-identity guard (Sol #2 /
|
|
14
25
|
// Fable F2) — the daemon refuses a mismatched/unbound journal unless this is set, then journals a
|
|
15
26
|
// graph-rehash event naming both hashes. Strip the flag before runId resolution so a bare id still wins.
|
|
16
|
-
const graphChanged =
|
|
17
|
-
const retryFailed =
|
|
27
|
+
const graphChanged = values["graph-changed"] ?? false;
|
|
28
|
+
const retryFailed = values["retry-failed"] ?? false;
|
|
18
29
|
const cfg = loadConfig(cwd);
|
|
19
30
|
// v1.87 T3 (OBS-162, twice-carried workaround): the preflight runs AFTER the graph is read and
|
|
20
31
|
// sees only the shapes the resumed graph carries. A deny∩prefer collision on a shape no resumed
|
|
@@ -32,7 +43,7 @@ export async function resume(argv, cwd = process.cwd()) {
|
|
|
32
43
|
graphChanged,
|
|
33
44
|
retryFailed,
|
|
34
45
|
// bound to the same sink the daemon gets, so a driver-journaled recovery reaches this rail too
|
|
35
|
-
driver: bindNarration(pickDriver(cfg), narrate),
|
|
46
|
+
driver: bindNarration(pickDriver(cfg, driverOverride), narrate),
|
|
36
47
|
// v1.99 T2: the ONE narration sink — the quiet rail on a TTY, the raw journal formatter on a
|
|
37
48
|
// pipe. A resumed run meets the same surface a fresh one does; printing the raw formatter here
|
|
38
49
|
// would leave `resume` as the last place the old unfiltered dump survives. Bound to the run id
|
package/dist/cli/commands/run.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { parseArgs } from "node:util";
|
|
2
2
|
import { allAdapters, discoverChannels, probeAll, readDoctor } from "../../adapters/registry.js";
|
|
3
3
|
import { ROUTING_MODES } from "../../config/config.js";
|
|
4
|
-
import { pickDriver } from "../../drivers/index.js";
|
|
4
|
+
import { parseDriverOverride, pickDriver } from "../../drivers/index.js";
|
|
5
5
|
import { loadGraph } from "../../graph/graph.js";
|
|
6
6
|
import { formatSummary, resolveRunMode, runDaemon } from "../../run/daemon.js";
|
|
7
7
|
import { isRunLockLive } from "../../run/lock.js";
|
|
@@ -393,6 +393,9 @@ export async function run(argv, cwd = process.cwd()) {
|
|
|
393
393
|
if (!Number.isInteger(n) || n <= 0)
|
|
394
394
|
throw new Error(`--concurrency must be a positive integer (got ${values.concurrency})`);
|
|
395
395
|
}
|
|
396
|
+
// Keep invalid input at the argv boundary. In particular, do not cast a string into the closed
|
|
397
|
+
// driver union and accidentally turn an unknown explicit choice into auto-selection.
|
|
398
|
+
const driverOverride = parseDriverOverride(values.driver);
|
|
396
399
|
if (values.quality && values.mode !== undefined) {
|
|
397
400
|
throw new Error("--quality is a compatibility alias for --mode partner-led and cannot be combined with an explicit --mode — pass one or the other");
|
|
398
401
|
}
|
|
@@ -443,7 +446,7 @@ export async function run(argv, cwd = process.cwd()) {
|
|
|
443
446
|
const s = await runDaemon(cwd, {
|
|
444
447
|
runId,
|
|
445
448
|
concurrency: values.concurrency ? Number(values.concurrency) : undefined,
|
|
446
|
-
driver: bindNarration(pickDriver(cfg,
|
|
449
|
+
driver: bindNarration(pickDriver(cfg, driverOverride), narrate),
|
|
447
450
|
mode: flagMode,
|
|
448
451
|
supersedes: values.supersedes,
|
|
449
452
|
narrate,
|
package/dist/cli/index.d.ts
CHANGED
|
@@ -5,7 +5,7 @@ export type CommandResult = string | {
|
|
|
5
5
|
};
|
|
6
6
|
export type CommandMap = Record<string, (argv: string[]) => Promise<CommandResult>>;
|
|
7
7
|
export declare const COMMANDS: CommandMap;
|
|
8
|
-
export declare const USAGE = "tickmarkr \u2014 spec-driven orchestration harness for AI coding agents\nusage: tickmarkr <command>\n init guided setup + doctor; init --agent [--force] [--docs] adds agent skills/docs\n doctor re-probe adapters, herdr, auth; print capability matrix (--fix writes the test-runner ignore when a safe edit exists)\n fleet interactive fleet editor (fleet --print for CI drift checks)\n compile <src> spec \u2192 .tickmarkr/graph.json (fails without acceptance criteria)\n scope <intent> draft a compiled native spec beside an answered intent (--force to overwrite)\n plan dry-run routing table + cost estimate + floor lints\n eval run checked-in fixtures against every channel in isolated temp repos\n run execute the graph (--concurrency N --driver herdr|subprocess --route-strict)\n status live run state\n verify run the gate battery standalone against merge-base(--base, HEAD)..HEAD \u2014 no daemon, one verdict (--base main --criteria <file> | --task <id> [--files <glob>] [--author adapter:model] [--no-review] [--json])\n resume <id> continue a run from its journal\n report <id> cost/quality report (--md for committable execution record)\n profile show learned routing profile (profile reset = forget history via cursor, keeps telemetry)\n ui open the Fleet Studio TUI (full-screen tabbed cockpit)\n unlock remove a stale/garbage run lock (refuses if the holder is alive)\n beat <tier> record one supervision beat for orchestrator|overseer|watch (--stand-down to hand off); a supervising seat's own watcher loop calls it, and status reads the tier STALE once the beats stop\n approve <id> <task> release a park (--uphold sides with the reviewer and funds a fixed attempt; --by <name> --reason <text>); takes effect on resume";
|
|
8
|
+
export declare const USAGE = "tickmarkr \u2014 spec-driven orchestration harness for AI coding agents\nusage: tickmarkr <command>\n init guided setup + doctor; init --agent [--force] [--docs] adds agent skills/docs\n doctor re-probe adapters, herdr, auth; print capability matrix (--fix writes the test-runner ignore when a safe edit exists)\n fleet interactive fleet editor (fleet --print for CI drift checks)\n compile <src> spec \u2192 .tickmarkr/graph.json (fails without acceptance criteria)\n scope <intent> draft a compiled native spec beside an answered intent (--force to overwrite)\n plan dry-run routing table + cost estimate + floor lints\n eval run checked-in fixtures against every channel in isolated temp repos\n run execute the graph (--concurrency N --driver auto|herdr|subprocess|orca --route-strict; orca runs only when named)\n status live run state\n verify run the gate battery standalone against merge-base(--base, HEAD)..HEAD \u2014 no daemon, one verdict (--base main --criteria <file> | --task <id> [--files <glob>] [--author adapter:model] [--no-review] [--json])\n resume <id> continue a run from its journal\n report <id> cost/quality report (--md for committable execution record)\n profile show learned routing profile (profile reset = forget history via cursor, keeps telemetry)\n ui open the Fleet Studio TUI (full-screen tabbed cockpit)\n unlock remove a stale/garbage run lock (refuses if the holder is alive)\n beat <tier> record one supervision beat for orchestrator|overseer|watch (--stand-down to hand off); a supervising seat's own watcher loop calls it, and status reads the tier STALE once the beats stop\n approve <id> <task> release a park (--uphold sides with the reviewer and funds a fixed attempt; --by <name> --reason <text>); takes effect on resume";
|
|
9
9
|
export declare function dispatch(cmd: string | undefined, argv: string[], commands?: CommandMap): Promise<{
|
|
10
10
|
out: string;
|
|
11
11
|
code: number;
|
package/dist/cli/index.js
CHANGED
|
@@ -35,7 +35,7 @@ usage: tickmarkr <command>
|
|
|
35
35
|
scope <intent> draft a compiled native spec beside an answered intent (--force to overwrite)
|
|
36
36
|
plan dry-run routing table + cost estimate + floor lints
|
|
37
37
|
eval run checked-in fixtures against every channel in isolated temp repos
|
|
38
|
-
run execute the graph (--concurrency N --driver herdr|subprocess --route-strict)
|
|
38
|
+
run execute the graph (--concurrency N --driver auto|herdr|subprocess|orca --route-strict; orca runs only when named)
|
|
39
39
|
status live run state
|
|
40
40
|
verify run the gate battery standalone against merge-base(--base, HEAD)..HEAD — no daemon, one verdict (--base main --criteria <file> | --task <id> [--files <glob>] [--author adapter:model] [--no-review] [--json])
|
|
41
41
|
resume <id> continue a run from its journal
|
package/dist/config/config.d.ts
CHANGED
package/dist/config/config.js
CHANGED
|
@@ -266,7 +266,7 @@ const ShapeGateParticipationSchema = z
|
|
|
266
266
|
});
|
|
267
267
|
export const TickmarkrConfigSchema = z.object({
|
|
268
268
|
concurrency: z.number().int().positive(),
|
|
269
|
-
driver: z.enum(["auto", "herdr", "subprocess"]),
|
|
269
|
+
driver: z.enum(["auto", "herdr", "subprocess", "orca"]),
|
|
270
270
|
integrationBranchPrefix: z
|
|
271
271
|
.string()
|
|
272
272
|
.regex(/^[A-Za-z0-9][A-Za-z0-9._/-]*$/, "must be branch-safe (letters/digits/._/-, no spaces or shell metacharacters)")
|
|
@@ -713,7 +713,7 @@ export function overlayBytesLoadError(repoRoot, bytes, opts = {}) {
|
|
|
713
713
|
export function configTemplate(overlay) {
|
|
714
714
|
const base = `# tickmarkr config overlay — merges over built-in defaults (repo beats global beats defaults)
|
|
715
715
|
# concurrency: 3
|
|
716
|
-
# driver: auto # auto | herdr | subprocess
|
|
716
|
+
# driver: auto # auto | herdr | subprocess | orca
|
|
717
717
|
# taskTimeoutMinutes: 30
|
|
718
718
|
# contextWarnTokens: 170000 # v1.23: journal+notify once per attempt when live worker context crosses this (status shows the sample)
|
|
719
719
|
# setup: npm ci --prefer-offline # run in each fresh task worktree before dispatch
|
package/dist/drivers/herdr.js
CHANGED
|
@@ -1114,9 +1114,11 @@ export class HerdrDriver {
|
|
|
1114
1114
|
// reported the stack. The documented `changed` flag is the verification; anything else — a
|
|
1115
1115
|
// nonzero exit, `changed:false`, an unparseable result — fails closed.
|
|
1116
1116
|
const swapped = await this.herdr(`pane swap --source-pane ${shq(pane)} --target-pane ${shq(this.callerPane)}`);
|
|
1117
|
+
// Flag lives at `result.swap.changed` (verbatim 0.8.0); see herdr-swap-shape.test.ts.
|
|
1117
1118
|
let swapChanged;
|
|
1118
1119
|
try {
|
|
1119
|
-
|
|
1120
|
+
const result = JSON.parse(swapped.stdout).result;
|
|
1121
|
+
swapChanged = result?.swap?.changed ?? result?.changed;
|
|
1120
1122
|
}
|
|
1121
1123
|
catch {
|
|
1122
1124
|
/* fail closed below */
|
package/dist/drivers/index.d.ts
CHANGED
|
@@ -1,3 +1,7 @@
|
|
|
1
1
|
import type { TickmarkrConfig } from "../config/config.js";
|
|
2
2
|
import type { ExecutorDriver } from "./types.js";
|
|
3
|
-
export declare
|
|
3
|
+
export declare const DRIVER_CHOICES: readonly ["auto", "herdr", "subprocess", "orca"];
|
|
4
|
+
export type DriverChoice = (typeof DRIVER_CHOICES)[number];
|
|
5
|
+
/** Validate argv at the CLI boundary rather than casting an arbitrary string into a driver choice. */
|
|
6
|
+
export declare function parseDriverOverride(override?: string): DriverChoice | undefined;
|
|
7
|
+
export declare function pickDriver(cfg: TickmarkrConfig, override?: string): ExecutorDriver;
|
package/dist/drivers/index.js
CHANGED
|
@@ -1,7 +1,18 @@
|
|
|
1
1
|
import { HerdrDriver } from "./herdr.js";
|
|
2
|
+
import { OrcaDriver } from "./orca.js";
|
|
2
3
|
import { SubprocessDriver } from "./subprocess.js";
|
|
4
|
+
export const DRIVER_CHOICES = ["auto", "herdr", "subprocess", "orca"];
|
|
5
|
+
/** Validate argv at the CLI boundary rather than casting an arbitrary string into a driver choice. */
|
|
6
|
+
export function parseDriverOverride(override) {
|
|
7
|
+
if (override === undefined)
|
|
8
|
+
return undefined;
|
|
9
|
+
for (const choice of DRIVER_CHOICES)
|
|
10
|
+
if (override === choice)
|
|
11
|
+
return choice;
|
|
12
|
+
throw new Error(`usage: --driver must be one of ${DRIVER_CHOICES.join(" | ")} (got ${override})`);
|
|
13
|
+
}
|
|
3
14
|
export function pickDriver(cfg, override) {
|
|
4
|
-
const want = override ?? cfg.driver;
|
|
15
|
+
const want = parseDriverOverride(override) ?? cfg.driver;
|
|
5
16
|
// VIS-09 item 2: plumb the per-tab cap into the HerdrDriver — the driver takes it as a constructor
|
|
6
17
|
// param and never imports config (cfg is the only seam). Guaranteed present: DEFAULT_CONFIG seeds
|
|
7
18
|
// workersPerTab:3 and deepMerge overlays on top, so a missing overlay key still resolves.
|
|
@@ -9,5 +20,9 @@ export function pickDriver(cfg, override) {
|
|
|
9
20
|
return new HerdrDriver("herdr", cfg.visibility.workersPerTab);
|
|
10
21
|
if (want === "subprocess")
|
|
11
22
|
return new SubprocessDriver();
|
|
23
|
+
// Orca is an operator-selected execution surface. Its runtime failure stays on Orca; selection
|
|
24
|
+
// must never substitute a hidden subprocess worker after this explicit choice.
|
|
25
|
+
if (want === "orca")
|
|
26
|
+
return new OrcaDriver();
|
|
12
27
|
return HerdrDriver.available() ? new HerdrDriver("herdr", cfg.visibility.workersPerTab) : new SubprocessDriver();
|
|
13
28
|
}
|
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
import { type ShResult } from "../run/git.js";
|
|
2
|
+
import { type ExecutorDriver, type NotifyOpts, type Slot, type SlotOpts } from "./types.js";
|
|
3
|
+
/** The response families the ONE shared envelope parser serves. There is no second JSON seam. */
|
|
4
|
+
export declare const ORCA_RESPONSE_FAMILIES: readonly ["status", "create", "list", "read", "send", "wait", "show", "close"];
|
|
5
|
+
export type OrcaFamily = (typeof ORCA_RESPONSE_FAMILIES)[number];
|
|
6
|
+
export declare const STALE_HANDLE_CODE = "terminal_handle_stale";
|
|
7
|
+
export declare const NOT_WRITABLE_CODE = "terminal_not_writable";
|
|
8
|
+
/** The ONLY terminal status that licenses reading a terminal's bytes or its agent state. */
|
|
9
|
+
export declare const RUNNING_STATUS = "running";
|
|
10
|
+
export declare const STATUS_GOVERNED_METHODS: readonly ["read", "waitOutput", "status", "waitAgentStatus"];
|
|
11
|
+
export interface OrcaExec {
|
|
12
|
+
(args: string[], cwd: string, timeoutMs?: number): Promise<ShResult>;
|
|
13
|
+
}
|
|
14
|
+
export interface OrcaTimeSource {
|
|
15
|
+
now: () => number;
|
|
16
|
+
sleep: (ms: number) => Promise<void>;
|
|
17
|
+
}
|
|
18
|
+
/** Every failure this driver produces is explicit and carries the raw bytes that produced it. */
|
|
19
|
+
export declare class OrcaError extends Error {
|
|
20
|
+
readonly family: string;
|
|
21
|
+
readonly reason: string;
|
|
22
|
+
readonly raw: string;
|
|
23
|
+
readonly code?: string;
|
|
24
|
+
readonly runtimeId?: string;
|
|
25
|
+
constructor(family: string, reason: string, raw: string, opts?: {
|
|
26
|
+
code?: string;
|
|
27
|
+
runtimeId?: string;
|
|
28
|
+
});
|
|
29
|
+
}
|
|
30
|
+
/** The slot cannot be addressed: dead/unknown terminal record, or a handle that cannot be recovered
|
|
31
|
+
* to exactly one owned terminal in the slot's own worktree. Never a silent false or empty string. */
|
|
32
|
+
export declare class OrcaUnavailableError extends OrcaError {
|
|
33
|
+
readonly terminalStatus?: string | undefined;
|
|
34
|
+
constructor(family: string, reason: string, raw: string, terminalStatus?: string | undefined);
|
|
35
|
+
}
|
|
36
|
+
export interface OrcaEnvelope {
|
|
37
|
+
result: Record<string, unknown>;
|
|
38
|
+
runtimeId: string;
|
|
39
|
+
raw: string;
|
|
40
|
+
}
|
|
41
|
+
/**
|
|
42
|
+
* The one JSON seam. Fails CLOSED on every degenerate response — empty, unparseable (a truncated
|
|
43
|
+
* body lands here), non-object, no boolean `ok`, `ok:false`, `ok:true` with no result object, or a
|
|
44
|
+
* successful response without a usable `_meta.runtimeId` —
|
|
45
|
+
* and preserves the raw bytes on the thrown error for diagnostics. Callers never see a partial
|
|
46
|
+
* envelope, so no caller can reinterpret a parse failure as empty output, an unknown-but-successful
|
|
47
|
+
* status, or a successful close.
|
|
48
|
+
*/
|
|
49
|
+
export declare function parseEnvelope(family: OrcaFamily, stdout: string, raw?: string): OrcaEnvelope;
|
|
50
|
+
/** The worktree a terminal record binds to. The spike pinned the record's shape but not this key's
|
|
51
|
+
* spelling, so the known aliases are accepted and nothing else — a record with none is unbound,
|
|
52
|
+
* which fails every identity comparison below rather than passing one by default. */
|
|
53
|
+
export declare function terminalWorktree(term: Record<string, unknown>): string | undefined;
|
|
54
|
+
/**
|
|
55
|
+
* The same checkout under two spellings. git hands tickmarkr one (`/tmp/...` on darwin, or anything
|
|
56
|
+
* below a symlinked parent) while Orca answers the canonicalized one (`/private/tmp/...`), and
|
|
57
|
+
* `resolve()` collapses `..` but never a symlink — so string equality on resolved paths reports two
|
|
58
|
+
* different checkouts and leaves a perfectly valid slot unreacquirable after a runtime restart.
|
|
59
|
+
* Identity is FILESYSTEM identity. A path that does not exist has no filesystem identity to read, so
|
|
60
|
+
* it keeps its resolved spelling: deterministic, and still comparable to another spelling of itself.
|
|
61
|
+
*/
|
|
62
|
+
export declare function canonicalWorktreePath(path: string): string;
|
|
63
|
+
/** Conservative agent-state mapping over orca's ACTUAL surfaces: `blocked` only when the show
|
|
64
|
+
* record reports agentWait:true, `idle` only when the `terminal wait --for tui-idle` condition is
|
|
65
|
+
* satisfied. The recorded 1.4.186 show response carries NO agent field at all — an absent signal
|
|
66
|
+
* is "unknown", never a fabricated definite status. */
|
|
67
|
+
export declare function mapAgentState(term: Record<string, unknown>, tuiIdle: boolean): string;
|
|
68
|
+
/**
|
|
69
|
+
* The renderer hard-wraps long lines, paints margin chrome, and a cursor page boundary splits a
|
|
70
|
+
* marker exactly like a wrap does. `parseWorkerResult` (src/adapters/prompt.ts) already de-wraps
|
|
71
|
+
* trailers this way, so marker matching gets the same joined view beside the raw one.
|
|
72
|
+
* ponytail: joining every line can in principle glue two unrelated lines into a marker — the same
|
|
73
|
+
* tolerance parseWorkerResult has carried since v1.2; raw is matched first, so an unwrapped hit
|
|
74
|
+
* never depends on this.
|
|
75
|
+
*/
|
|
76
|
+
export declare function joinWrapped(raw: string): string;
|
|
77
|
+
export interface OrcaDriverOpts {
|
|
78
|
+
bin?: string;
|
|
79
|
+
exec?: OrcaExec;
|
|
80
|
+
time?: OrcaTimeSource;
|
|
81
|
+
pageLines?: number;
|
|
82
|
+
pollMs?: number;
|
|
83
|
+
/** Bounded, seam-adjustable staleness window for runtime probes before mutations. */
|
|
84
|
+
probeStalenessMs?: number;
|
|
85
|
+
}
|
|
86
|
+
export declare class OrcaDriver implements ExecutorDriver {
|
|
87
|
+
id: string;
|
|
88
|
+
interactive: boolean;
|
|
89
|
+
private slots;
|
|
90
|
+
private n;
|
|
91
|
+
private bin;
|
|
92
|
+
private exec;
|
|
93
|
+
private time;
|
|
94
|
+
private pageLines;
|
|
95
|
+
private pollMs;
|
|
96
|
+
private probeStalenessMs;
|
|
97
|
+
constructor(opts?: OrcaDriverOpts);
|
|
98
|
+
private call;
|
|
99
|
+
/** The live runtime's identity, or an explicit failure. A missing or unreachable runtime is a
|
|
100
|
+
* driver-level failure carrying the raw refusal — never a reachable-looking default. */
|
|
101
|
+
private runtimeEnv;
|
|
102
|
+
/** Explicit runtime probe. Also T3's doctor probe. */
|
|
103
|
+
probeRuntime(cwd?: string): Promise<string>;
|
|
104
|
+
slot(cwd: string, name: string, opts?: SlotOpts): Promise<Slot>;
|
|
105
|
+
/** Where to invoke the CLI for this slot's calls (see OrcaSlotState.dir). */
|
|
106
|
+
private cliCwd;
|
|
107
|
+
private state;
|
|
108
|
+
private latched;
|
|
109
|
+
private assertAvailable;
|
|
110
|
+
run(slot: Slot, cmd: string): Promise<void>;
|
|
111
|
+
private create;
|
|
112
|
+
/**
|
|
113
|
+
* Every terminal-addressed call — read AND write — goes through here, and the runtime identity is
|
|
114
|
+
* established BEFORE the runtime-scoped handle goes on the wire. Discarding a lookalike's answer
|
|
115
|
+
* after reading it is still having addressed it, so the probe comes first; the post-call check
|
|
116
|
+
* only closes the narrow race of a restart landing between probe and call. Same for an explicit
|
|
117
|
+
* `terminal_handle_stale`. Either way the driver relists the slot's exact worktree and replaces
|
|
118
|
+
* the handle exactly once, then re-issues the operation against the replacement.
|
|
119
|
+
*/
|
|
120
|
+
private terminalOp;
|
|
121
|
+
private recover;
|
|
122
|
+
/** Validated READ terminal record, or an explicit unavailable failure. Called BEFORE any caller
|
|
123
|
+
* looks at tail bytes — on every page, on every read-governed method. Read records are the one
|
|
124
|
+
* place orca reports a literal `status` (recorded: "running" live, "exited" on the dead record). */
|
|
125
|
+
private validated;
|
|
126
|
+
/** Validated SHOW terminal record, or an explicit unavailable failure. The recorded 1.4.186 show
|
|
127
|
+
* response reports liveness through connected/orphaned and carries NO status and NO agent field,
|
|
128
|
+
* so this is the status discipline's show leg: a terminal that cannot prove connected-and-not-
|
|
129
|
+
* orphaned is unavailable for state questions, exactly as a non-running read record is for bytes. */
|
|
130
|
+
private liveShowTerm;
|
|
131
|
+
private tailText;
|
|
132
|
+
private readPage;
|
|
133
|
+
/** A single UNPAGED tail read — exactly what the caller asked for and nothing more. Markers split
|
|
134
|
+
* across cursor pages are not reassembled here; that is waitOutput's job. */
|
|
135
|
+
read(slot: Slot, lines: number): Promise<string>;
|
|
136
|
+
/**
|
|
137
|
+
* Bounded cursor-paged sweep into the slot's accumulated buffer. The first read of a slot carries
|
|
138
|
+
* no cursor: it is the ANCHOR, whose `oldestCursor` says where the retained buffer starts (its own
|
|
139
|
+
* tail is the newest lines, not the oldest, so it is not appended). Every page after it appends,
|
|
140
|
+
* and every one of them — anchor included — is status-validated before a single byte is matched.
|
|
141
|
+
*/
|
|
142
|
+
private sweep;
|
|
143
|
+
waitOutput(slot: Slot, pattern: string, timeoutMs: number, opts?: {
|
|
144
|
+
regex?: boolean;
|
|
145
|
+
}): Promise<boolean>;
|
|
146
|
+
status(slot: Slot): Promise<string>;
|
|
147
|
+
/** One `terminal wait` through the full identity machinery. The recorded 1.4.186 elapsed answer
|
|
148
|
+
* is rc 1 + ok:true + {handle, condition, satisfied:false, status:"running"}; it is "not yet"
|
|
149
|
+
* only after this method validates all four fields. Any malformed/refused wait remains explicit. */
|
|
150
|
+
private waitCondition;
|
|
151
|
+
waitAgentStatus(slot: Slot, status: string, timeoutMs: number): Promise<boolean>;
|
|
152
|
+
notify(msg: string, opts?: NotifyOpts): Promise<void>;
|
|
153
|
+
close(slot: Slot): Promise<void>;
|
|
154
|
+
/**
|
|
155
|
+
* The one destructive call in this driver, for a slot's own terminal AND for a reconcile candidate
|
|
156
|
+
* alike. It goes through terminalOp deliberately: the live runtime identity is proven immediately
|
|
157
|
+
* BEFORE the handle goes on the wire, and a runtime that changed does not merely fail the close —
|
|
158
|
+
* the handle is DISCARDED and re-derived from the owned tab title in that exact checkout, where a
|
|
159
|
+
* handle value the new runtime happened to reissue to somebody else's terminal is refused by
|
|
160
|
+
* construction (recover()). Checking identity on the receipt afterwards could not undo a close.
|
|
161
|
+
*/
|
|
162
|
+
private closeTerminal;
|
|
163
|
+
/**
|
|
164
|
+
* The WHOLE terminal table. `terminal list` caps rows at its own default and says so through
|
|
165
|
+
* `truncated`/`totalCount`; a capped listing is not an ownership snapshot, because the row it
|
|
166
|
+
* dropped is precisely the older run's leftover no later sweep would ever see again.
|
|
167
|
+
* ponytail: two asks, not a paging loop — `--limit` takes the whole table in one go, and a runtime
|
|
168
|
+
* that still reports truncated at totalCount rows is a listing this sweep declines to judge on.
|
|
169
|
+
*/
|
|
170
|
+
private listAll;
|
|
171
|
+
/**
|
|
172
|
+
* Sweep tickmarkr-owned terminals down to `desired`. Ownership is decided ONLY by parseOwnedName
|
|
173
|
+
* over the owned TAB title, through the same panesToClose fold herdr uses (drivers/types.ts): an
|
|
174
|
+
* owned-and-undesired terminal closes whichever run — and whichever daemon — created it, and a
|
|
175
|
+
* title that does not parse is never a candidate however much it resembles one.
|
|
176
|
+
*
|
|
177
|
+
* The listing is UNSCOPED and layout-bearing. Unscoped because an older run's leftover sits in a
|
|
178
|
+
* checkout this run never knew, so a `--worktree`-filtered sweep is exactly how such a leftover
|
|
179
|
+
* survives forever. Layout-bearing because the owned title survives at TAB identity only — a list
|
|
180
|
+
* row's `title` is the shell-controlled pane title, and closing on that is how a foreign pane that
|
|
181
|
+
* happens to be running an owned-looking command gets killed.
|
|
182
|
+
*
|
|
183
|
+
* Cosmetic by contract: every failure is swallowed, per candidate and overall.
|
|
184
|
+
*/
|
|
185
|
+
reconcile(desired: Set<string>, runId: string, opts?: {
|
|
186
|
+
spareLiveLlm?: boolean;
|
|
187
|
+
}): Promise<void>;
|
|
188
|
+
worktree(repo: string, branch: string, baseRef: string): Promise<string>;
|
|
189
|
+
}
|