tickmarkr 2.2.1 → 2.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -9
- package/dist/adapters/types.d.ts +20 -1
- package/dist/adapters/types.js +42 -2
- package/dist/cli/commands/approve.js +5 -4
- package/dist/cli/commands/beat.js +7 -4
- package/dist/cli/commands/doctor.d.ts +5 -1
- package/dist/cli/commands/doctor.js +67 -7
- package/dist/cli/commands/init.js +36 -21
- package/dist/cli/commands/plan.js +16 -2
- package/dist/cli/commands/report.js +37 -1
- package/dist/cli/commands/verify.d.ts +5 -0
- package/dist/cli/commands/verify.js +140 -25
- package/dist/compile/collateral.js +15 -9
- package/dist/compile/native.js +3 -3
- package/dist/config/config.js +1 -1
- package/dist/drivers/index.d.ts +6 -0
- package/dist/drivers/index.js +19 -4
- package/dist/drivers/orca.d.ts +22 -1
- package/dist/drivers/orca.js +147 -6
- package/dist/drivers/subprocess.d.ts +3 -3
- package/dist/drivers/subprocess.js +16 -9
- package/dist/drivers/types.d.ts +2 -0
- package/dist/gates/baseline.d.ts +2 -0
- package/dist/gates/baseline.js +21 -4
- package/dist/gates/llm.d.ts +6 -0
- package/dist/gates/llm.js +25 -9
- package/dist/gates/review.d.ts +3 -1
- package/dist/gates/review.js +40 -12
- package/dist/gates/run-gates.js +17 -10
- package/dist/gates/verdict-cause.d.ts +6 -2
- package/dist/gates/verdict-cause.js +8 -4
- package/dist/run/consult.js +5 -1
- package/dist/run/daemon.d.ts +12 -0
- package/dist/run/daemon.js +218 -25
- package/dist/run/git.d.ts +1 -0
- package/dist/run/git.js +4 -0
- package/dist/run/journal.d.ts +15 -2
- package/dist/run/journal.js +70 -12
- package/dist/run/supervision.d.ts +6 -0
- package/dist/run/supervision.js +29 -1
- package/dist/tui/ink/init-app.js +4 -4
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +77 -18
- package/skills/tickmarkr-overseer/scripts/seat-send.sh +88 -18
- package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +36 -2
- package/skills/tickmarkr-overseer/scripts/watch-contamination.sh +36 -15
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +33 -7
- package/skills/tickmarkr-overseer/scripts/watch-pending-input.sh +32 -8
package/dist/drivers/orca.js
CHANGED
|
@@ -2,8 +2,9 @@ import { realpathSync } from "node:fs";
|
|
|
2
2
|
import { resolve } from "node:path";
|
|
3
3
|
import { shq } from "../adapters/types.js";
|
|
4
4
|
import { createWorktree, sh } from "../run/git.js";
|
|
5
|
+
import { Journal } from "../run/journal.js";
|
|
5
6
|
import { MAX_BUF } from "./subprocess.js";
|
|
6
|
-
import { formatOwnedName, panesToClose } from "./types.js";
|
|
7
|
+
import { formatOwnedName, panesToClose, parseOwnedName } from "./types.js";
|
|
7
8
|
// Orca (onorca.dev) as a third execution surface beside herdr and subprocess. tickmarkr keeps
|
|
8
9
|
// worktrees, routing, gates, journal and merges; orca supplies visible terminals only. Everything
|
|
9
10
|
// here is bound by the 1.4.186 conformance spike
|
|
@@ -31,7 +32,10 @@ import { formatOwnedName, panesToClose } from "./types.js";
|
|
|
31
32
|
// - `terminal list`'s `--worktree` is OPTIONAL (same help): the reconcile sweep omits it, because
|
|
32
33
|
// an older run's leftover sits in a checkout this run never knew (T2).
|
|
33
34
|
/** The response families the ONE shared envelope parser serves. There is no second JSON seam. */
|
|
34
|
-
export const ORCA_RESPONSE_FAMILIES = [
|
|
35
|
+
export const ORCA_RESPONSE_FAMILIES = [
|
|
36
|
+
"status", "create", "list", "read", "send", "wait", "show", "close",
|
|
37
|
+
"worktree-current", "hooks-status",
|
|
38
|
+
];
|
|
35
39
|
export const ORCA_FIXTURE_VERSION = "1.4.195";
|
|
36
40
|
export const ORCA_CLI_COMMAND_ENV = "ORCA_CLI_COMMAND";
|
|
37
41
|
export const STALE_HANDLE_CODE = "terminal_handle_stale";
|
|
@@ -50,6 +54,9 @@ const PAGE_LINES = 500; // per-page ask; orca caps server-side and reports `limi
|
|
|
50
54
|
const LIST_LIMIT = 10000; // well past orca's own row default; `truncated` still decides (listAll)
|
|
51
55
|
const MAX_PAGES = 400; // runaway guard: a cursor that stops advancing ends the sweep, never loops
|
|
52
56
|
const POLL_MS = 200;
|
|
57
|
+
export const WORKTREE_ADOPTION_TIMEOUT_MS = 60_000;
|
|
58
|
+
const WORKTREE_ADOPTION_POLL_MS = 1_000;
|
|
59
|
+
const WORKTREE_ADOPTION_JOURNAL_MS = 2_000;
|
|
53
60
|
const SYSTEM_TIME = {
|
|
54
61
|
now: () => Date.now(),
|
|
55
62
|
sleep: (ms) => new Promise((resolve) => setTimeout(resolve, ms)),
|
|
@@ -252,6 +259,9 @@ export class OrcaDriver {
|
|
|
252
259
|
pageLines;
|
|
253
260
|
pollMs;
|
|
254
261
|
probeStalenessMs;
|
|
262
|
+
journalRoots = new Map();
|
|
263
|
+
narrate;
|
|
264
|
+
hookCoverage;
|
|
255
265
|
constructor(opts = {}) {
|
|
256
266
|
this.bin = opts.bin ?? resolveOrcaCliBinary(process.cwd(), { env: opts.env, platform: opts.platform }) ?? "orca";
|
|
257
267
|
// Config values flow into a shell here: every argv element is quoted, always.
|
|
@@ -347,7 +357,7 @@ export class OrcaDriver {
|
|
|
347
357
|
// Canonical: git and Orca can spell one checkout two ways, and every later comparison — create
|
|
348
358
|
// receipt, relist, reconcile — is against THIS value.
|
|
349
359
|
const worktree = canonicalWorktreePath(cwd);
|
|
350
|
-
this.slots.set(id, { title, cwd: worktree, buf: "", recoveries: 0, recovering: false });
|
|
360
|
+
this.slots.set(id, { title, cwd: worktree, agent: opts?.agent, buf: "", recoveries: 0, recovering: false });
|
|
351
361
|
return { id, name: title, cwd: worktree, group: opts?.group };
|
|
352
362
|
}
|
|
353
363
|
/** Where to invoke the CLI for this slot's calls (see OrcaSlotState.dir). */
|
|
@@ -396,6 +406,7 @@ export class OrcaDriver {
|
|
|
396
406
|
}
|
|
397
407
|
async create(st, cmd) {
|
|
398
408
|
await this.probeRuntime(this.cliCwd(st));
|
|
409
|
+
await this.awaitWorktreeAdoption(st);
|
|
399
410
|
// The selector names THIS slot's checkout outright, and the CLI child is bound to it too, so
|
|
400
411
|
// neither the UI's active worktree (`active`/`current`) nor the daemon's cwd can place it.
|
|
401
412
|
const env = await this.call("create", [
|
|
@@ -429,6 +440,64 @@ export class OrcaDriver {
|
|
|
429
440
|
await this.notify(`tickmarkr orca terminal created on ${surface} surface`, { tier: "attention" });
|
|
430
441
|
}
|
|
431
442
|
}
|
|
443
|
+
/**
|
|
444
|
+
* A freshly-created git checkout does not become a valid Orca selector atomically. Ask
|
|
445
|
+
* `worktree current` FROM that checkout until Orca itself resolves the exact filesystem identity;
|
|
446
|
+
* an enclosing checkout is still not adoption. Only selector_not_found is a retryable refusal —
|
|
447
|
+
* malformed envelopes and every other refusal remain explicit driver failures.
|
|
448
|
+
*/
|
|
449
|
+
async awaitWorktreeAdoption(st) {
|
|
450
|
+
const started = this.time.now();
|
|
451
|
+
const deadline = started + WORKTREE_ADOPTION_TIMEOUT_MS;
|
|
452
|
+
let lastReported;
|
|
453
|
+
for (;;) {
|
|
454
|
+
const left = deadline - this.time.now();
|
|
455
|
+
if (left < 0)
|
|
456
|
+
break;
|
|
457
|
+
try {
|
|
458
|
+
const env = await this.call("worktree-current", ["worktree", "current"], st.cwd, Math.max(1, left));
|
|
459
|
+
const worktree = env.result.worktree;
|
|
460
|
+
if (typeof worktree !== "object" || worktree === null || Array.isArray(worktree)) {
|
|
461
|
+
throw new OrcaError("worktree-current", "response carries no worktree record", env.raw, { runtimeId: env.runtimeId });
|
|
462
|
+
}
|
|
463
|
+
const reported = terminalWorktree(worktree);
|
|
464
|
+
if (!reported) {
|
|
465
|
+
throw new OrcaError("worktree-current", "worktree record carries no path", env.raw, { runtimeId: env.runtimeId });
|
|
466
|
+
}
|
|
467
|
+
lastReported = canonicalWorktreePath(reported);
|
|
468
|
+
if (lastReported === st.cwd) {
|
|
469
|
+
const waitedMs = this.time.now() - started;
|
|
470
|
+
if (waitedMs > WORKTREE_ADOPTION_JOURNAL_MS)
|
|
471
|
+
this.appendAdoptionWait(st, waitedMs);
|
|
472
|
+
return;
|
|
473
|
+
}
|
|
474
|
+
}
|
|
475
|
+
catch (error) {
|
|
476
|
+
if (!(error instanceof OrcaError) || error.code !== "selector_not_found")
|
|
477
|
+
throw error;
|
|
478
|
+
lastReported = undefined;
|
|
479
|
+
}
|
|
480
|
+
const remaining = deadline - this.time.now();
|
|
481
|
+
if (remaining <= 0)
|
|
482
|
+
break;
|
|
483
|
+
await this.time.sleep(Math.min(WORKTREE_ADOPTION_POLL_MS, remaining));
|
|
484
|
+
}
|
|
485
|
+
const waitedMs = this.time.now() - started;
|
|
486
|
+
throw new OrcaUnavailableError("worktree-current", `Orca did not adopt ${st.cwd} within ${WORKTREE_ADOPTION_TIMEOUT_MS}ms${lastReported ? ` (last answered ${lastReported})` : ""}`, `waitedMs=${waitedMs}`);
|
|
487
|
+
}
|
|
488
|
+
/** Same repo/run/narration path Herdr uses for its driver-owned dispatch-retry row. */
|
|
489
|
+
appendAdoptionWait(st, waitedMs) {
|
|
490
|
+
const owned = parseOwnedName(st.title);
|
|
491
|
+
if (!owned)
|
|
492
|
+
throw new Error(`cannot journal worktree-adoption-wait: slot ${st.title} carries no run identity`);
|
|
493
|
+
const repoRoot = this.journalRoots.get(st.cwd);
|
|
494
|
+
if (!repoRoot) {
|
|
495
|
+
throw new Error(`cannot journal worktree-adoption-wait: slot ${st.title} has no daemon repo binding for ${st.cwd}`);
|
|
496
|
+
}
|
|
497
|
+
Journal.open(repoRoot, owned.runId, this.narrate).append("worktree-adoption-wait", owned.taskId, {
|
|
498
|
+
milliseconds: waitedMs,
|
|
499
|
+
});
|
|
500
|
+
}
|
|
432
501
|
// ---- handle identity and restart recovery ----------------------------------------------------
|
|
433
502
|
/**
|
|
434
503
|
* Every terminal-addressed call — read AND write — goes through here, and the runtime identity is
|
|
@@ -628,6 +697,18 @@ export class OrcaDriver {
|
|
|
628
697
|
], this.cliCwd(st)), { onRecovered: () => { recovered = true; } });
|
|
629
698
|
return { term: this.validated(family, st, env), raw: env.raw, recovered };
|
|
630
699
|
}
|
|
700
|
+
/** A rendered-frame liveness read. `--screen` and `--cursor` are mutually exclusive in Orca. */
|
|
701
|
+
async readScreen(st) {
|
|
702
|
+
const env = await this.terminalOp("status", st, (h) => this.call("read", [
|
|
703
|
+
"terminal", "read", "--terminal", h, "--screen",
|
|
704
|
+
], this.cliCwd(st)));
|
|
705
|
+
const term = this.validated("status", st, env);
|
|
706
|
+
const source = str(term.source);
|
|
707
|
+
if (source !== "screen" && source !== "screen-unavailable") {
|
|
708
|
+
throw new OrcaError("read", `screen read reports source ${source ?? "absent"}, not screen or screen-unavailable`, env.raw);
|
|
709
|
+
}
|
|
710
|
+
return { term, source };
|
|
711
|
+
}
|
|
631
712
|
/** A single UNPAGED tail read — exactly what the caller asked for and nothing more. Markers split
|
|
632
713
|
* across cursor pages are not reassembled here; that is waitOutput's job. */
|
|
633
714
|
async read(slot, lines) {
|
|
@@ -704,10 +785,14 @@ export class OrcaDriver {
|
|
|
704
785
|
// reporting even when the show record alone would still look connected (show carries no
|
|
705
786
|
// status field of its own, so a terminal can report "unknown"/"exited" on read while its
|
|
706
787
|
// show row still says connected — the read leg is the only place that catches that).
|
|
707
|
-
await this.
|
|
788
|
+
const screen = await this.readScreen(st);
|
|
708
789
|
if (`${st.runtimeId}:${st.handle}:${st.recoveries}` !== gen) {
|
|
709
790
|
continue;
|
|
710
791
|
}
|
|
792
|
+
// No rendered frame means no trustworthy TUI state. In particular, the stream fragments this
|
|
793
|
+
// call replaced cannot license an idle verdict or a wait probe.
|
|
794
|
+
if (screen.source === "screen-unavailable")
|
|
795
|
+
return "unknown";
|
|
711
796
|
const env = await this.terminalOp("show", st, (h) => this.call("show", ["terminal", "show", "--terminal", h], this.cliCwd(st)));
|
|
712
797
|
if (`${st.runtimeId}:${st.handle}:${st.recoveries}` !== gen) {
|
|
713
798
|
continue;
|
|
@@ -715,6 +800,11 @@ export class OrcaDriver {
|
|
|
715
800
|
const term = this.liveShowTerm("status", st, env);
|
|
716
801
|
if (term.agentWait === true)
|
|
717
802
|
return "blocked";
|
|
803
|
+
// Orca can only report tui-idle/agentWait for agents whose managed hook is installed. A
|
|
804
|
+
// definitively unhooked adapter is unknown; an agent absent from Orca's table keeps the
|
|
805
|
+
// legacy probe because absence is not proof that the CLI has no compatible status surface.
|
|
806
|
+
if (await this.agentHookAvailable(st) === false)
|
|
807
|
+
return "unknown";
|
|
718
808
|
// `idle` is proven only by orca's own tui-idle condition: a 1ms wait is a point-in-time probe —
|
|
719
809
|
// satisfied now → idle; elapsed (the recorded `timeout` refusal) → not idle.
|
|
720
810
|
const isIdle = await this.waitCondition(st, "tui-idle", 1);
|
|
@@ -724,6 +814,50 @@ export class OrcaDriver {
|
|
|
724
814
|
return mapAgentState(term, isIdle);
|
|
725
815
|
}
|
|
726
816
|
}
|
|
817
|
+
hookAgent(adapter) {
|
|
818
|
+
if (adapter === "claude-code")
|
|
819
|
+
return "claude";
|
|
820
|
+
if (adapter === "cursor-agent")
|
|
821
|
+
return "cursor";
|
|
822
|
+
return adapter;
|
|
823
|
+
}
|
|
824
|
+
/** true = hooked, false = definitively unhooked, undefined = agent absent from Orca's table. */
|
|
825
|
+
async agentHookAvailable(st) {
|
|
826
|
+
if (!st.agent)
|
|
827
|
+
return undefined;
|
|
828
|
+
this.hookCoverage ??= this.loadHookCoverage(st.cwd);
|
|
829
|
+
const coverage = await this.hookCoverage;
|
|
830
|
+
const state = coverage.states.get(this.hookAgent(st.agent));
|
|
831
|
+
// The table's omission is deliberately inconclusive even when managed hooks are disabled:
|
|
832
|
+
// Orca may not know this agent, so preserve the legacy probe exactly as an unlisted row does.
|
|
833
|
+
if (state === undefined)
|
|
834
|
+
return undefined;
|
|
835
|
+
return coverage.enabled && state === "installed";
|
|
836
|
+
}
|
|
837
|
+
async loadHookCoverage(cwd) {
|
|
838
|
+
const env = await this.call("hooks-status", ["agent", "hooks", "status"], cwd);
|
|
839
|
+
if (typeof env.result.enabled !== "boolean") {
|
|
840
|
+
throw new OrcaError("hooks-status", "response carries no boolean enabled", env.raw, { runtimeId: env.runtimeId });
|
|
841
|
+
}
|
|
842
|
+
const statuses = env.result.statuses;
|
|
843
|
+
if (!Array.isArray(statuses)) {
|
|
844
|
+
throw new OrcaError("hooks-status", "response carries no statuses array", env.raw, { runtimeId: env.runtimeId });
|
|
845
|
+
}
|
|
846
|
+
const states = new Map();
|
|
847
|
+
const allowed = new Set(["installed", "not_installed", "partial", "error"]);
|
|
848
|
+
for (const row of statuses) {
|
|
849
|
+
if (typeof row !== "object" || row === null || Array.isArray(row)) {
|
|
850
|
+
throw new OrcaError("hooks-status", "statuses carries a non-object row", env.raw, { runtimeId: env.runtimeId });
|
|
851
|
+
}
|
|
852
|
+
const agent = str(row.agent);
|
|
853
|
+
const state = str(row.state);
|
|
854
|
+
if (!agent || !state || !allowed.has(state)) {
|
|
855
|
+
throw new OrcaError("hooks-status", "status row carries no valid agent/state pair", env.raw, { runtimeId: env.runtimeId });
|
|
856
|
+
}
|
|
857
|
+
states.set(agent, state);
|
|
858
|
+
}
|
|
859
|
+
return { enabled: env.result.enabled, states };
|
|
860
|
+
}
|
|
727
861
|
/** One `terminal wait` through the full identity machinery. The recorded 1.4.186 elapsed answer
|
|
728
862
|
* is rc 1 + ok:true + {handle, condition, satisfied:false, status:"running"}; it is "not yet"
|
|
729
863
|
* only after this method validates all four fields. Any malformed/refused wait remains explicit. */
|
|
@@ -791,6 +925,9 @@ export class OrcaDriver {
|
|
|
791
925
|
return;
|
|
792
926
|
console.log(`[tickmarkr] ${msg}`); // console fallback only — notification injection is out of scope
|
|
793
927
|
}
|
|
928
|
+
narrateWith(narrate) {
|
|
929
|
+
this.narrate = narrate;
|
|
930
|
+
}
|
|
794
931
|
async close(slot) {
|
|
795
932
|
const st = this.slots.get(slot.id);
|
|
796
933
|
if (!st)
|
|
@@ -909,7 +1046,11 @@ export class OrcaDriver {
|
|
|
909
1046
|
catch { /* cosmetic — visibility hygiene never fails the run */ }
|
|
910
1047
|
}
|
|
911
1048
|
// tickmarkr's own createWorktree stays the sole checkout authority — orca never makes worktrees.
|
|
912
|
-
worktree(repo, branch, baseRef) {
|
|
913
|
-
|
|
1049
|
+
async worktree(repo, branch, baseRef) {
|
|
1050
|
+
const worktree = await createWorktree(repo, branch, baseRef);
|
|
1051
|
+
const repoRoot = canonicalWorktreePath(repo);
|
|
1052
|
+
this.journalRoots.set(repoRoot, repoRoot);
|
|
1053
|
+
this.journalRoots.set(canonicalWorktreePath(worktree), repoRoot);
|
|
1054
|
+
return worktree;
|
|
914
1055
|
}
|
|
915
1056
|
}
|
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
import type { ExecutorDriver, NotifyOpts, Slot } from "./types.js";
|
|
2
2
|
export declare const MAX_BUF: number;
|
|
3
|
-
export declare const HERDR_CONTROL_VARS: readonly ["HERDR_ENV", "HERDR_SOCKET_PATH"];
|
|
3
|
+
export declare const HERDR_CONTROL_VARS: readonly ["HERDR_ENV", "HERDR_SOCKET_PATH", "ORCA_TERMINAL_HANDLE", "ORCA_PANE_KEY", "ORCA_TAB_ID"];
|
|
4
4
|
/**
|
|
5
|
-
* Copy of worker env with the fork cap applied and
|
|
5
|
+
* Copy of worker env with the fork cap applied and host control-plane vars stripped.
|
|
6
6
|
* The cap is the one the enclosing run resolved (resolvedForkCap) — a worker's suites divide the
|
|
7
7
|
* same machine the gate shells do, so both seams have to read the same run-owned number rather
|
|
8
8
|
* than a flat constant. The operator's own export still wins.
|
|
9
9
|
*/
|
|
10
10
|
export declare function sealHerdrEnv(env?: NodeJS.ProcessEnv): NodeJS.ProcessEnv;
|
|
11
|
-
/** Pane/login-shell form of the same worker env seal (herdr seed + daemon setup). */
|
|
11
|
+
/** Pane/login-shell form of the same host-neutral worker env seal (herdr seed + daemon setup). */
|
|
12
12
|
export declare function herdrSealShellPrefix(env?: NodeJS.ProcessEnv): string;
|
|
13
13
|
export declare class SubprocessDriver implements ExecutorDriver {
|
|
14
14
|
id: string;
|
|
@@ -8,13 +8,20 @@ import { createWorktree, FORK_CAP_ENV, resolvedForkCap } from "../run/git.js";
|
|
|
8
8
|
// waitOutput poll (~10MB/s sustained, far beyond agent-CLI rates). Tail-truncate, never head:
|
|
9
9
|
// consumers only ever tail-read.
|
|
10
10
|
export const MAX_BUF = 2 * 1024 * 1024;
|
|
11
|
-
// OBS-17 / v1.22 T3: control-plane vars that let a process
|
|
12
|
-
// Workers, judges, reviewers, and consults must never inherit them — only the
|
|
13
|
-
//
|
|
14
|
-
//
|
|
15
|
-
|
|
11
|
+
// OBS-17 / v1.22 T3 and OBS-843: host control-plane vars that let a process address the operator's
|
|
12
|
+
// herdr or Orca UI. Workers, judges, reviewers, and consults must never inherit them — only the
|
|
13
|
+
// daemon process and its driver calls keep the live session. TERM_PROGRAM is host description, not
|
|
14
|
+
// an addressing capability, and ORCA_AGENT_HOOK_* belongs to agent status transport, so both stay.
|
|
15
|
+
// Keep the exported name for compatibility even though the list is now host-neutral.
|
|
16
|
+
export const HERDR_CONTROL_VARS = [
|
|
17
|
+
"HERDR_ENV",
|
|
18
|
+
"HERDR_SOCKET_PATH",
|
|
19
|
+
"ORCA_TERMINAL_HANDLE",
|
|
20
|
+
"ORCA_PANE_KEY",
|
|
21
|
+
"ORCA_TAB_ID",
|
|
22
|
+
];
|
|
16
23
|
/**
|
|
17
|
-
* Copy of worker env with the fork cap applied and
|
|
24
|
+
* Copy of worker env with the fork cap applied and host control-plane vars stripped.
|
|
18
25
|
* The cap is the one the enclosing run resolved (resolvedForkCap) — a worker's suites divide the
|
|
19
26
|
* same machine the gate shells do, so both seams have to read the same run-owned number rather
|
|
20
27
|
* than a flat constant. The operator's own export still wins.
|
|
@@ -27,7 +34,7 @@ export function sealHerdrEnv(env = process.env) {
|
|
|
27
34
|
delete out[k];
|
|
28
35
|
return out;
|
|
29
36
|
}
|
|
30
|
-
/** Pane/login-shell form of the same worker env seal (herdr seed + daemon setup). */
|
|
37
|
+
/** Pane/login-shell form of the same host-neutral worker env seal (herdr seed + daemon setup). */
|
|
31
38
|
export function herdrSealShellPrefix(env = process.env) {
|
|
32
39
|
const forkCap = sealHerdrEnv(env)[FORK_CAP_ENV] ?? resolvedForkCap();
|
|
33
40
|
return `export ${FORK_CAP_ENV}=${shq(forkCap)}; ` +
|
|
@@ -56,8 +63,8 @@ export class SubprocessDriver {
|
|
|
56
63
|
// HARD-05: interactive=false — no operator, so an open stdin pipe is a promise tickmarkr can never
|
|
57
64
|
// keep; codex exec appends a piped stdin as a <stdin> block (`codex exec --help`) and blocks on a
|
|
58
65
|
// read that never EOFs. One spawn site covers every adapter (D-06).
|
|
59
|
-
// v1.22 T3: seal
|
|
60
|
-
// operator's herdr
|
|
66
|
+
// v1.22 T3 / OBS-843: seal host control vars so worker/judge/review/consult children cannot
|
|
67
|
+
// address the operator's herdr or Orca UI. process.env of the daemon is untouched.
|
|
61
68
|
const p = spawn("bash", ["-lc", cmd], {
|
|
62
69
|
cwd: slot.cwd,
|
|
63
70
|
stdio: ["ignore", "pipe", "pipe"],
|
package/dist/drivers/types.d.ts
CHANGED
|
@@ -36,6 +36,8 @@ export interface SlotOpts {
|
|
|
36
36
|
group?: string;
|
|
37
37
|
label?: string;
|
|
38
38
|
owned?: OwnedName;
|
|
39
|
+
/** Adapter id for execution surfaces whose liveness support is agent-specific. */
|
|
40
|
+
agent?: string;
|
|
39
41
|
}
|
|
40
42
|
export declare const OWNED_ROLES: readonly ["worker", "judge", "review", "consult", "watch", "other"];
|
|
41
43
|
export type OwnedRole = (typeof OWNED_ROLES)[number];
|
package/dist/gates/baseline.d.ts
CHANGED
|
@@ -23,6 +23,8 @@ export interface BaselineCommand {
|
|
|
23
23
|
exitCode?: number;
|
|
24
24
|
fingerprints: string[];
|
|
25
25
|
missingCommand?: boolean;
|
|
26
|
+
/** Why a capture returned no verdict. */
|
|
27
|
+
invalidCause?: "ceiling-kill" | "resource-exhaustion";
|
|
26
28
|
/** What this command actually took at capture, on a pristine tree. Absent in pre-v1.90 baselines. */
|
|
27
29
|
durationMs?: number;
|
|
28
30
|
/** Sum of the per-file durations named by the runner; null when its output names none. */
|
package/dist/gates/baseline.js
CHANGED
|
@@ -427,8 +427,9 @@ export function ceilingKillResult(gate, r, ceilingMs) {
|
|
|
427
427
|
* shell asks it to finish faster than the thing it is measuring.
|
|
428
428
|
*/
|
|
429
429
|
export const CAPTURE_CEILING_MS = 1_800_000;
|
|
430
|
-
const invalidCaptureEntry = (durationMs, invalidatingLines = []) => ({
|
|
430
|
+
const invalidCaptureEntry = (durationMs, invalidCause, invalidatingLines = []) => ({
|
|
431
431
|
infra: true,
|
|
432
|
+
invalidCause,
|
|
432
433
|
fingerprints: [],
|
|
433
434
|
durationMs,
|
|
434
435
|
fileDurationSumMs: null,
|
|
@@ -463,7 +464,7 @@ export async function captureBaseline(cwd, commands) {
|
|
|
463
464
|
console.error(`tickmarkr: baseline capture for "${name}" was killed at its ${CAPTURE_CEILING_MS}ms ceiling — `
|
|
464
465
|
+ `it recorded NO fingerprints, so nothing is forgiven and every gate will treat a pre-existing `
|
|
465
466
|
+ `failure as a fresh one. Raise the ceiling or shorten the command.`);
|
|
466
|
-
base.commands[name] = invalidCaptureEntry(durationMs);
|
|
467
|
+
base.commands[name] = invalidCaptureEntry(durationMs, "ceiling-kill");
|
|
467
468
|
continue;
|
|
468
469
|
}
|
|
469
470
|
const combinedOutput = r.stdout + "\n" + r.stderr;
|
|
@@ -479,7 +480,7 @@ export async function captureBaseline(cwd, commands) {
|
|
|
479
480
|
+ `it recorded NO exit-code verdict and NO fingerprints, so nothing is forgiven for this command; `
|
|
480
481
|
+ `the measurement cannot distinguish a pre-existing failure from one caused by exhaustion. `
|
|
481
482
|
+ `First invalidating line: ${invalidatingLines[0]}`);
|
|
482
|
-
base.commands[name] = invalidCaptureEntry(durationMs, invalidatingLines);
|
|
483
|
+
base.commands[name] = invalidCaptureEntry(durationMs, "resource-exhaustion", invalidatingLines);
|
|
483
484
|
continue;
|
|
484
485
|
}
|
|
485
486
|
base.commands[name] = {
|
|
@@ -591,6 +592,19 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
|
591
592
|
record(killed);
|
|
592
593
|
continue;
|
|
593
594
|
}
|
|
595
|
+
const runner = shellToken(cmd) ?? name;
|
|
596
|
+
if (entry?.missingCommand === true || r.code === 127) {
|
|
597
|
+
const cause = entry?.missingCommand === true && r.code !== 127
|
|
598
|
+
? "the baseline capture recorded this runner as missing"
|
|
599
|
+
: `the head command exited 127${/\bENOENT\b/i.test(`${r.stdout}\n${r.stderr}`) ? " (spawn ENOENT)" : ""}`;
|
|
600
|
+
record({
|
|
601
|
+
gate: name,
|
|
602
|
+
pass: false,
|
|
603
|
+
details: `unreadable — ${cause}; runner ${JSON.stringify(runner)} did not produce a trustworthy verdict`,
|
|
604
|
+
meta: { unreadable: true, runner },
|
|
605
|
+
});
|
|
606
|
+
continue;
|
|
607
|
+
}
|
|
594
608
|
if (r.code === 0) {
|
|
595
609
|
record({ gate: name, pass: true, details: "exit 0" });
|
|
596
610
|
continue;
|
|
@@ -629,8 +643,11 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
|
629
643
|
// entries with no exitCode keep the `?? 1` red default both readers share (merge.ts:131).
|
|
630
644
|
const baselineRed = entry?.infra !== true && (entry?.exitCode ?? 1) !== 0;
|
|
631
645
|
if (!failing.length && !baselineRed) {
|
|
646
|
+
const recordedCause = entry?.invalidCause === "resource-exhaustion" || entry?.invalidatingLines?.length
|
|
647
|
+
? `was invalidated by its recorded process/resource-exhaustion cause${entry.invalidatingLines?.[0] ? ` (${entry.invalidatingLines[0]})` : ""}`
|
|
648
|
+
: "was killed at its ceiling";
|
|
632
649
|
const closed = entry?.infra === true
|
|
633
|
-
? `the baseline capture for this command
|
|
650
|
+
? `the baseline capture for this command ${recordedCause} and recorded no verdict, so nothing here is forgivable — it now exits ${r.code} with no recognizable failure lines — failing closed`
|
|
634
651
|
: `command was green at baseline but now exits ${r.code} with no recognizable failure lines — failing closed`;
|
|
635
652
|
const evidence = unrecognizedEvidence(raw);
|
|
636
653
|
record({
|
package/dist/gates/llm.d.ts
CHANGED
|
@@ -57,9 +57,15 @@ export declare function captureLlmOutput<T>(run: () => Promise<T>): Promise<{
|
|
|
57
57
|
value: T;
|
|
58
58
|
outputs: string[];
|
|
59
59
|
}>;
|
|
60
|
+
export interface LlmRunResult {
|
|
61
|
+
output: string;
|
|
62
|
+
exitCode?: number;
|
|
63
|
+
timedOut: boolean;
|
|
64
|
+
}
|
|
60
65
|
export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number): Promise<string>;
|
|
61
66
|
export declare function runViaDriver(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via: LlmVia, timeoutMs?: number): Promise<string>;
|
|
62
67
|
export declare function dewrapPaneVerdict(out: string, nonce: string): string;
|
|
68
|
+
export declare function runLlmDetailed(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via?: LlmVia, timeoutMs?: number): Promise<LlmRunResult>;
|
|
63
69
|
export declare function runLlm(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via?: LlmVia, timeoutMs?: number): Promise<string>;
|
|
64
70
|
export declare function extractJson<T>(raw: string): T | null;
|
|
65
71
|
/** Fable F3: verdict JSON must echo the call nonce — skip unbound or mismatched objects. */
|
package/dist/gates/llm.js
CHANGED
|
@@ -129,21 +129,24 @@ export async function captureLlmOutput(run) {
|
|
|
129
129
|
const value = await llmOutputCapture.run(outputs, run);
|
|
130
130
|
return { value, outputs };
|
|
131
131
|
}
|
|
132
|
-
|
|
132
|
+
async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 300000) {
|
|
133
133
|
const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
|
|
134
134
|
try {
|
|
135
135
|
const pf = join(dir, "prompt.md");
|
|
136
136
|
writeFileSync(pf, prompt);
|
|
137
137
|
const r = await sh(adapter.headlessCommand(pf, model), cwd, timeoutMs);
|
|
138
|
-
return r.stdout + "\n" + r.stderr;
|
|
138
|
+
return { output: r.stdout + "\n" + r.stderr, exitCode: r.code, timedOut: r.timedOut === true };
|
|
139
139
|
}
|
|
140
140
|
finally {
|
|
141
141
|
rmSync(dir, { recursive: true, force: true });
|
|
142
142
|
}
|
|
143
143
|
}
|
|
144
|
+
export async function runHeadless(adapter, model, prompt, cwd, timeoutMs = 300000) {
|
|
145
|
+
return (await runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs)).output;
|
|
146
|
+
}
|
|
144
147
|
// v1.1 default path: the same headless CLI call, but dispatched through the driver
|
|
145
148
|
// as a visible named agent (herdr pane), with the quote-split completion wrapper.
|
|
146
|
-
|
|
149
|
+
async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
|
|
147
150
|
const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
|
|
148
151
|
let slot;
|
|
149
152
|
let accountant;
|
|
@@ -176,6 +179,7 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
|
|
|
176
179
|
// false-complete — same guard the worker path uses (daemon.ts:330-331).
|
|
177
180
|
const exitPattern = `TICKMARKR_EXIT_${nonce}:\\d`;
|
|
178
181
|
let out;
|
|
182
|
+
let timedOut = false;
|
|
179
183
|
const gatePrompt = prompt.startsWith("TICKMARKR-JUDGE") || prompt.startsWith("TICKMARKR-REVIEW");
|
|
180
184
|
if (!gatePrompt) {
|
|
181
185
|
await via.driver.waitOutput(slot, exitPattern, timeoutMs, { regex: true });
|
|
@@ -241,8 +245,14 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
|
|
|
241
245
|
break;
|
|
242
246
|
}
|
|
243
247
|
}
|
|
248
|
+
timedOut = Date.now() - startedAt >= timeoutMs && !new RegExp(exitPattern).test(out);
|
|
244
249
|
}
|
|
245
|
-
|
|
250
|
+
const exitCode = Number(new RegExp(`TICKMARKR_EXIT_${nonce}:(\\d+)`).exec(out)?.[1]);
|
|
251
|
+
return {
|
|
252
|
+
output: dewrapPaneVerdict(out, nonce),
|
|
253
|
+
...(Number.isFinite(exitCode) ? { exitCode } : {}),
|
|
254
|
+
timedOut,
|
|
255
|
+
};
|
|
246
256
|
}
|
|
247
257
|
finally {
|
|
248
258
|
try {
|
|
@@ -256,6 +266,9 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
|
|
|
256
266
|
}
|
|
257
267
|
}
|
|
258
268
|
}
|
|
269
|
+
export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
|
|
270
|
+
return (await runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs)).output;
|
|
271
|
+
}
|
|
259
272
|
// OBS-155: a TUI renders the verdict as a bullet and HARD-wraps it at pane width with a 2-space
|
|
260
273
|
// continuation indent, splitting words mid-token — so literal newlines land inside JSON string
|
|
261
274
|
// literals and ZERO lines begin with `{`. `--source recent-unwrapped` cannot undo it: the wrap is
|
|
@@ -315,12 +328,15 @@ export function dewrapPaneVerdict(out, nonce) {
|
|
|
315
328
|
}
|
|
316
329
|
return out;
|
|
317
330
|
}
|
|
331
|
+
export async function runLlmDetailed(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
|
|
332
|
+
const result = await (via
|
|
333
|
+
? runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs)
|
|
334
|
+
: runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs));
|
|
335
|
+
llmOutputCapture.getStore()?.push(result.output);
|
|
336
|
+
return result;
|
|
337
|
+
}
|
|
318
338
|
export async function runLlm(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
|
|
319
|
-
|
|
320
|
-
? runViaDriver(adapter, model, prompt, cwd, via, timeoutMs)
|
|
321
|
-
: runHeadless(adapter, model, prompt, cwd, timeoutMs));
|
|
322
|
-
llmOutputCapture.getStore()?.push(out);
|
|
323
|
-
return out;
|
|
339
|
+
return (await runLlmDetailed(adapter, model, prompt, cwd, via, timeoutMs)).output;
|
|
324
340
|
}
|
|
325
341
|
export function extractJson(raw) {
|
|
326
342
|
const fenced = [...raw.matchAll(/```json\s*\n([\s\S]*?)```/g)].at(-1);
|
package/dist/gates/review.d.ts
CHANGED
|
@@ -41,13 +41,15 @@ export type TaskDiffMeasurement = {
|
|
|
41
41
|
readonly fullMeasurement: ArtifactDiffMeasurement;
|
|
42
42
|
readonly capMeasurement: ArtifactDiffMeasurement;
|
|
43
43
|
};
|
|
44
|
-
export declare function fetchTaskDiff(worktree: string, baseRef: string): Promise<TaskDiffMeasurement>;
|
|
44
|
+
export declare function fetchTaskDiff(worktree: string, baseRef: string, files?: readonly string[]): Promise<TaskDiffMeasurement>;
|
|
45
45
|
export declare function checkDiffCap(gate: string, measured: number, cap: number, prefix?: string): GateResult | null;
|
|
46
46
|
/** Apply the strict reviewable-logic cap and the finite, larger capture cap independently. */
|
|
47
47
|
export declare function checkTaskDiffCaps(gate: string, measured: Pick<TaskDiffMeasurement, "logicBytes" | "captureBytes">, logicCap: number, prefix?: string): GateResult | null;
|
|
48
48
|
export declare function isDiffCapPark(result: GateResult): boolean;
|
|
49
49
|
export declare function diffCapParkReason(results: GateResult[]): string | null;
|
|
50
50
|
export declare function modelId(model: string): string;
|
|
51
|
+
/** Provider identity comes from the served model, not a gateway adapter's stamped vendor. */
|
|
52
|
+
export declare function modelProvider(model: string, fallback?: string): string;
|
|
51
53
|
export declare function pickReviewer(author: Assignment, channels: BillingChannel[], exclude?: string[], // v1.1 failover: reviewer channels that already produced garbage for this task
|
|
52
54
|
prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
|
|
53
55
|
floor?: Tier): BillingChannel | null;
|