tickmarkr 2.6.1 → 2.6.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -3
- package/dist/adapters/catalog-remote.js +89 -47
- package/dist/adapters/claude-code.js +9 -6
- package/dist/adapters/codex.js +7 -4
- package/dist/adapters/prompt.d.ts +1 -0
- package/dist/adapters/prompt.js +14 -6
- package/dist/adapters/registry.js +3 -3
- package/dist/adapters/types.d.ts +12 -4
- package/dist/adapters/types.js +6 -0
- package/dist/cli/commands/approve.d.ts +5 -1
- package/dist/cli/commands/approve.js +66 -23
- package/dist/cli/commands/compile.js +13 -3
- package/dist/cli/commands/doctor.d.ts +2 -0
- package/dist/cli/commands/doctor.js +11 -3
- package/dist/cli/commands/fleet.js +45 -7
- package/dist/cli/commands/report.js +18 -2
- package/dist/cli/commands/resume.js +4 -2
- package/dist/cli/commands/status.js +24 -19
- package/dist/cli/help.d.ts +2 -0
- package/dist/cli/help.js +9 -2
- package/dist/config/config.d.ts +20 -0
- package/dist/config/config.js +47 -8
- package/dist/config/fleet-overlay.d.ts +1 -0
- package/dist/config/fleet-overlay.js +56 -0
- package/dist/drivers/orca.d.ts +26 -1
- package/dist/drivers/orca.js +193 -60
- package/dist/eval/canary.d.ts +2 -1
- package/dist/eval/canary.js +2 -2
- package/dist/eval/dispatch.js +1 -0
- package/dist/gates/acceptance.d.ts +2 -1
- package/dist/gates/acceptance.js +7 -2
- package/dist/gates/baseline.d.ts +12 -1
- package/dist/gates/baseline.js +11 -4
- package/dist/gates/llm.d.ts +5 -4
- package/dist/gates/llm.js +13 -13
- package/dist/gates/review.d.ts +8 -0
- package/dist/gates/review.js +40 -4
- package/dist/gates/run-gates.d.ts +2 -1
- package/dist/gates/run-gates.js +27 -13
- package/dist/gates/test-manifest.d.ts +3 -1
- package/dist/gates/test-manifest.js +9 -2
- package/dist/graph/schema.d.ts +2 -0
- package/dist/graph/schema.js +2 -0
- package/dist/plan/scope.js +2 -2
- package/dist/route/preference.d.ts +20 -2
- package/dist/route/preference.js +48 -13
- package/dist/route/router.js +30 -15
- package/dist/run/consult.d.ts +13 -1
- package/dist/run/consult.js +14 -5
- package/dist/run/daemon.d.ts +37 -2
- package/dist/run/daemon.js +579 -141
- package/dist/run/git.d.ts +8 -0
- package/dist/run/git.js +14 -0
- package/dist/run/journal.d.ts +126 -3
- package/dist/run/journal.js +410 -37
- package/dist/run/merge.d.ts +3 -1
- package/dist/run/merge.js +3 -2
- package/dist/run/operator-summary.d.ts +3 -0
- package/dist/run/operator-summary.js +3 -1
- package/dist/run/protocol.d.ts +31 -1
- package/dist/run/protocol.js +3 -1
- package/dist/run/supervision.d.ts +7 -1
- package/dist/run/supervision.js +5 -2
- package/dist/tui/cockpit/board.js +3 -3
- package/dist/tui/cockpit/decision-actions.d.ts +8 -5
- package/dist/tui/cockpit/decision-actions.js +55 -32
- package/dist/tui/cockpit/derive.js +13 -2
- package/dist/tui/cockpit/live-runtime.d.ts +10 -0
- package/dist/tui/cockpit/live-runtime.js +50 -3
- package/dist/tui/cockpit/run-cockpit.d.ts +3 -0
- package/dist/tui/cockpit/run-cockpit.js +26 -1
- package/dist/tui/cockpit/run-view.d.ts +9 -3
- package/dist/tui/cockpit/run-view.js +60 -7
- package/dist/tui/cockpit/setup-cockpit.d.ts +4 -0
- package/dist/tui/cockpit/setup-cockpit.js +6 -3
- package/dist/tui/ink/fleet-app.d.ts +15 -3
- package/dist/tui/ink/fleet-app.js +91 -22
- package/package.json +2 -1
- package/schema/config.schema.json +818 -0
- package/skills/tickmarkr-loop/SKILL.md +8 -2
- package/skills/tickmarkr-overseer/SKILL.md +42 -0
- package/skills/tickmarkr-overseer/scripts/classify-vitest-log.sh +88 -0
- package/skills/tickmarkr-overseer/scripts/context-statusline.sh +81 -0
- package/skills/tickmarkr-overseer/scripts/grade-ci.sh +36 -34
- package/skills/tickmarkr-overseer/scripts/watch-journal.sh +6 -4
package/dist/run/git.d.ts
CHANGED
|
@@ -277,6 +277,14 @@ export type BaseContainment = {
|
|
|
277
277
|
* future `git cherry` ever emitted a `-` line for a merge.
|
|
278
278
|
*/
|
|
279
279
|
export declare function declaredBaseContainment(cwd: string, declaredRef: string, targetRef?: string): Promise<BaseContainment>;
|
|
280
|
+
/**
|
|
281
|
+
* OBS-1107: does `dest` already hold everything `commit` changed? Replays the commit's OWN
|
|
282
|
+
* parent-to-commit change onto `dest` as a three-way merge; a clean merge whose tree IS `dest`'s tree
|
|
283
|
+
* brings nothing the destination lacks — a content-empty commit, or a patch already represented there
|
|
284
|
+
* under another hash. Content, never hash identity. Any git failure (a root commit, a git without
|
|
285
|
+
* `merge-tree --merge-base`) answers false: unproven work stays unaccounted.
|
|
286
|
+
*/
|
|
287
|
+
export declare function changeRepresented(cwd: string, commit: string, dest?: string): Promise<boolean>;
|
|
280
288
|
export declare function removeWorktree(repo: string, dir: string): Promise<void>;
|
|
281
289
|
export declare const REFS_PREFLIGHT_PREFIX = "refs/tickmarkr/preflight";
|
|
282
290
|
export declare const REFS_PROBE_REMEDY = "run from the main repository or a full clone; a sandbox that denies writes under that path cannot host a run";
|
package/dist/run/git.js
CHANGED
|
@@ -863,6 +863,20 @@ export async function declaredBaseContainment(cwd, declaredRef, targetRef = "HEA
|
|
|
863
863
|
}
|
|
864
864
|
return missing.length > 0 ? { result: "drifted", missing } : { result: "contained", via: "patch-id" };
|
|
865
865
|
}
|
|
866
|
+
/**
|
|
867
|
+
* OBS-1107: does `dest` already hold everything `commit` changed? Replays the commit's OWN
|
|
868
|
+
* parent-to-commit change onto `dest` as a three-way merge; a clean merge whose tree IS `dest`'s tree
|
|
869
|
+
* brings nothing the destination lacks — a content-empty commit, or a patch already represented there
|
|
870
|
+
* under another hash. Content, never hash identity. Any git failure (a root commit, a git without
|
|
871
|
+
* `merge-tree --merge-base`) answers false: unproven work stays unaccounted.
|
|
872
|
+
*/
|
|
873
|
+
export async function changeRepresented(cwd, commit, dest = "HEAD") {
|
|
874
|
+
const merged = await shGit(`git merge-tree --write-tree --merge-base=${shq(`${commit}^`)} ${shq(dest)} ${shq(commit)}`, cwd);
|
|
875
|
+
if (merged.code !== 0)
|
|
876
|
+
return false;
|
|
877
|
+
const tree = await shGit(`git rev-parse ${shq(`${dest}^{tree}`)}`, cwd);
|
|
878
|
+
return tree.code === 0 && merged.stdout.split("\n")[0].trim() === tree.stdout.trim();
|
|
879
|
+
}
|
|
866
880
|
export async function removeWorktree(repo, dir) {
|
|
867
881
|
await shGit(`git worktree remove --force ${shq(dir)}`, repo); // best-effort; stale dirs are re-added with -B
|
|
868
882
|
await shGit(`rm -rf ${shq(dir)}`, repo);
|
package/dist/run/journal.d.ts
CHANGED
|
@@ -18,6 +18,8 @@ export interface ResumeState {
|
|
|
18
18
|
tried: string[];
|
|
19
19
|
lastAssignment?: Assignment;
|
|
20
20
|
upheldFeedback?: string;
|
|
21
|
+
escalated?: Record<string, string>;
|
|
22
|
+
escalatedAdapters?: string[];
|
|
21
23
|
}
|
|
22
24
|
export interface CurrentAttemptGateReplay {
|
|
23
25
|
commit: string;
|
|
@@ -29,12 +31,70 @@ export declare const ATTEMPT_CAP_RELEASE: "attempt-cap";
|
|
|
29
31
|
export declare const GATE_SATISFIED_RELEASE: "gate-satisfied";
|
|
30
32
|
export declare const REVIEW_UPHELD_RELEASE: "review-upheld";
|
|
31
33
|
export declare const RECHECK_RELEASE: "recheck";
|
|
34
|
+
export declare const APPROVAL_REFUSED: "approval-refused";
|
|
35
|
+
/** OBS-1178: the journal row a decision answers — its physical 1-based line plus its timestamp. */
|
|
36
|
+
export interface DecisionBinding {
|
|
37
|
+
line: number;
|
|
38
|
+
ts: string;
|
|
39
|
+
}
|
|
40
|
+
/** The token every surface prints and `approve --park` accepts: `<line>@<ts>`. */
|
|
41
|
+
export declare const bindingToken: (binding: DecisionBinding) => string;
|
|
42
|
+
export declare function parseBindingToken(token: string): DecisionBinding | undefined;
|
|
43
|
+
/** The binding a task-approved row recorded under `park` or `failure`, when well formed. */
|
|
44
|
+
export declare function recordedBinding(value: unknown): DecisionBinding | undefined;
|
|
45
|
+
/** The newest failed gate before a park row — the gate a waive of that park would satisfy. */
|
|
46
|
+
export declare function failedGateBeforePark(events: readonly JournalEvent[], taskId: string, parkIndex: number): GateName | undefined;
|
|
47
|
+
/** The physical journal line of `events[i]` (see SOURCE_LINE). */
|
|
48
|
+
export declare const physicalLine: (events: readonly JournalEvent[], i: number) => number;
|
|
49
|
+
/** A row parsed outside this module (a cockpit capture) names its own physical line to the decision fold. */
|
|
50
|
+
export declare const withPhysicalLine: <T extends JournalEvent>(row: T, line: number) => T;
|
|
51
|
+
/** OBS-1178: an approval neither enacted nor refused yet, and — when it may not be enacted — why. */
|
|
52
|
+
export interface OpenDecision {
|
|
53
|
+
taskId: string;
|
|
54
|
+
line: number;
|
|
55
|
+
stale?: string;
|
|
56
|
+
}
|
|
57
|
+
/**
|
|
58
|
+
* OBS-1178: THE decision fold. Every reader of task-approved rows — the replay folds, the daemon's
|
|
59
|
+
* startup, sweep and pending-action folds, scope-amendment replay and its audits, status and both
|
|
60
|
+
* cockpits — reads decisions through it, so no two surfaces can disagree on whether one happened.
|
|
61
|
+
*
|
|
62
|
+
* `effective` holds the PHYSICAL lines of the task-approved rows whose effects may apply: a row that
|
|
63
|
+
* was sound when an enactment row consumed it (bound by line and timestamp to the task's newest park,
|
|
64
|
+
* a waive also to that park's failed gate, or a recheck to the newest failure with no park after it;
|
|
65
|
+
* a legacy row carrying no binding only when no park or failure landed between it and that
|
|
66
|
+
* enactment), plus a row still open that is sound NOW. A refused row is never effective, and neither is an open row that no
|
|
67
|
+
* longer binds — a newer park or failure landed, or it names none — whatever unrelated rows landed
|
|
68
|
+
* around it. An enactment row also consumes the task's park and failure: a decision naming either
|
|
69
|
+
* after the task moved on is stale, while the decisions that enactment already consumed stay effective.
|
|
70
|
+
* `open` lists every unenacted, unrefused row with why it may not be enacted, and `refused` the lines
|
|
71
|
+
* an approval-refused row answered.
|
|
72
|
+
*/
|
|
73
|
+
export declare function foldDecisions(events: readonly JournalEvent[]): {
|
|
74
|
+
effective: Set<number>;
|
|
75
|
+
open: OpenDecision[];
|
|
76
|
+
refused: Set<number>;
|
|
77
|
+
};
|
|
78
|
+
/** The physical lines of the task-approved rows whose effects may apply (see foldDecisions). */
|
|
79
|
+
export declare const effectiveDecisions: (events: readonly JournalEvent[]) => Set<number>;
|
|
80
|
+
/**
|
|
81
|
+
* The journal as every decision reader must see it: task-approved rows that are not effective removed,
|
|
82
|
+
* every other row (and its physical line) kept. Order-only folds iterate this instead of skipping rows
|
|
83
|
+
* on their own.
|
|
84
|
+
*/
|
|
85
|
+
export declare function effectiveEvents(events: readonly JournalEvent[]): JournalEvent[];
|
|
86
|
+
/** Per task, the open decisions that may not be enacted now — what the daemon refuses before enactment. */
|
|
87
|
+
export interface StaleApprovals {
|
|
88
|
+
reason: string;
|
|
89
|
+
lines: number[];
|
|
90
|
+
}
|
|
91
|
+
export declare function staleApprovals(events: readonly JournalEvent[]): Map<string, StaleApprovals>;
|
|
32
92
|
export interface PreservedRef {
|
|
33
93
|
ref: string;
|
|
34
94
|
diffCommand: string;
|
|
35
95
|
}
|
|
36
96
|
export declare function preservedRefsByTask(events: JournalEvent[]): Map<string, PreservedRef[]>;
|
|
37
|
-
export declare function reviewRoundsSinceApproval(events: JournalEvent[], taskId: string): number;
|
|
97
|
+
export declare function reviewRoundsSinceApproval(events: JournalEvent[], taskId: string, rounds?: (decided: JournalEvent[]) => JournalEvent[]): number;
|
|
38
98
|
export declare function upheldFeedbackByTask(events: JournalEvent[]): Map<string, string>;
|
|
39
99
|
export interface StructuredFinding {
|
|
40
100
|
class: string;
|
|
@@ -177,6 +237,7 @@ export type PendingApprovalAction = {
|
|
|
177
237
|
* enacts it.
|
|
178
238
|
*/
|
|
179
239
|
export declare function pendingApprovalActions(events: JournalEvent[]): Map<string, PendingApprovalAction>;
|
|
240
|
+
export declare function approvalAction(taskId: string, e: JournalEvent): PendingApprovalAction;
|
|
180
241
|
/**
|
|
181
242
|
* Why the last attempt failed, one row per journaled cause, in the daemon's own `source: details`
|
|
182
243
|
* shape. The daemon builds that brief in a loop-local variable, which dies with the process: a resumed
|
|
@@ -192,10 +253,20 @@ export declare function pendingApprovalActions(events: JournalEvent[]): Map<stri
|
|
|
192
253
|
* failures OR the delivery failure that preceded it. Of the approvals, only a WAIVE clears (the operator
|
|
193
254
|
* retired the findings by fiat — the uphold case re-derives its own brief separately). OBS-1074: a
|
|
194
255
|
* plain approve, a scope grant or a recheck re-funds an attempt that must still see why the last one
|
|
195
|
-
* parked
|
|
196
|
-
*
|
|
256
|
+
* parked — v2.5.7's T11 looped four times on one hygiene oracle because every approval erased exactly
|
|
257
|
+
* the finding the fresh attempt was funded to fix. The operator's stated reasons ride after those rows
|
|
258
|
+
* as `standingRulings` (OBS-1150): no launch and no waive resets them.
|
|
197
259
|
*/
|
|
198
260
|
export declare function journaledFailureBrief(events: JournalEvent[], taskId: string): string[];
|
|
261
|
+
/**
|
|
262
|
+
* OBS-1150: an operator's approval reason is a ruling on the TASK, not on the attempt it released.
|
|
263
|
+
* The failure rows above are spent at the next worker-launch and the review context once bound only
|
|
264
|
+
* the newest reason, so ruling A vanished at the first launch after it and a later ruling B replaced
|
|
265
|
+
* it. Every effective reason therefore stands, oldest first, for the whole run; a repeat is carried
|
|
266
|
+
* once. A gate-satisfied release accepts a gate's verdict rather than ruling on the work, so its
|
|
267
|
+
* reason adds no standing ruling — and, being no ruling, it retires none either.
|
|
268
|
+
*/
|
|
269
|
+
export declare function standingRulings(events: JournalEvent[], taskId: string): string[];
|
|
199
270
|
/** Never infer a cross-path alias from a symbol, sentence, or hash. */
|
|
200
271
|
export declare function reviewFingerprintMatches(candidate: unknown, fingerprint: string): boolean;
|
|
201
272
|
export declare function observedReviewFingerprints(finding: StructuredFinding): string[];
|
|
@@ -256,6 +327,32 @@ export type WorkerResultCause = (typeof WORKER_RESULT_CAUSES)[number];
|
|
|
256
327
|
export declare function isQualityFailureParkKind(kind: ParkKind): boolean;
|
|
257
328
|
export declare function classifyTaskFailure(taskEvents: JournalEvent[]): ParkKind;
|
|
258
329
|
export declare function recordedTaskFailureKind(events: JournalEvent[], taskId: string): ParkKind | undefined;
|
|
330
|
+
export declare const RESUME_HARVEST_SOURCE: "resume";
|
|
331
|
+
export interface InterruptedAttempt {
|
|
332
|
+
attempt: number;
|
|
333
|
+
/** The producing dispatch's own assignment — the author the harvested work is gated under. */
|
|
334
|
+
assignment: Assignment;
|
|
335
|
+
/** Owned evidence still to read: nothing of this attempt's harvest is recorded yet. */
|
|
336
|
+
launch?: {
|
|
337
|
+
nonce: string;
|
|
338
|
+
dispatchScript: string;
|
|
339
|
+
slot: {
|
|
340
|
+
id: string;
|
|
341
|
+
name: string;
|
|
342
|
+
cwd: string;
|
|
343
|
+
};
|
|
344
|
+
};
|
|
345
|
+
/** A worker-result with no harvest row after it: finish that record. `reaped` only when a resume wrote
|
|
346
|
+
* it — a resume records the result after its owned-process cleanup; a live daemon records it BEFORE
|
|
347
|
+
* handling a failed reap, so its result alone never proves cleanup and `launch` must be reaped again. */
|
|
348
|
+
result?: {
|
|
349
|
+
finished: boolean;
|
|
350
|
+
summary: string;
|
|
351
|
+
reaped: boolean;
|
|
352
|
+
};
|
|
353
|
+
}
|
|
354
|
+
export declare function resumeHarvestAuthor(events: JournalEvent[], taskId: string): Assignment | undefined;
|
|
355
|
+
export declare function interruptedAttempt(events: JournalEvent[], taskId: string): InterruptedAttempt | undefined;
|
|
259
356
|
export declare function runHasEnded(events: JournalEvent[]): boolean;
|
|
260
357
|
/** OBS-53: classify worker-result failures so retries and routing see the true signal, not one lumped bucket. */
|
|
261
358
|
export declare function classifyWorkerResultCause(opts: {
|
|
@@ -392,6 +489,8 @@ export type EngagementCompare = {
|
|
|
392
489
|
export declare function engagementComparable(events: JournalEvent[], loadedHash: string): EngagementCompare;
|
|
393
490
|
export declare function newRunId(now?: Date): string;
|
|
394
491
|
export declare function parseRunId(runId: string): string;
|
|
492
|
+
/** The journal reader rule over bytes in hand: skip blanks, drop a torn line, keep each row's physical line. */
|
|
493
|
+
export declare function parseJournalText(raw: string): JournalEvent[];
|
|
395
494
|
export declare function readAllTelemetry(repoRoot: string, lastK: number, opts?: {
|
|
396
495
|
after?: string;
|
|
397
496
|
}): (TelemetryRow & {
|
|
@@ -402,6 +501,23 @@ export declare const PRIOR_JOURNAL_RUN_WINDOW = 50;
|
|
|
402
501
|
export declare function readPriorRunEvidence(repoRoot: string, tasks: readonly Pick<Task, "id" | "goal" | "files" | "acceptance">[], opts?: {
|
|
403
502
|
suppressRunId?: string;
|
|
404
503
|
}): PriorRunEvidence;
|
|
504
|
+
export declare const REVIEW_NO_VERDICT_RUN_WINDOW = 10;
|
|
505
|
+
export declare const REVIEW_NO_VERDICT_ADVISORY_AT = 2;
|
|
506
|
+
export interface ReviewNoVerdictHistory {
|
|
507
|
+
/** completed runs measured, oldest first */
|
|
508
|
+
runs: string[];
|
|
509
|
+
/** window journals with a row that did not parse — their events are unknown, never zero */
|
|
510
|
+
unreadable: string[];
|
|
511
|
+
/** reviewer channel → review-no-verdict events; every channel seen reviewing is present, 0 included */
|
|
512
|
+
counts: Map<string, number>;
|
|
513
|
+
}
|
|
514
|
+
export declare function readReviewNoVerdictHistory(repoRoot: string, window?: number): ReviewNoVerdictHistory;
|
|
515
|
+
/** One row per measured channel (plus one naming unreadable journals); doctor prints all, Fleet the warn rows. */
|
|
516
|
+
export declare function reviewNoVerdictRows(history: ReviewNoVerdictHistory): {
|
|
517
|
+
channel: string;
|
|
518
|
+
verdict: "pass" | "warn";
|
|
519
|
+
value: string;
|
|
520
|
+
}[];
|
|
405
521
|
export declare function readProfileCursor(repoRoot: string): string | undefined;
|
|
406
522
|
export declare function profileDiscountsPath(repoRoot: string): string;
|
|
407
523
|
export declare function readProfileDiscounts(repoRoot: string): ProfileDiscount[];
|
|
@@ -424,6 +540,13 @@ export declare class Journal {
|
|
|
424
540
|
append(event: string, taskId?: string, data?: Record<string, unknown>): void;
|
|
425
541
|
phaseStart(taskId: string, phase: TaskPhase, data?: Record<string, unknown>): void;
|
|
426
542
|
read(): JournalEvent[];
|
|
543
|
+
/** Parsed rows paired with their physical 1-based journal lines (OBS-1178 bindings name lines). */
|
|
544
|
+
readSourced(): {
|
|
545
|
+
events: JournalEvent[];
|
|
546
|
+
lines: number[];
|
|
547
|
+
};
|
|
548
|
+
/** OBS-1178: the binding of a task's newest park (or `task-failed`) row — what a decision on it names. */
|
|
549
|
+
newestBinding(taskId: string, event?: "task-human" | "task-failed"): DecisionBinding | undefined;
|
|
427
550
|
readTracked(): TrackedJournalRow[];
|
|
428
551
|
replayStatuses(): Map<string, TaskStatus>;
|
|
429
552
|
/**
|