tickmarkr 1.84.0 → 1.86.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -2
- package/dist/adapters/catalog-remote.d.ts +64 -0
- package/dist/adapters/catalog-remote.js +287 -0
- package/dist/adapters/catalog.d.ts +96 -0
- package/dist/adapters/catalog.js +176 -0
- package/dist/adapters/claude-code.d.ts +1 -0
- package/dist/adapters/claude-code.js +59 -1
- package/dist/adapters/fake.js +42 -4
- package/dist/adapters/model-lints.d.ts +25 -5
- package/dist/adapters/model-lints.js +184 -50
- package/dist/adapters/model-windows.d.ts +31 -0
- package/dist/adapters/model-windows.js +69 -0
- package/dist/adapters/prompt.d.ts +5 -1
- package/dist/adapters/prompt.js +13 -4
- package/dist/adapters/registry.d.ts +25 -26
- package/dist/adapters/registry.js +173 -110
- package/dist/adapters/types.d.ts +3 -0
- package/dist/adapters/types.js +36 -3
- package/dist/brand.d.ts +5 -1
- package/dist/brand.js +18 -2
- package/dist/cli/commands/doctor.d.ts +3 -0
- package/dist/cli/commands/doctor.js +43 -21
- package/dist/cli/commands/fleet.d.ts +7 -0
- package/dist/cli/commands/fleet.js +94 -74
- package/dist/cli/commands/init.js +118 -5
- package/dist/cli/commands/status.js +202 -46
- package/dist/compile/collateral.d.ts +86 -2
- package/dist/compile/collateral.js +294 -3
- package/dist/compile/gsd.d.ts +2 -1
- package/dist/compile/gsd.js +68 -2
- package/dist/compile/native.d.ts +14 -0
- package/dist/compile/native.js +161 -12
- package/dist/config/config.d.ts +82 -5
- package/dist/config/config.js +253 -66
- package/dist/config/fleet-overlay.d.ts +25 -20
- package/dist/config/fleet-overlay.js +195 -77
- package/dist/config/fleet-why.d.ts +23 -0
- package/dist/config/fleet-why.js +42 -0
- package/dist/drivers/herdr.d.ts +21 -3
- package/dist/drivers/herdr.js +344 -110
- package/dist/gates/acceptance.js +7 -2
- package/dist/gates/baseline.d.ts +1 -0
- package/dist/gates/baseline.js +91 -13
- package/dist/gates/llm.d.ts +0 -1
- package/dist/gates/llm.js +5 -30
- package/dist/gates/review.d.ts +9 -1
- package/dist/gates/review.js +105 -10
- package/dist/gates/run-gates.d.ts +9 -0
- package/dist/gates/run-gates.js +285 -41
- package/dist/gates/verdict-cause.d.ts +4 -0
- package/dist/gates/verdict-cause.js +63 -0
- package/dist/graph/schema.d.ts +6 -0
- package/dist/graph/schema.js +8 -5
- package/dist/route/router.d.ts +0 -5
- package/dist/route/router.js +16 -20
- package/dist/run/consult.d.ts +6 -0
- package/dist/run/consult.js +35 -25
- package/dist/run/daemon.d.ts +48 -2
- package/dist/run/daemon.js +1488 -330
- package/dist/run/journal.d.ts +56 -3
- package/dist/run/journal.js +358 -4
- package/dist/run/stall.d.ts +35 -1
- package/dist/run/stall.js +118 -8
- package/dist/tui/cockpit/capture.d.ts +12 -0
- package/dist/tui/cockpit/capture.js +37 -1
- package/dist/tui/cockpit/components.d.ts +2 -0
- package/dist/tui/cockpit/components.js +8 -8
- package/dist/tui/cockpit/derive.d.ts +29 -2
- package/dist/tui/cockpit/derive.js +219 -23
- package/dist/tui/cockpit/run-cockpit.js +128 -27
- package/dist/tui/cockpit/theme.d.ts +32 -26
- package/dist/tui/cockpit/theme.js +11 -5
- package/dist/tui/ink/components.d.ts +0 -15
- package/dist/tui/ink/components.js +0 -17
- package/dist/tui/ink/fleet-app.d.ts +4 -1
- package/dist/tui/ink/fleet-app.js +134 -13
- package/fixtures/sample.native.md +1 -1
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +354 -34
- package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +70 -0
- package/skills/tickmarkr-overseer/scripts/watch-panes.sh +1 -1
- package/dist/tui/ink/studio-app.d.ts +0 -59
- package/dist/tui/ink/studio-app.js +0 -320
- package/dist/tui/save.d.ts +0 -38
- package/dist/tui/save.js +0 -96
- package/dist/tui/staging.d.ts +0 -29
- package/dist/tui/staging.js +0 -78
package/dist/run/journal.d.ts
CHANGED
|
@@ -25,9 +25,61 @@ export declare const GATE_SATISFIED_RELEASE: "gate-satisfied";
|
|
|
25
25
|
export declare const REVIEW_UPHELD_RELEASE: "review-upheld";
|
|
26
26
|
export declare const RECHECK_RELEASE: "recheck";
|
|
27
27
|
export declare function reviewRoundsSinceApproval(events: JournalEvent[], taskId: string): number;
|
|
28
|
+
export declare function upheldFeedbackByTask(events: JournalEvent[]): Map<string, string>;
|
|
29
|
+
export interface StructuredFinding {
|
|
30
|
+
class: string;
|
|
31
|
+
path: string;
|
|
32
|
+
symbol: string;
|
|
33
|
+
note: string;
|
|
34
|
+
fingerprint: string;
|
|
35
|
+
}
|
|
36
|
+
export declare const UNIDENTIFIED = "<unidentified>";
|
|
37
|
+
/**
|
|
38
|
+
* Structured findings for a BLOCKING review/judge gate result, parsed from the details the gate
|
|
39
|
+
* already writes (D-03: no gate-module change, so an older gate's prose degrades to one unclassified
|
|
40
|
+
* finding rather than to none). Never empty for a blocking result — a finding the journal cannot
|
|
41
|
+
* classify is still a finding the next retry must not lose.
|
|
42
|
+
*
|
|
43
|
+
* Rule: a finding's path is its verdict row's own evidence path. An inline path and an anchored row's
|
|
44
|
+
* path therefore resolve; a different anchor or the task's declared scope never substitutes for a
|
|
45
|
+
* pathless finding. That row fails closed as UNIDENTIFIED instead of manufacturing an R4 identity.
|
|
46
|
+
* A symbol the row's own prose does not name falls back to the criterion id, then to the row's own
|
|
47
|
+
* normalized words (see toFinding).
|
|
48
|
+
*/
|
|
49
|
+
export declare function structuredFindings(gate: string, details: string, _scopeFiles?: string[]): StructuredFinding[];
|
|
50
|
+
/** Normalized identity of a gate failure: the same defect, seen twice, normalizes to the same bytes. */
|
|
51
|
+
export declare function normalizeGateFailure(details: string): string;
|
|
52
|
+
export declare const GATE_FINGERPRINT_CAP = 2;
|
|
53
|
+
export declare function identicalGateFailures(events: JournalEvent[], taskId: string, gate: string, normalized: string): number;
|
|
54
|
+
/** Repair attempts this engagement has already funded — journal-derived, so a resume inherits it. */
|
|
55
|
+
export declare function repairsSinceApproval(events: JournalEvent[], taskId: string): number;
|
|
56
|
+
/**
|
|
57
|
+
* Why the last attempt failed, one row per journaled cause, in the daemon's own `source: details`
|
|
58
|
+
* shape. The daemon builds that brief in a loop-local variable, which dies with the process: a resumed
|
|
59
|
+
* or `--retry-failed` run rebuilt the prompt from nothing and dispatched a retry that had lost the
|
|
60
|
+
* reason it was retrying — OBS-254's class, one layer below the upheld brief. Re-derived here so the
|
|
61
|
+
* bytes the journal already holds cannot be taken away by any reset of attempt or channel state.
|
|
62
|
+
*
|
|
63
|
+
* The same rule governs a dead DISPATCH: its exact task-failed error is retained until
|
|
64
|
+
* `worker-launch`, never retired at
|
|
65
|
+
* `task-dispatch`: everything between the two — worktree recreation, setup, prompt write, slot
|
|
66
|
+
* allocation, the launch itself — can still die with no worker having read a word, and clearing at
|
|
67
|
+
* task-dispatch meant `--retry-failed` after exactly that death rebuilt the prompt without the gate
|
|
68
|
+
* failures OR the delivery failure that preceded it. `task-approved` also clears (an operator approval
|
|
69
|
+
* retires the findings it settled — the uphold case re-derives its own brief separately).
|
|
70
|
+
*/
|
|
71
|
+
export declare function journaledFailureBrief(events: JournalEvent[], taskId: string): string[];
|
|
72
|
+
/** The findings a funded repair must carry into the next dispatch, or undefined if none is pending. */
|
|
73
|
+
export declare function pendingRepairFindings(events: JournalEvent[], taskId: string): string | undefined;
|
|
74
|
+
/**
|
|
75
|
+
* The gate whose identical failure banned an identical retry of the NEXT dispatch — bound to the
|
|
76
|
+
* channel that produced it, so a verdict that has already moved the work elsewhere is not refused for
|
|
77
|
+
* a channel it is no longer using, and a later unrelated failure is not parked under a stale reason.
|
|
78
|
+
*/
|
|
79
|
+
export declare function activeRetryBan(events: JournalEvent[], taskId: string, channel: string): string | undefined;
|
|
28
80
|
export declare const PARK_KINDS: readonly ["human-gate", "ladder-exhausted", "attempt-cap", "gate-fail", "quota", "reroute-exhausted", "setup", "stall", "merge-conflict", "tip-moved", "infra", "dispatch"];
|
|
29
81
|
export type ParkKind = (typeof PARK_KINDS)[number];
|
|
30
|
-
export declare const RETRY_MODES: readonly ["resume", "fresh"];
|
|
82
|
+
export declare const RETRY_MODES: readonly ["resume", "fresh", "repair"];
|
|
31
83
|
export type RetryMode = (typeof RETRY_MODES)[number];
|
|
32
84
|
export declare const WORKER_RESULT_CAUSES: readonly ["provider-death", "stall-timeout", "malformed-trailer", "clean-exit-no-trailer"];
|
|
33
85
|
export type WorkerResultCause = (typeof WORKER_RESULT_CAUSES)[number];
|
|
@@ -66,13 +118,13 @@ export declare const TelemetryRowSchema: z.ZodObject<{
|
|
|
66
118
|
"attempt-cap": "attempt-cap";
|
|
67
119
|
"gate-fail": "gate-fail";
|
|
68
120
|
quota: "quota";
|
|
121
|
+
dispatch: "dispatch";
|
|
69
122
|
"human-gate": "human-gate";
|
|
70
123
|
"reroute-exhausted": "reroute-exhausted";
|
|
71
124
|
stall: "stall";
|
|
72
125
|
"merge-conflict": "merge-conflict";
|
|
73
126
|
"tip-moved": "tip-moved";
|
|
74
127
|
infra: "infra";
|
|
75
|
-
dispatch: "dispatch";
|
|
76
128
|
}>>;
|
|
77
129
|
tokens: z.ZodCatch<z.ZodOptional<z.ZodObject<{
|
|
78
130
|
input: z.ZodNumber;
|
|
@@ -87,13 +139,14 @@ export declare const TelemetryRowSchema: z.ZodObject<{
|
|
|
87
139
|
retryMode: z.ZodOptional<z.ZodEnum<{
|
|
88
140
|
resume: "resume";
|
|
89
141
|
fresh: "fresh";
|
|
142
|
+
repair: "repair";
|
|
90
143
|
}>>;
|
|
91
144
|
signalQuality: z.ZodOptional<z.ZodUnion<readonly [z.ZodLiteral<0>, z.ZodLiteral<0.25>, z.ZodLiteral<0.5>, z.ZodLiteral<0.75>, z.ZodLiteral<1>]>>;
|
|
92
145
|
signalBasis: z.ZodOptional<z.ZodEnum<{
|
|
146
|
+
"judge-only": "judge-only";
|
|
93
147
|
skipped: "skipped";
|
|
94
148
|
proved: "proved";
|
|
95
149
|
"review-agree": "review-agree";
|
|
96
|
-
"judge-only": "judge-only";
|
|
97
150
|
legacy: "legacy";
|
|
98
151
|
vacuous: "vacuous";
|
|
99
152
|
}>>;
|
package/dist/run/journal.js
CHANGED
|
@@ -70,6 +70,329 @@ export function reviewRoundsSinceApproval(events, taskId) {
|
|
|
70
70
|
}
|
|
71
71
|
return rounds;
|
|
72
72
|
}
|
|
73
|
+
// OBS-189/OBS-254: the uphold brief is the operator's funded decision, not attempt state. ONE fold,
|
|
74
|
+
// two consumers — replayResumeState seeds it, and the daemon re-derives it from the journal at
|
|
75
|
+
// prompt-build time so no reset of attempt/channel state can take the findings with it (OBS-254 deleted
|
|
76
|
+
// the whole resume entry and dispatched a funded attempt with an empty "fix these specifically" heading).
|
|
77
|
+
export function upheldFeedbackByTask(events) {
|
|
78
|
+
const upheld = new Map();
|
|
79
|
+
const lastReviewFail = new Map(); // newest failed review details per task
|
|
80
|
+
for (const e of events) {
|
|
81
|
+
if (!e.taskId)
|
|
82
|
+
continue;
|
|
83
|
+
if (e.event === "gate-result" && e.data.gate === "review" && e.data.pass === false
|
|
84
|
+
&& typeof e.data.details === "string") {
|
|
85
|
+
lastReviewFail.set(e.taskId, e.data.details);
|
|
86
|
+
}
|
|
87
|
+
else if (e.event === "task-approved") {
|
|
88
|
+
// any later approval supersedes: a plain accept-the-diff approval retires the uphold brief.
|
|
89
|
+
if (e.data.release === REVIEW_UPHELD_RELEASE) {
|
|
90
|
+
const details = lastReviewFail.get(e.taskId);
|
|
91
|
+
if (details)
|
|
92
|
+
upheld.set(e.taskId, details);
|
|
93
|
+
else
|
|
94
|
+
upheld.delete(e.taskId);
|
|
95
|
+
}
|
|
96
|
+
else {
|
|
97
|
+
upheld.delete(e.taskId);
|
|
98
|
+
}
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
return upheld;
|
|
102
|
+
}
|
|
103
|
+
// Reserved for a finding whose OWN evidence names no path. Reporting a blank path a reader would take
|
|
104
|
+
// for a resolved one is the silent-lie shape the gates exist to refuse, so the field says so outright.
|
|
105
|
+
export const UNIDENTIFIED = "<unidentified>";
|
|
106
|
+
const ANCHORED_RE = /^- (\S+?):(\d+) — (.*)$/; // "## Anchored review" rows (llm.ts)
|
|
107
|
+
const REVIEW_ROW_RE = /^- \[([^\]]+)\] (.*)$/; // "- [material] …" (review.ts)
|
|
108
|
+
const JUDGE_ROW_RE = /^✗ ([\w.-]+): (.*)$/; // "✗ c1: …" (acceptance.ts) — id, then reason
|
|
109
|
+
const PATH_RE = /\b((?:[\w.@~+-]+\/)+[\w.@~+-]+\.\w{1,6})\b/;
|
|
110
|
+
const LINE_REF_RE = /(:\d+(?::\d+)?\b)|(\bline \d+\b)/gi;
|
|
111
|
+
// ponytail: repo-relative tail from the first known top-level directory — enough to make an absolute
|
|
112
|
+
// worktree path and its repo-relative twin the same identity. Widen the marker list if a run ever
|
|
113
|
+
// names findings outside these roots.
|
|
114
|
+
function canonicalPath(raw) {
|
|
115
|
+
const cleaned = raw.replace(/^["'`(]+/, "").replace(/["'`),.]+$/, "").replace(/^\.\//, "");
|
|
116
|
+
const m = /(?:^|\/)((?:src|tests|scripts|docs|fixtures|specs|schema|skills|assets)\/.+)$/.exec(cleaned);
|
|
117
|
+
return m ? m[1] : cleaned;
|
|
118
|
+
}
|
|
119
|
+
// The code identity a finding names, if it names one: a backticked identifier, then a call/member
|
|
120
|
+
// expression. Line references are stripped first so no identity can carry one. "" means the prose
|
|
121
|
+
// named no symbol — the caller decides what stands in, rather than this guessing from prose.
|
|
122
|
+
function identifierIn(note) {
|
|
123
|
+
const text = note.replace(LINE_REF_RE, " ");
|
|
124
|
+
const ticked = /`([^`]{1,80})`/.exec(text);
|
|
125
|
+
if (ticked)
|
|
126
|
+
return ticked[1].trim();
|
|
127
|
+
// no whitespace before the paren: "the brief (see …)" is prose, not a call expression, and a prose
|
|
128
|
+
// word standing in for a symbol is the guessing this function exists to refuse.
|
|
129
|
+
const call = /\b([A-Za-z_$][\w$]*(?:\.[A-Za-z_$][\w$]*)*)\(/.exec(text);
|
|
130
|
+
return call ? call[1] : "";
|
|
131
|
+
}
|
|
132
|
+
// The SYMBOL of last resort. R4 admits "stable symbol/test title" — a finding whose prose names no
|
|
133
|
+
// code identity still has one stable identity of its own: its own words, with the volatile tokens
|
|
134
|
+
// swept out so line/path churn cannot mint a new symbol for the same finding. It is the reviewer's
|
|
135
|
+
// own bytes, never a guess, and it can never fuse two different findings into one.
|
|
136
|
+
function toFinding(cls, note, path, symbol) {
|
|
137
|
+
const p = path || UNIDENTIFIED;
|
|
138
|
+
const s = symbol || normalizeGateFailure(note) || UNIDENTIFIED;
|
|
139
|
+
return { class: cls, path: p, symbol: s, note, fingerprint: `${cls}|${p}|${s}` };
|
|
140
|
+
}
|
|
141
|
+
/**
|
|
142
|
+
* Structured findings for a BLOCKING review/judge gate result, parsed from the details the gate
|
|
143
|
+
* already writes (D-03: no gate-module change, so an older gate's prose degrades to one unclassified
|
|
144
|
+
* finding rather than to none). Never empty for a blocking result — a finding the journal cannot
|
|
145
|
+
* classify is still a finding the next retry must not lose.
|
|
146
|
+
*
|
|
147
|
+
* Rule: a finding's path is its verdict row's own evidence path. An inline path and an anchored row's
|
|
148
|
+
* path therefore resolve; a different anchor or the task's declared scope never substitutes for a
|
|
149
|
+
* pathless finding. That row fails closed as UNIDENTIFIED instead of manufacturing an R4 identity.
|
|
150
|
+
* A symbol the row's own prose does not name falls back to the criterion id, then to the row's own
|
|
151
|
+
* normalized words (see toFinding).
|
|
152
|
+
*/
|
|
153
|
+
export function structuredFindings(gate, details, _scopeFiles = []) {
|
|
154
|
+
const lines = details.split("\n");
|
|
155
|
+
const rows = [];
|
|
156
|
+
const push = (cls, note, ownPath, fallbackSymbol = "") => {
|
|
157
|
+
const own = canonicalPath(ownPath || PATH_RE.exec(note)?.[1] || "");
|
|
158
|
+
const sym = identifierIn(note) || fallbackSymbol;
|
|
159
|
+
rows.push(toFinding(cls, note, own, sym));
|
|
160
|
+
};
|
|
161
|
+
for (const line of lines) {
|
|
162
|
+
const a = ANCHORED_RE.exec(line);
|
|
163
|
+
if (a) {
|
|
164
|
+
push(`${gate}:anchored`, a[3], a[1]);
|
|
165
|
+
continue;
|
|
166
|
+
}
|
|
167
|
+
if (gate === "review") {
|
|
168
|
+
const r = REVIEW_ROW_RE.exec(line);
|
|
169
|
+
if (r) {
|
|
170
|
+
push(`review:${r[1]}`, r[2], "");
|
|
171
|
+
continue;
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
if (gate === "acceptance") {
|
|
175
|
+
const j = JUDGE_ROW_RE.exec(line);
|
|
176
|
+
// the criterion id IS a stable symbol for an unmet acceptance criterion — the same criterion is
|
|
177
|
+
// the same finding however the judge rephrases its reason — so it backs the prose-derived one.
|
|
178
|
+
if (j) {
|
|
179
|
+
push("acceptance:unmet", j[2], "", j[1]);
|
|
180
|
+
continue;
|
|
181
|
+
}
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
if (rows.length === 0) {
|
|
185
|
+
const head = lines.map((l) => l.trim()).find(Boolean) ?? "";
|
|
186
|
+
push(`${gate}:unclassified`, head, "");
|
|
187
|
+
}
|
|
188
|
+
return rows;
|
|
189
|
+
}
|
|
190
|
+
// v1.85 T3: volatile tokens carry no information about WHY a gate failed — ~663m across 5 runs went to
|
|
191
|
+
// re-dispatching against failures that differed only in these. Every rule below erases a token PROVEN
|
|
192
|
+
// to be a diagnostic location or a clock reading; nothing erases a value the failure asserts ABOUT.
|
|
193
|
+
// Ordered: styling, then timestamps (they contain colon-digits), then paths (they end before a :line),
|
|
194
|
+
// then line refs, durations, long hex.
|
|
195
|
+
const VOLATILE_TOKENS = [
|
|
196
|
+
[/\u001b\[[0-9;]*[a-zA-Z]/g, ""], // ANSI styling
|
|
197
|
+
[/\b\d{4}-\d{2}-\d{2}[T ][\d:]+(?:\.\d+)?(?:Z|[+-]\d{2}:?\d{2})?\b/g, "<ts>"], // timestamps
|
|
198
|
+
[/\brun-\d{8}-\d{6}(?:-\d{16})?\b/g, "<run>"], // run identifiers
|
|
199
|
+
[/\b0x[0-9a-fA-F]+\b/g, "<addr>"], // memory addresses
|
|
200
|
+
// An absolute path INTO the repo keeps its repo-relative tail — that tail IS identity (a defect in
|
|
201
|
+
// daemon.ts is not a defect in journal.ts); only the machine/worktree prefix ahead of it is volatile.
|
|
202
|
+
[/\/(?:[\w.@~+%-]+\/)*((?:src|tests|scripts|docs|fixtures|specs|schema|skills|assets)\/[\w.@~+%/-]+)/g, "<path>/$1"],
|
|
203
|
+
// Rule: an absolute diagnostic path's machine/worktree prefix is volatile, but its named file is
|
|
204
|
+
// identity. Therefore paths outside the repo-marker set keep their final segment: two machines
|
|
205
|
+
// naming parse.js normalize together, while parse.js and render.js can never spend one another's
|
|
206
|
+
// retry budget. A path-shaped VALUE ("/api/v1/users") is rooted nowhere real and survives.
|
|
207
|
+
[/\/(?:tmp|private|var|Users|home|opt|workspace|w)(?:\/[\w.@~+%-]+)*\/([\w.@~+%-]+)\/?/g, "<path>/$1"],
|
|
208
|
+
// A line[:col] ref counts as one only when it hangs off a file-ish token (a dot or a slash in it):
|
|
209
|
+
// R4 says the line number is evidence, not identity. "exit 1" and "expected 3" are neither.
|
|
210
|
+
[/([\w.@~+%-]*[./][\w.@~+%-]*):\d+(?::\d+)?\b/g, "$1:<line>"],
|
|
211
|
+
[/\bline \d+\b/gi, "line <line>"],
|
|
212
|
+
[/\b\d+(?:[.,]\d+)?\s?(?:ms|µs|us|ns|s|sec|secs|m|min|mins|h|hrs)\b/g, "<dur>"], // durations
|
|
213
|
+
[/\b[0-9a-f]{12,40}\b/g, "<hex>"], // sha / worktree ids
|
|
214
|
+
];
|
|
215
|
+
// Rule: a quoted span is protected IFF it is assertion payload. Quoting alone is ordinary diagnostic
|
|
216
|
+
// rendering, so paths/timestamps inside ENOENT and worker messages still normalize. A value introduced
|
|
217
|
+
// by an assertion cue is payload whether quoted or bare: `expected /tmp/actual-a to be /tmp/want-a`
|
|
218
|
+
// must not collapse with an assertion about actual-b.
|
|
219
|
+
//
|
|
220
|
+
// The asymmetry is deliberate: a missed cap costs one extra round, a false cap bans a legitimate retry.
|
|
221
|
+
const ASSERTION_CUE = "expected|received|actual|got|to be|to equal|to match|to contain|instead of|but was|but got|but received";
|
|
222
|
+
const PAYLOAD_SPAN = new RegExp(`(?<=\\b(?:${ASSERTION_CUE})[:=]?[ \\t])(?:'[^'\\n]*'|"[^"\\n]*"|\`[^\`\\n]*\`|[^\\s,;)]+)`, "gi");
|
|
223
|
+
const eraseVolatile = (text) => VOLATILE_TOKENS.reduce((out, [re, replacement]) => out.replace(re, replacement), text);
|
|
224
|
+
// Whitespace RUNS are rendering, so they collapse — but only outside a payload, exactly like every
|
|
225
|
+
// other rule here. Inside one it is part of what the failure asserts: `expected "a b"` and
|
|
226
|
+
// `expected "a b"` are two different assertions, and collapsing the joined string erased that
|
|
227
|
+
// difference and banned a retry that was never redundant. Newlines survive (payload spans cannot
|
|
228
|
+
// cross one) and the line-wise trim below finishes the job.
|
|
229
|
+
const collapseRuns = (text) => text.replace(/[^\S\n]+/g, " ");
|
|
230
|
+
// v1.85 T34: the runner's TALLY moves with the base, not with the defect. When another task merges
|
|
231
|
+
// ahead and adds test files, vitest re-counts the suite — `192 passed (197)` becomes `193 passed
|
|
232
|
+
// (198)` — while the FAIL headlines name the same defect in the same order, so a substantively
|
|
233
|
+
// identical gate failure re-fingerprinted and gate-fingerprint-cap never counted the repeat
|
|
234
|
+
// (measured on T21 in run-20260805-164546: details at 17:33:02 and 18:11:01 differ only so).
|
|
235
|
+
//
|
|
236
|
+
// Why provenance AND shape, and why not digits as a class: a blanket \d+ collapse also erases counts
|
|
237
|
+
// that ARE identity — the failed-assertion count, the failing-suite count, an exit status, an error
|
|
238
|
+
// code — and the rule above governs: a missed cap costs one extra round, a false cap bans a
|
|
239
|
+
// legitimate retry. So the mask is gated on PROVENANCE first: only lines inside the `failing tests:`
|
|
240
|
+
// block baseline.ts:209-219 emits (up to its blank-line/`new failure fingerprints` boundary) are
|
|
241
|
+
// eligible, which keeps review prose, quoted diffs and assertion payloads structurally unreachable —
|
|
242
|
+
// and that block's own `FAIL … > test name` headlines are excluded by SHAPE: only a line that starts
|
|
243
|
+
// with the runner's own `Tests`/`Test Files` summary token is a tally line.
|
|
244
|
+
//
|
|
245
|
+
// Masked tally FIELDS, exactly: the passed count, the skipped and todo counts, and the derived
|
|
246
|
+
// parenthesized total immediately following the tally sequence. Deliberately KEPT (a reader uses
|
|
247
|
+
// each of them to tell two failures apart): every `N failed` count on the line (which suites and how
|
|
248
|
+
// many assertions actually broke — the defect's identity), everything after the derived total (an
|
|
249
|
+
// appended `| exit 1`, an appended `(404)` code), and every other number anywhere.
|
|
250
|
+
const RUNNER_TALLY_LINE = /^([ \t]*(?:Test Files|Tests)[ \t]+)((?:\d+ (?:failed|passed|skipped|todo)[ \t]*\|[ \t]*)*\d+ (?:failed|passed|skipped|todo))([ \t]*\(\d+\))?(.*)$/;
|
|
251
|
+
const TALLY_MOVED_FIELD = /\d+ (?=passed|skipped|todo)/g;
|
|
252
|
+
const maskTallyLine = (line) => {
|
|
253
|
+
const m = RUNNER_TALLY_LINE.exec(line);
|
|
254
|
+
if (!m)
|
|
255
|
+
return line;
|
|
256
|
+
const fields = m[2].replace(TALLY_MOVED_FIELD, "#");
|
|
257
|
+
const total = (m[3] ?? "").replace(/\d+/, "#");
|
|
258
|
+
return m[1] + fields + total + m[4];
|
|
259
|
+
};
|
|
260
|
+
// Computed on the FULL details text BEFORE any payload split: every recorded tally-bearing detail
|
|
261
|
+
// carries the `failing tests:` header, so gating on it closes the false-line-start axis (the mask
|
|
262
|
+
// never sees a slice boundary) and the wrong-provenance axis in one place.
|
|
263
|
+
const maskRunnerTallies = (text) => {
|
|
264
|
+
let inBlock = false;
|
|
265
|
+
return text
|
|
266
|
+
.split("\n")
|
|
267
|
+
.map((line) => {
|
|
268
|
+
if (!inBlock) {
|
|
269
|
+
if (line.trim() === "failing tests:")
|
|
270
|
+
inBlock = true;
|
|
271
|
+
return line;
|
|
272
|
+
}
|
|
273
|
+
if (line.trim() === "" || line.startsWith("new failure fingerprints")) {
|
|
274
|
+
inBlock = false;
|
|
275
|
+
return line;
|
|
276
|
+
}
|
|
277
|
+
return maskTallyLine(line);
|
|
278
|
+
})
|
|
279
|
+
.join("\n");
|
|
280
|
+
};
|
|
281
|
+
/** Normalized identity of a gate failure: the same defect, seen twice, normalizes to the same bytes. */
|
|
282
|
+
export function normalizeGateFailure(details) {
|
|
283
|
+
const masked = maskRunnerTallies(details);
|
|
284
|
+
let out = "";
|
|
285
|
+
let last = 0;
|
|
286
|
+
for (const m of masked.matchAll(PAYLOAD_SPAN)) {
|
|
287
|
+
out += collapseRuns(eraseVolatile(masked.slice(last, m.index))) + m[0];
|
|
288
|
+
last = m.index + m[0].length;
|
|
289
|
+
}
|
|
290
|
+
out += collapseRuns(eraseVolatile(masked.slice(last)));
|
|
291
|
+
return out.split("\n").map((l) => l.trim()).filter(Boolean).join("\n");
|
|
292
|
+
}
|
|
293
|
+
// Two normalized-identical failures of one gate on one task buy no more rounds (the ladder cannot fix
|
|
294
|
+
// what it already re-ran verbatim). Engagement-scoped exactly like reviewRoundsSinceApproval: an
|
|
295
|
+
// operator approval is a new engagement, and nothing else resets the count.
|
|
296
|
+
export const GATE_FINGERPRINT_CAP = 2;
|
|
297
|
+
export function identicalGateFailures(events, taskId, gate, normalized) {
|
|
298
|
+
let n = 0;
|
|
299
|
+
for (const e of events) {
|
|
300
|
+
if (e.taskId !== taskId)
|
|
301
|
+
continue;
|
|
302
|
+
if (e.event === "task-approved")
|
|
303
|
+
n = 0;
|
|
304
|
+
else if (e.event === "gate-result" && e.data.gate === gate && e.data.pass === false
|
|
305
|
+
&& typeof e.data.details === "string"
|
|
306
|
+
&& normalizeGateFailure(e.data.details) === normalized)
|
|
307
|
+
n++;
|
|
308
|
+
}
|
|
309
|
+
return n;
|
|
310
|
+
}
|
|
311
|
+
/** Repair attempts this engagement has already funded — journal-derived, so a resume inherits it. */
|
|
312
|
+
export function repairsSinceApproval(events, taskId) {
|
|
313
|
+
let n = 0;
|
|
314
|
+
for (const e of events) {
|
|
315
|
+
if (e.taskId !== taskId)
|
|
316
|
+
continue;
|
|
317
|
+
if (e.event === "task-approved")
|
|
318
|
+
n = 0;
|
|
319
|
+
else if (e.event === "repair-attempt")
|
|
320
|
+
n++;
|
|
321
|
+
}
|
|
322
|
+
return n;
|
|
323
|
+
}
|
|
324
|
+
// Both retry decisions below govern exactly ONE dispatch: the next one. So both are read back from the
|
|
325
|
+
// journal at the moment that dispatch is built, never carried in a process variable — a stop between
|
|
326
|
+
// the decision and the dispatch (OBS-254's shape, one layer up) would otherwise send a normal prompt
|
|
327
|
+
// with the findings gone, or re-run an assignment that was banned.
|
|
328
|
+
//
|
|
329
|
+
// Rule: retry state is spent iff a worker actually launches. The expiry is `worker-launch`, NOT
|
|
330
|
+
// `task-dispatch`: task-dispatch is journaled before worktree
|
|
331
|
+
// recreation, setup, prompt writing and slot allocation, so spending the decision there hands it to a
|
|
332
|
+
// dispatch that may still die before any worker sees it — and `--retry-failed` would then send a fresh
|
|
333
|
+
// prompt with the repair findings gone, or re-run the banned channel. worker-launch is appended only
|
|
334
|
+
// once the prompt has actually been delivered to a worker, which is the dispatch the decision governs.
|
|
335
|
+
const DECISION_SPENT = "worker-launch";
|
|
336
|
+
function decisionForNextDispatch(events, taskId, event) {
|
|
337
|
+
let pending;
|
|
338
|
+
for (const e of events) {
|
|
339
|
+
if (e.taskId !== taskId)
|
|
340
|
+
continue;
|
|
341
|
+
if (e.event === event)
|
|
342
|
+
pending = e;
|
|
343
|
+
else if (e.event === DECISION_SPENT)
|
|
344
|
+
pending = undefined;
|
|
345
|
+
}
|
|
346
|
+
return pending;
|
|
347
|
+
}
|
|
348
|
+
/**
|
|
349
|
+
* Why the last attempt failed, one row per journaled cause, in the daemon's own `source: details`
|
|
350
|
+
* shape. The daemon builds that brief in a loop-local variable, which dies with the process: a resumed
|
|
351
|
+
* or `--retry-failed` run rebuilt the prompt from nothing and dispatched a retry that had lost the
|
|
352
|
+
* reason it was retrying — OBS-254's class, one layer below the upheld brief. Re-derived here so the
|
|
353
|
+
* bytes the journal already holds cannot be taken away by any reset of attempt or channel state.
|
|
354
|
+
*
|
|
355
|
+
* The same rule governs a dead DISPATCH: its exact task-failed error is retained until
|
|
356
|
+
* `worker-launch`, never retired at
|
|
357
|
+
* `task-dispatch`: everything between the two — worktree recreation, setup, prompt write, slot
|
|
358
|
+
* allocation, the launch itself — can still die with no worker having read a word, and clearing at
|
|
359
|
+
* task-dispatch meant `--retry-failed` after exactly that death rebuilt the prompt without the gate
|
|
360
|
+
* failures OR the delivery failure that preceded it. `task-approved` also clears (an operator approval
|
|
361
|
+
* retires the findings it settled — the uphold case re-derives its own brief separately).
|
|
362
|
+
*/
|
|
363
|
+
export function journaledFailureBrief(events, taskId) {
|
|
364
|
+
let rows = [];
|
|
365
|
+
for (const e of events) {
|
|
366
|
+
if (e.taskId !== taskId)
|
|
367
|
+
continue;
|
|
368
|
+
if (e.event === "worker-launch" || e.event === "task-approved")
|
|
369
|
+
rows = [];
|
|
370
|
+
else if (e.event === "gate-result" && e.data.pass === false && e.data.skipped !== true
|
|
371
|
+
&& typeof e.data.details === "string")
|
|
372
|
+
rows.push(`${e.data.gate}: ${e.data.details}`);
|
|
373
|
+
else if (e.event === "delivery-readiness-failed" && typeof e.data.transcript === "string") {
|
|
374
|
+
rows.push(`dispatch: delivery readiness failed after ${e.data.waitedMs}ms; pane transcript:\n${e.data.transcript}`);
|
|
375
|
+
}
|
|
376
|
+
else if (e.event === "task-failed" && e.data.kind === "dispatch" && typeof e.data.error === "string") {
|
|
377
|
+
rows.push(`dispatch: ${e.data.error}`);
|
|
378
|
+
}
|
|
379
|
+
}
|
|
380
|
+
return rows;
|
|
381
|
+
}
|
|
382
|
+
/** The findings a funded repair must carry into the next dispatch, or undefined if none is pending. */
|
|
383
|
+
export function pendingRepairFindings(events, taskId) {
|
|
384
|
+
const e = decisionForNextDispatch(events, taskId, "repair-attempt");
|
|
385
|
+
return typeof e?.data.findings === "string" ? e.data.findings : undefined;
|
|
386
|
+
}
|
|
387
|
+
/**
|
|
388
|
+
* The gate whose identical failure banned an identical retry of the NEXT dispatch — bound to the
|
|
389
|
+
* channel that produced it, so a verdict that has already moved the work elsewhere is not refused for
|
|
390
|
+
* a channel it is no longer using, and a later unrelated failure is not parked under a stale reason.
|
|
391
|
+
*/
|
|
392
|
+
export function activeRetryBan(events, taskId, channel) {
|
|
393
|
+
const e = decisionForNextDispatch(events, taskId, "gate-fingerprint-cap");
|
|
394
|
+
return e && e.data.channel === channel && typeof e.data.gate === "string" ? e.data.gate : undefined;
|
|
395
|
+
}
|
|
73
396
|
// Fail-closed shape for a dispatched assignment (journal.ts:75-90 posture): a malformed assignment in
|
|
74
397
|
// one dispatch degrades that single task toward today's behavior — counts toward attempts, contributes
|
|
75
398
|
// nothing to tried, poisons only lastAssignment — never crashes resume, never poisons other tasks.
|
|
@@ -81,7 +404,10 @@ const DispatchAssignmentSchema = z.object({
|
|
|
81
404
|
});
|
|
82
405
|
export const PARK_KINDS = ["human-gate", "ladder-exhausted", "attempt-cap", "gate-fail", "quota",
|
|
83
406
|
"reroute-exhausted", "setup", "stall", "merge-conflict", "tip-moved", "infra", "dispatch"];
|
|
84
|
-
|
|
407
|
+
// v1.85 T3: "repair" is a third dispatch mode beside the v1.29 session pair — a fix-only attempt that
|
|
408
|
+
// carries the failing findings and the diff CONTENT of the work already landed, instead of re-buying
|
|
409
|
+
// ~20m of onboarding to rediscover them (62 of 68 measured re-dispatches were fresh).
|
|
410
|
+
export const RETRY_MODES = ["resume", "fresh", "repair"];
|
|
85
411
|
export const WORKER_RESULT_CAUSES = ["provider-death", "stall-timeout", "malformed-trailer", "clean-exit-no-trailer"];
|
|
86
412
|
// Status consumes the routing profile's existing quality split directly: verified park kinds classify
|
|
87
413
|
// to 0, while availability/recovery noise classifies to null. Keep the synthetic row here at the
|
|
@@ -264,9 +590,37 @@ export function engagementComparable(events, loadedHash) {
|
|
|
264
590
|
return { comparable: false, reason: "unbound" };
|
|
265
591
|
return recorded === loadedHash ? { comparable: true, recorded } : { comparable: false, reason: "mismatch", recorded };
|
|
266
592
|
}
|
|
593
|
+
const RUN_SEQUENCE_WIDTH = 16;
|
|
594
|
+
// The daemon mints before acquiring graph.lock, and each CLI invocation has fresh module state. Claim
|
|
595
|
+
// a repository-state ticket with mkdir's atomic EEXIST boundary instead: a losing process advances and
|
|
596
|
+
// retries, while the winning directory remains as the durable high-water evidence for later invocations.
|
|
597
|
+
function claimRunSequence(repoRoot = process.cwd()) {
|
|
598
|
+
const dir = join(tickmarkrDir(repoRoot), "run-id-sequence");
|
|
599
|
+
mkdirSync(dir, { recursive: true });
|
|
600
|
+
const allocated = readdirSync(dir).filter((name) => /^\d{16}$/.test(name));
|
|
601
|
+
let candidate = allocated.reduce((max, name) => {
|
|
602
|
+
const value = BigInt(name);
|
|
603
|
+
return value > max ? value : max;
|
|
604
|
+
}, 0n) + 1n;
|
|
605
|
+
while (true) {
|
|
606
|
+
const suffix = candidate.toString().padStart(RUN_SEQUENCE_WIDTH, "0");
|
|
607
|
+
if (suffix.length > RUN_SEQUENCE_WIDTH)
|
|
608
|
+
throw new Error("run id sequence exhausted");
|
|
609
|
+
try {
|
|
610
|
+
mkdirSync(join(dir, suffix));
|
|
611
|
+
return suffix;
|
|
612
|
+
}
|
|
613
|
+
catch (error) {
|
|
614
|
+
if (error.code !== "EEXIST")
|
|
615
|
+
throw error;
|
|
616
|
+
candidate += 1n;
|
|
617
|
+
}
|
|
618
|
+
}
|
|
619
|
+
}
|
|
267
620
|
export function newRunId(now = new Date()) {
|
|
268
621
|
const p = (n, w = 2) => String(n).padStart(w, "0");
|
|
269
|
-
|
|
622
|
+
const instant = `${now.getUTCFullYear()}${p(now.getUTCMonth() + 1)}${p(now.getUTCDate())}-${p(now.getUTCHours())}${p(now.getUTCMinutes())}${p(now.getUTCSeconds())}`;
|
|
623
|
+
return `run-${instant}-${claimRunSequence()}`;
|
|
270
624
|
}
|
|
271
625
|
// Sol #4: one strict parser for every journal open/create path — generated run-… ids plus test
|
|
272
626
|
// suffix chars only; forbid path separators, dot-segments, and empty ids.
|
|
@@ -304,7 +658,7 @@ function readJsonl(path) {
|
|
|
304
658
|
return out;
|
|
305
659
|
}
|
|
306
660
|
// Cross-run telemetry for Phase-12 profile derivation: the last K runs' rows, each
|
|
307
|
-
// tagged with its runId (runIds are zero-padded run-
|
|
661
|
+
// tagged with its runId (runIds are zero-padded run-UTCYYYYMMDD-HHMMSS-sequence ⇒ plain .sort() is
|
|
308
662
|
// chronological, same as latestRunId). Rows are facts, not classifications — Phase 12
|
|
309
663
|
// owns the quality-denominator/reward policy. A safeParse failure drops that one row
|
|
310
664
|
// (same posture as a torn line); a garbage row must never crash profile derivation.
|
|
@@ -315,7 +669,7 @@ export function readAllTelemetry(repoRoot, lastK, opts = {}) {
|
|
|
315
669
|
if (!existsSync(dir))
|
|
316
670
|
return [];
|
|
317
671
|
let runIds = readdirSync(dir).filter((d) => d.startsWith("run-")).sort();
|
|
318
|
-
// VIS-03 reset cursor:
|
|
672
|
+
// VIS-03 reset cursor: UTC clock fields and a fixed-width sequence make string > chronological
|
|
319
673
|
if (opts.after)
|
|
320
674
|
runIds = runIds.filter((id) => id > opts.after);
|
|
321
675
|
runIds = runIds.slice(-lastK);
|
package/dist/run/stall.d.ts
CHANGED
|
@@ -6,6 +6,14 @@ export interface StallProgressSample {
|
|
|
6
6
|
seedSubmitted?: boolean;
|
|
7
7
|
contextTokens?: number;
|
|
8
8
|
}
|
|
9
|
+
export declare const NUDGEABLE_ADAPTERS: Set<string>;
|
|
10
|
+
export declare const QUOTA_BANNER_TAIL_ROWS = 12;
|
|
11
|
+
export declare function stallSnapshotTail(text: string, rows?: number): string;
|
|
12
|
+
export declare function stallSnapshotBannerRows(text: string, rows?: number): string;
|
|
13
|
+
export declare const ROW_REARM_TOKEN_FLAT_MS: number;
|
|
14
|
+
export declare function setRowRearmTokenFlatMsForTests(ms: number): void;
|
|
15
|
+
export declare function resetRowRearmTokenFlatMsForTests(): void;
|
|
16
|
+
export declare const PANE_READ_ROWS = 1000;
|
|
9
17
|
/**
|
|
10
18
|
* Monotonic worker-progress measure for the stall watchdog.
|
|
11
19
|
*
|
|
@@ -13,12 +21,38 @@ export interface StallProgressSample {
|
|
|
13
21
|
* evidence of work. A rendered transcript is only known to have grown when it occupies more
|
|
14
22
|
* non-empty rows than any prior sample. Same-row rewrites are deliberately ambiguous and do not
|
|
15
23
|
* advance the clock: a recoverable early consult is safer than silencing the watchdog forever.
|
|
24
|
+
*
|
|
25
|
+
* CEILING: `transcriptRows` is a monotone high-water over the daemon's bounded pane read
|
|
26
|
+
* (PANE_READ_ROWS lines), so it is a one-way ratchet whose signal goes blind once the pane's
|
|
27
|
+
* content exceeds the read window — the same full window slides and `observe()` can never
|
|
28
|
+
* report row growth again. Past that point a `false` return means "unmeasurable", not "no
|
|
29
|
+
* output" — consumers making a kill decision (the dead-channel fast-kill) must check
|
|
30
|
+
* `rowSignalSaturated` and stand down on it.
|
|
16
31
|
*/
|
|
17
32
|
export declare class StallProgressTracker {
|
|
18
33
|
private transcriptRows;
|
|
34
|
+
private rawWindowLines;
|
|
19
35
|
private seedSubmitted;
|
|
20
36
|
private contextTokens;
|
|
21
|
-
|
|
37
|
+
private lastTokenGrowthAt;
|
|
38
|
+
private rowGrowthAt;
|
|
39
|
+
/** True once a sample FILLED the bounded read window on RAW lines (blanks and chrome-only
|
|
40
|
+
* rows included): the pane's real extent is then unknown — genuinely new content scrolls out
|
|
41
|
+
* of the read and the row high-water can never advance again — so a flat tracker is blindness,
|
|
42
|
+
* not silence. The raw window is the saturation signal, NOT the normalized non-empty count: a
|
|
43
|
+
* production `read(slot, PANE_READ_ROWS)` returns at most PANE_READ_ROWS lines including blank
|
|
44
|
+
* and chrome-only rows (measured 730 non-empty of 1000 on the codex-mcp-spinner fixture), so
|
|
45
|
+
* comparing the non-empty high-water against PANE_READ_ROWS could never engage and the
|
|
46
|
+
* fast-kill's stand-down was unreachable. Sticky by construction (the high-water never
|
|
47
|
+
* decreases). */
|
|
48
|
+
get rowSignalSaturated(): boolean;
|
|
49
|
+
/** Raw row-growth clock: the last observe() that advanced the row high-water, recorded even
|
|
50
|
+
* when the flat-token rule suppresses the progress REPORT (observe returns false). T1 review:
|
|
51
|
+
* the dead-channel fast-kill's "no output growth" leg must read this, not the suppressed
|
|
52
|
+
* progress clock — a metered adapter whose sticky token counter freezes the report while the
|
|
53
|
+
* pane keeps streaming rows is alive, and only this clock sees it. */
|
|
54
|
+
get lastRowGrowthAt(): number | undefined;
|
|
55
|
+
observe(sample: StallProgressSample, now?: number): boolean;
|
|
22
56
|
}
|
|
23
57
|
/** Filter transcript text bound for an LLM prompt (consult dossiers, gate prompts). The classify
|
|
24
58
|
* seam exists for fault injection in tests only — production callers pass text alone. */
|