tickmarkr 2.3.0 → 2.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/catalog-remote.d.ts +12 -4
- package/dist/adapters/catalog-remote.js +97 -45
- package/dist/adapters/catalog.js +5 -3
- package/dist/adapters/claude-code.d.ts +1 -1
- package/dist/adapters/claude-code.js +8 -5
- package/dist/adapters/codex.js +6 -7
- package/dist/adapters/model-lints.d.ts +9 -5
- package/dist/adapters/model-lints.js +56 -15
- package/dist/adapters/model-windows.js +11 -0
- package/dist/adapters/prompt.js +1 -0
- package/dist/adapters/qwen.d.ts +5 -0
- package/dist/adapters/qwen.js +153 -0
- package/dist/adapters/registry.js +13 -1
- package/dist/adapters/types.d.ts +1 -0
- package/dist/adapters/types.js +1 -0
- package/dist/cli/commands/compile.d.ts +3 -0
- package/dist/cli/commands/compile.js +91 -34
- package/dist/cli/commands/doctor.d.ts +4 -3
- package/dist/cli/commands/doctor.js +28 -9
- package/dist/cli/commands/fleet.d.ts +4 -0
- package/dist/cli/commands/fleet.js +60 -15
- package/dist/cli/commands/init.js +12 -13
- package/dist/cli/commands/plan.js +75 -11
- package/dist/cli/commands/run.js +20 -1
- package/dist/cli/commands/status.d.ts +1 -0
- package/dist/cli/commands/status.js +59 -16
- package/dist/cli/commands/verify.d.ts +1 -0
- package/dist/cli/commands/verify.js +5 -0
- package/dist/cli/commands/version.js +2 -2
- package/dist/cli/index.d.ts +1 -1
- package/dist/cli/index.js +2 -2
- package/dist/compile/collateral.d.ts +14 -12
- package/dist/compile/collateral.js +32 -33
- package/dist/compile/index.d.ts +4 -1
- package/dist/compile/index.js +53 -8
- package/dist/compile/native.d.ts +4 -2
- package/dist/compile/native.js +63 -6
- package/dist/compile/ownership.js +41 -10
- package/dist/config/config.d.ts +1 -0
- package/dist/config/config.js +51 -5
- package/dist/drivers/herdr.d.ts +2 -0
- package/dist/drivers/herdr.js +43 -4
- package/dist/drivers/orca.d.ts +18 -1
- package/dist/drivers/orca.js +163 -15
- package/dist/drivers/types.d.ts +10 -0
- package/dist/gates/baseline.d.ts +26 -2
- package/dist/gates/baseline.js +115 -13
- package/dist/gates/review.d.ts +6 -4
- package/dist/gates/review.js +26 -31
- package/dist/gates/run-gates.d.ts +5 -2
- package/dist/gates/run-gates.js +34 -17
- package/dist/graph/graph.d.ts +20 -0
- package/dist/graph/graph.js +66 -1
- package/dist/route/preference.d.ts +6 -0
- package/dist/route/preference.js +40 -0
- package/dist/route/router.js +15 -2
- package/dist/run/consult.d.ts +1 -0
- package/dist/run/consult.js +35 -7
- package/dist/run/daemon.d.ts +9 -0
- package/dist/run/daemon.js +266 -59
- package/dist/run/git.d.ts +4 -0
- package/dist/run/git.js +51 -6
- package/dist/run/journal.d.ts +1 -1
- package/dist/run/journal.js +5 -2
- package/dist/run/lock.d.ts +6 -0
- package/dist/run/lock.js +41 -1
- package/dist/tui/ink/fleet-app.d.ts +4 -0
- package/dist/tui/ink/fleet-app.js +45 -16
- package/package.json +59 -1
- package/skills/tickmarkr-overseer/SKILL.md +39 -4
- package/skills/tickmarkr-overseer/scripts/grade-ci.sh +79 -0
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +90 -4
package/dist/run/git.d.ts
CHANGED
|
@@ -90,6 +90,10 @@ export interface ShResult {
|
|
|
90
90
|
stdout: string;
|
|
91
91
|
stderr: string;
|
|
92
92
|
timedOut?: boolean;
|
|
93
|
+
/** The shell exited, but descendants still held its process group and pipes past the grace. */
|
|
94
|
+
reapedGroup?: boolean;
|
|
95
|
+
/** The grace reap's group kill failed with something other than ESRCH (for example EPERM). */
|
|
96
|
+
reapError?: string;
|
|
93
97
|
durationMs?: number;
|
|
94
98
|
/** T7: the capacity THIS child ran under, stamped where its environment was built (see `shell`). */
|
|
95
99
|
capacity?: RunCapacity;
|
package/dist/run/git.js
CHANGED
|
@@ -138,6 +138,9 @@ export const resetSpawnForTests = () => { spawnChild = undefined; };
|
|
|
138
138
|
const RETRYABLE_SPAWN_CODE = "EAGAIN";
|
|
139
139
|
export const SPAWN_ATTEMPT_LIMIT = 4;
|
|
140
140
|
export const SPAWN_RETRY_BACKOFF_MS = 50;
|
|
141
|
+
// RULING-P99-28: SHIP — every gate user can hit the inherited-pipe hang at this product seam.
|
|
142
|
+
/** Give a normally-exited shell's descendants this long to close their inherited pipes themselves. */
|
|
143
|
+
const SHELL_REAP_GRACE_MS = 2000;
|
|
141
144
|
// stdin "ignore": same class as HARD-05 / SubprocessDriver — never leave an open pipe a child can block on
|
|
142
145
|
// (pi -p / codex exec wait for stdin EOF). timedOut distinguishes SIGKILL-timeout from a real nonzero exit.
|
|
143
146
|
function shell(cmd, cwd, timeoutMs, login) {
|
|
@@ -170,23 +173,42 @@ function shell(cmd, cwd, timeoutMs, login) {
|
|
|
170
173
|
let stdout = "", stderr = "";
|
|
171
174
|
const stdoutDecoder = new StringDecoder("utf8");
|
|
172
175
|
const stderrDecoder = new StringDecoder("utf8");
|
|
173
|
-
let timedOut = false, done = false, started = false, outputSeen = false;
|
|
176
|
+
let timedOut = false, reapedGroup = false, done = false, started = false, outputSeen = false;
|
|
177
|
+
let reapError, exitedCode;
|
|
178
|
+
let reapTimer;
|
|
174
179
|
const finish = (code, err) => {
|
|
175
180
|
if (done)
|
|
176
181
|
return;
|
|
177
182
|
done = true;
|
|
178
183
|
clearTimeout(timer);
|
|
184
|
+
clearTimeout(reapTimer);
|
|
179
185
|
stdout += stdoutDecoder.end();
|
|
180
186
|
stderr += stderrDecoder.end();
|
|
181
|
-
resolve({
|
|
187
|
+
resolve({
|
|
188
|
+
code,
|
|
189
|
+
stdout,
|
|
190
|
+
stderr: err ?? stderr,
|
|
191
|
+
timedOut,
|
|
192
|
+
durationMs: Date.now() - startedAt,
|
|
193
|
+
capacity,
|
|
194
|
+
...(reapedGroup ? { reapedGroup: true } : {}),
|
|
195
|
+
...(reapError ? { reapError } : {}),
|
|
196
|
+
});
|
|
182
197
|
};
|
|
183
198
|
const timer = setTimeout(() => {
|
|
184
199
|
timedOut = true;
|
|
185
200
|
try {
|
|
186
201
|
process.kill(-p.pid, "SIGKILL");
|
|
187
202
|
}
|
|
188
|
-
catch {
|
|
203
|
+
catch (error) {
|
|
204
|
+
const code = error?.code;
|
|
205
|
+
if (code !== "ESRCH") {
|
|
206
|
+
reapError ??= `${code ?? "kill"}: ${error instanceof Error ? error.message : String(error)}`;
|
|
207
|
+
}
|
|
189
208
|
p.kill("SIGKILL");
|
|
209
|
+
// A failed group kill can leave inherited pipes open forever; the command ceiling still wins.
|
|
210
|
+
if (code !== "ESRCH")
|
|
211
|
+
finish(exitedCode ?? 1);
|
|
190
212
|
}
|
|
191
213
|
}, timeoutMs);
|
|
192
214
|
p.on("spawn", () => { started = true; }); // the command exists from here on — never retryable past it
|
|
@@ -215,9 +237,32 @@ function shell(cmd, cwd, timeoutMs, login) {
|
|
|
215
237
|
finish(127, String(e));
|
|
216
238
|
});
|
|
217
239
|
p.on("close", (code) => finish(code ?? 1));
|
|
218
|
-
// "close" waits for stdio to drain
|
|
219
|
-
|
|
220
|
-
|
|
240
|
+
// "close" waits for stdio to drain. Once bash exits normally, give descendants a bounded grace
|
|
241
|
+
// to exit with it; a survivor still in bash's detached group is then reaped so its inherited pipe
|
|
242
|
+
// cannot hold this promise until the command ceiling. A real timeout wins first and is never
|
|
243
|
+
// reclassified as a grace reap.
|
|
244
|
+
p.on("exit", (code) => {
|
|
245
|
+
exitedCode = code ?? 1;
|
|
246
|
+
if (timedOut) {
|
|
247
|
+
finish(exitedCode);
|
|
248
|
+
return;
|
|
249
|
+
}
|
|
250
|
+
reapTimer = setTimeout(() => {
|
|
251
|
+
if (done)
|
|
252
|
+
return;
|
|
253
|
+
try {
|
|
254
|
+
process.kill(-p.pid, "SIGKILL");
|
|
255
|
+
reapedGroup = true;
|
|
256
|
+
}
|
|
257
|
+
catch (error) {
|
|
258
|
+
// ESRCH means the group closed in the grace race; every other error must survive to the
|
|
259
|
+
// result while the command timeout remains the explicit backstop.
|
|
260
|
+
const code = error?.code;
|
|
261
|
+
if (code !== "ESRCH")
|
|
262
|
+
reapError = `${code ?? "kill"}: ${error instanceof Error ? error.message : String(error)}`;
|
|
263
|
+
}
|
|
264
|
+
}, SHELL_REAP_GRACE_MS);
|
|
265
|
+
});
|
|
221
266
|
});
|
|
222
267
|
return (async () => {
|
|
223
268
|
const startedAt = Date.now();
|
package/dist/run/journal.d.ts
CHANGED
|
@@ -196,7 +196,7 @@ export declare const PARK_KINDS: readonly ["human-gate", "ladder-exhausted", "at
|
|
|
196
196
|
export type ParkKind = (typeof PARK_KINDS)[number];
|
|
197
197
|
export declare const RETRY_MODES: readonly ["resume", "fresh", "repair"];
|
|
198
198
|
export type RetryMode = (typeof RETRY_MODES)[number];
|
|
199
|
-
export declare const WORKER_RESULT_CAUSES: readonly ["provider-death", "dead-channel", "stall-timeout", "malformed-trailer", "clean-exit-no-trailer"];
|
|
199
|
+
export declare const WORKER_RESULT_CAUSES: readonly ["provider-death", "dead-channel", "stall-timeout", "malformed-trailer", "clean-exit-no-trailer", "startup-failure"];
|
|
200
200
|
export type WorkerResultCause = (typeof WORKER_RESULT_CAUSES)[number];
|
|
201
201
|
export declare function isQualityFailureParkKind(kind: ParkKind): boolean;
|
|
202
202
|
export declare function classifyTaskFailure(taskEvents: JournalEvent[]): ParkKind;
|
package/dist/run/journal.js
CHANGED
|
@@ -5,6 +5,7 @@ import { z } from "zod";
|
|
|
5
5
|
import { channelKey, shq, TokenUsageSchema } from "../adapters/types.js";
|
|
6
6
|
import { stateDirName, taskContentDigest, tickmarkrDir } from "../graph/graph.js";
|
|
7
7
|
import { GATE_NAMES, TIERS } from "../graph/schema.js";
|
|
8
|
+
import { channelRouteIdentity } from "../route/preference.js";
|
|
8
9
|
import { buildProfile, classify } from "../route/profile.js";
|
|
9
10
|
import { DecisionEventSchema, trackJournalRows, } from "./protocol.js";
|
|
10
11
|
import { normalizeGateOutcome } from "./outcome.js";
|
|
@@ -752,7 +753,9 @@ export function activeRetryBan(events, taskId, channel) {
|
|
|
752
753
|
|| (e.event === "task-approved" && e.data.release === RECHECK_RELEASE))
|
|
753
754
|
pending = undefined;
|
|
754
755
|
}
|
|
755
|
-
return pending &&
|
|
756
|
+
return pending && typeof pending.data.channel === "string"
|
|
757
|
+
&& channelRouteIdentity(pending.data.channel) === channelRouteIdentity(channel)
|
|
758
|
+
&& typeof pending.data.gate === "string"
|
|
756
759
|
? pending.data.gate
|
|
757
760
|
: undefined;
|
|
758
761
|
}
|
|
@@ -779,7 +782,7 @@ export const RETRY_MODES = ["resume", "fresh", "repair"];
|
|
|
779
782
|
// concludes a worker while the rolling stall window still has most of its time left, so labelling it
|
|
780
783
|
// stall-timeout points every downstream reader — the repair brief, the consult — at a mechanism that
|
|
781
784
|
// cannot have fired.
|
|
782
|
-
export const WORKER_RESULT_CAUSES = ["provider-death", "dead-channel", "stall-timeout", "malformed-trailer", "clean-exit-no-trailer"];
|
|
785
|
+
export const WORKER_RESULT_CAUSES = ["provider-death", "dead-channel", "stall-timeout", "malformed-trailer", "clean-exit-no-trailer", "startup-failure"];
|
|
783
786
|
// Status consumes the routing profile's existing quality split directly: verified park kinds classify
|
|
784
787
|
// to 0, while availability/recovery noise classifies to null. Keep the synthetic row here at the
|
|
785
788
|
// run↔route seam so presentation code never grows a second list of "bad" park kinds.
|
package/dist/run/lock.d.ts
CHANGED
|
@@ -9,6 +9,10 @@ export interface Inspection {
|
|
|
9
9
|
mtimeMs: number;
|
|
10
10
|
ino: number;
|
|
11
11
|
}
|
|
12
|
+
export type RunLineEvent = {
|
|
13
|
+
event: string;
|
|
14
|
+
ts: string;
|
|
15
|
+
};
|
|
12
16
|
export declare function shouldRefuse(i: Pick<Inspection, "garbage" | "dead">): boolean;
|
|
13
17
|
export declare function isPidLive(pid: number): boolean;
|
|
14
18
|
export declare function acquireRunLock(repoRoot: string, runId: string): {
|
|
@@ -29,6 +33,8 @@ export declare function runLockOwner(repoRoot: string): {
|
|
|
29
33
|
runId?: string;
|
|
30
34
|
live: boolean;
|
|
31
35
|
} | undefined;
|
|
36
|
+
export declare function runLockRunId(repoRoot: string): string | undefined;
|
|
37
|
+
export declare function runStatusLine(repoRoot: string, runId: string, events: readonly RunLineEvent[]): string | null;
|
|
32
38
|
export declare function isRunLockLive(repoRoot: string): boolean;
|
|
33
39
|
export declare function unlockRun(repoRoot: string): {
|
|
34
40
|
held: false;
|
package/dist/run/lock.js
CHANGED
|
@@ -2,6 +2,7 @@ import { linkSync, readFileSync, statSync, unlinkSync, utimesSync, writeFileSync
|
|
|
2
2
|
import { join } from "node:path";
|
|
3
3
|
import { z } from "zod";
|
|
4
4
|
import { tickmarkrDir, stateDirName } from "../graph/graph.js";
|
|
5
|
+
import { parseRunId } from "./journal.js";
|
|
5
6
|
// HARD-01/02: coarse per-run advisory lock over .tickmarkr/graph.json. LOCK-02: the lock is created by
|
|
6
7
|
// the link(2) idiom — write the full payload to graph.lock.<pid>.tmp, then linkSync(tmp, lockPath),
|
|
7
8
|
// which is atomic and throws EEXIST if the lock already exists (the mutual-exclusion primitive).
|
|
@@ -238,7 +239,7 @@ export async function acquireApprovalSerialization(repoRoot, runId) {
|
|
|
238
239
|
export function runLockOwner(repoRoot) {
|
|
239
240
|
let insp;
|
|
240
241
|
try {
|
|
241
|
-
insp = inspect(
|
|
242
|
+
insp = inspect(join(repoRoot, stateDirName(repoRoot), "graph.lock"));
|
|
242
243
|
}
|
|
243
244
|
catch {
|
|
244
245
|
return undefined;
|
|
@@ -247,6 +248,45 @@ export function runLockOwner(repoRoot) {
|
|
|
247
248
|
// REPOSITORY-wide, so a live pid here is not proof it is running the run the caller cares about.
|
|
248
249
|
return { pid: insp.pid, runId: insp.runId, live: shouldRefuse(insp) };
|
|
249
250
|
}
|
|
251
|
+
export function runLockRunId(repoRoot) {
|
|
252
|
+
const runId = runLockOwner(repoRoot)?.runId;
|
|
253
|
+
if (runId === undefined)
|
|
254
|
+
return undefined;
|
|
255
|
+
try {
|
|
256
|
+
return parseRunId(runId);
|
|
257
|
+
}
|
|
258
|
+
catch {
|
|
259
|
+
return undefined;
|
|
260
|
+
}
|
|
261
|
+
}
|
|
262
|
+
export function runStatusLine(repoRoot, runId, events) {
|
|
263
|
+
let parsedRunId;
|
|
264
|
+
try {
|
|
265
|
+
parsedRunId = parseRunId(runId);
|
|
266
|
+
}
|
|
267
|
+
catch {
|
|
268
|
+
return null;
|
|
269
|
+
}
|
|
270
|
+
const owner = runLockOwner(repoRoot);
|
|
271
|
+
const ownerRunId = owner?.runId === undefined ? undefined : (() => {
|
|
272
|
+
try {
|
|
273
|
+
return parseRunId(owner.runId);
|
|
274
|
+
}
|
|
275
|
+
catch {
|
|
276
|
+
return undefined;
|
|
277
|
+
}
|
|
278
|
+
})();
|
|
279
|
+
if (owner && ownerRunId === parsedRunId) {
|
|
280
|
+
if (owner.live)
|
|
281
|
+
return `run ${parsedRunId} active`;
|
|
282
|
+
return owner.pid === undefined
|
|
283
|
+
? `run ${parsedRunId} stale lock`
|
|
284
|
+
: `run ${parsedRunId} stale lock naming dead holder pid ${owner.pid}`;
|
|
285
|
+
}
|
|
286
|
+
if (events.some((event) => event.event === "run-end"))
|
|
287
|
+
return null;
|
|
288
|
+
return `run ${parsedRunId} abandoned since ${events.at(-1)?.ts ?? "unknown"}`;
|
|
289
|
+
}
|
|
250
290
|
// Read-only predicate: true iff a lock exists that the decision table would REFUSE on (alive,
|
|
251
291
|
// EPERM, or ANY garbage). A provably-dead holder (ESRCH) reads not-live (LOCK-02/OBS-05). Never
|
|
252
292
|
// mutates the lock. compile now acquires via acquireRunLock; this remains for drift oracles/tests.
|
|
@@ -68,6 +68,10 @@ export type FleetModelGroup = {
|
|
|
68
68
|
detectedAt?: string;
|
|
69
69
|
suggestion?: FleetModelSuggestion;
|
|
70
70
|
evidence?: FleetModelEvidence;
|
|
71
|
+
classifyModel?: string;
|
|
72
|
+
variants?: string[];
|
|
73
|
+
foldedModels?: string[];
|
|
74
|
+
score?: number;
|
|
71
75
|
}>;
|
|
72
76
|
};
|
|
73
77
|
export type FleetClassification = {
|
|
@@ -96,7 +96,8 @@ export function FleetApp({ ageMs, agents, initialDenyAdapters, initialDenyModels
|
|
|
96
96
|
};
|
|
97
97
|
const groupRows = (group) => {
|
|
98
98
|
const rows = group.rows.map((row) => {
|
|
99
|
-
const staged = ui.classifications.find((classification) => classification.adapter === group.adapter
|
|
99
|
+
const staged = ui.classifications.find((classification) => classification.adapter === group.adapter
|
|
100
|
+
&& (classification.model === row.model || classification.model === row.classifyModel));
|
|
100
101
|
return {
|
|
101
102
|
adapter: group.adapter,
|
|
102
103
|
model: row.model,
|
|
@@ -105,10 +106,14 @@ export function FleetApp({ ageMs, agents, initialDenyAdapters, initialDenyModels
|
|
|
105
106
|
suggestion: row.suggestion,
|
|
106
107
|
evidence: row.evidence,
|
|
107
108
|
channel: group.channel,
|
|
109
|
+
classifyModel: row.classifyModel,
|
|
110
|
+
variants: row.variants,
|
|
111
|
+
foldedModels: row.foldedModels,
|
|
112
|
+
score: row.score,
|
|
108
113
|
denied: ui.denyModels.has(`${group.adapter}:${row.model}`),
|
|
109
114
|
};
|
|
110
115
|
});
|
|
111
|
-
const known = new Set(rows.
|
|
116
|
+
const known = new Set(rows.flatMap((row) => row.classifyModel ? [row.model, row.classifyModel] : [row.model]));
|
|
112
117
|
for (const staged of ui.classifications) {
|
|
113
118
|
if (staged.adapter === group.adapter && !known.has(staged.model)) {
|
|
114
119
|
rows.push({
|
|
@@ -125,10 +130,21 @@ export function FleetApp({ ageMs, agents, initialDenyAdapters, initialDenyModels
|
|
|
125
130
|
const matches = (label, f) => f === "" || label.toLowerCase().includes(f.toLowerCase());
|
|
126
131
|
// Operator directive 2026-08-13: retired shapes (dated snapshots, previews, non-worker SKUs,
|
|
127
132
|
// legacy families) hide by DEFAULT. `a` reveals them; a CLASSIFIED row is never hidden.
|
|
128
|
-
const modelRows = (f = ui.filter) =>
|
|
129
|
-
|
|
130
|
-
|
|
133
|
+
const modelRows = (f = ui.filter) => {
|
|
134
|
+
const rows = scopedGroups().flatMap(groupRows).filter((row) => matches(`${row.adapter}/${row.model} ${(row.foldedModels ?? []).join(" ")}`, f)
|
|
135
|
+
&& (ui.showAll || row.tier !== undefined || retiredModelReason(row.model) === null));
|
|
136
|
+
const classified = rows.filter((row) => row.tier !== undefined);
|
|
137
|
+
const unclassified = rows.filter((row) => row.tier === undefined).sort((a, b) => Number(!!b.suggestion) - Number(!!a.suggestion)
|
|
138
|
+
|| (b.detectedAt ?? "").localeCompare(a.detectedAt ?? "")
|
|
139
|
+
|| (b.score ?? Number.NEGATIVE_INFINITY) - (a.score ?? Number.NEGATIVE_INFINITY)
|
|
140
|
+
|| `${a.adapter}/${a.model}`.localeCompare(`${b.adapter}/${b.model}`));
|
|
141
|
+
return [...classified, ...unclassified];
|
|
142
|
+
};
|
|
143
|
+
const hiddenModelCount = (f = ui.filter) => scopedGroups().flatMap(groupRows)
|
|
144
|
+
.filter((row) => matches(`${row.adapter}/${row.model} ${(row.foldedModels ?? []).join(" ")}`, f)).length
|
|
131
145
|
- modelRows(f).length;
|
|
146
|
+
const foldedModelCount = (f = ui.filter) => modelRows(f)
|
|
147
|
+
.reduce((count, row) => count + Math.max((row.foldedModels?.length ?? 1) - 1, 0), 0);
|
|
132
148
|
/** rail counts mirror the list's retired-hide (never the text filter) so numbers agree on screen */
|
|
133
149
|
const visibleCount = (scope) => {
|
|
134
150
|
const groups = scope === -1 ? enabledGroups() : (modelGroups[scope] ? [modelGroups[scope]] : []);
|
|
@@ -232,14 +248,17 @@ export function FleetApp({ ageMs, agents, initialDenyAdapters, initialDenyModels
|
|
|
232
248
|
return;
|
|
233
249
|
// OBS-508: catalog evidence prefills the flow — the tier cursor lands on the suggested band and
|
|
234
250
|
// (when the operator keeps that band) the provenance note arrives pre-typed. Free to override.
|
|
235
|
-
const
|
|
251
|
+
const source = group.rows.find((row) => row.model === model);
|
|
252
|
+
const suggestion = source?.suggestion ?? null;
|
|
253
|
+
const classifyModel = source?.classifyModel ?? model;
|
|
236
254
|
const tierAt = suggestion ? Math.max(TIERS.indexOf(suggestion.tier), 0) : 0;
|
|
237
255
|
const firstTouch = !group.rows.some((row) => row.tier !== undefined)
|
|
238
256
|
&& !ui.classifications.some((c) => c.adapter === adapter);
|
|
239
257
|
const base = {
|
|
240
258
|
kind: "classify",
|
|
241
259
|
adapter,
|
|
242
|
-
model,
|
|
260
|
+
model: classifyModel,
|
|
261
|
+
...(classifyModel !== model ? { displayModel: model } : {}),
|
|
243
262
|
channelAt: 0,
|
|
244
263
|
tierAt,
|
|
245
264
|
note: "",
|
|
@@ -966,7 +985,7 @@ export function FleetApp({ ageMs, agents, initialDenyAdapters, initialDenyModels
|
|
|
966
985
|
const group = modelGroups[ui.adapterAt];
|
|
967
986
|
const suggested = rows
|
|
968
987
|
.filter((candidate) => candidate.adapter === group.adapter && !candidate.tier && candidate.suggestion !== undefined)
|
|
969
|
-
.map((candidate) => ({ model: candidate.model, suggestion: candidate.suggestion }));
|
|
988
|
+
.map((candidate) => ({ model: candidate.classifyModel ?? candidate.model, suggestion: candidate.suggestion }));
|
|
970
989
|
if (!suggested.length) {
|
|
971
990
|
ui.notice = "no catalog tier suggestions among the visible unclassified models — evidence comes from the cached catalogs (AA index / API pricing)";
|
|
972
991
|
bump();
|
|
@@ -1101,6 +1120,9 @@ export function FleetApp({ ageMs, agents, initialDenyAdapters, initialDenyModels
|
|
|
1101
1120
|
const parts = [
|
|
1102
1121
|
`${row.adapter}:${row.model}`,
|
|
1103
1122
|
row.tier ?? "unclassified",
|
|
1123
|
+
row.variants?.length ? `variants ${row.variants.join(", ")}` : "",
|
|
1124
|
+
row.classifyModel ? `classify writes ${row.classifyModel}` : "",
|
|
1125
|
+
row.foldedModels?.length ? `${row.foldedModels.length} gateway ids folded: ${row.foldedModels.join(", ")}` : "",
|
|
1104
1126
|
row.evidence?.unauthed !== undefined ? `UNAUTHED — ${row.evidence.unauthed}` : "",
|
|
1105
1127
|
row.evidence?.contextWindow !== undefined ? `${fmtCtx(row.evidence.contextWindow)} ctx` : "",
|
|
1106
1128
|
row.evidence?.outputWindow !== undefined ? `${fmtCtx(row.evidence.outputWindow)} out` : "",
|
|
@@ -1120,7 +1142,8 @@ export function FleetApp({ ageMs, agents, initialDenyAdapters, initialDenyModels
|
|
|
1120
1142
|
const nameW = bodyW - 4 - 11 - 6 - (showPrice ? 12 : 0) - (showProbe ? 7 : 0);
|
|
1121
1143
|
// OBS-531: deep router ids (omp/prime-agent) clip tail-preserving — the LAST segment is the
|
|
1122
1144
|
// distinguishing half; end-clipping rendered ten identical "prime-agent/anthropic/claude-…" rows.
|
|
1123
|
-
const
|
|
1145
|
+
const folded = row.foldedModels?.length ?? 1;
|
|
1146
|
+
const name = clipPathTail(`${row.adapter}/${row.model}${folded > 1 ? ` ×${folded}` : ""}`, nameW);
|
|
1124
1147
|
const prefixLen = Math.min(name.length, row.adapter.length + 1);
|
|
1125
1148
|
return (_jsxs(Text, { wrap: "truncate", children: [_jsx(Pointer, { on: selected }), row.denied
|
|
1126
1149
|
? _jsx(Glyph, { kind: "off" })
|
|
@@ -1174,7 +1197,9 @@ export function FleetApp({ ageMs, agents, initialDenyAdapters, initialDenyModels
|
|
|
1174
1197
|
if (overlay.kind === "classify") {
|
|
1175
1198
|
const subject = overlay.bulk
|
|
1176
1199
|
? `${overlay.adapter} · ${overlay.bulk.rows.length} suggested models`
|
|
1177
|
-
:
|
|
1200
|
+
: overlay.displayModel
|
|
1201
|
+
? `${overlay.adapter}:${overlay.displayModel} · writes ${overlay.model}`
|
|
1202
|
+
: `${overlay.adapter}:${overlay.model}`;
|
|
1178
1203
|
return (_jsxs(OverlayPanel, { title: `classify · ${subject}`, width: bodyW, children: [overlay.stage === "channel" && (_jsxs(_Fragment, { children: [_jsxs(Text, { dimColor: true, children: ["first touch for ", overlay.adapter, " \u2014 how is this CLI billed?"] }), _jsx(Text, { children: " " }), CHANNELS.map((channel, index) => (_jsxs(Text, { children: [_jsx(Pointer, { on: overlay.channelAt === index }), _jsx(Text, { bold: overlay.channelAt === index, children: padCell(channel, 5) }), _jsx(Text, { dimColor: true, children: channel === "sub" ? "flat-rate subscription quota" : "metered API billing" })] }, channel)))] })), overlay.stage === "tier" && (_jsxs(_Fragment, { children: [_jsx(Text, { dimColor: true, children: overlay.suggestion
|
|
1179
1204
|
? `catalog suggests ${overlay.suggestion.tier} — keep it and the provenance note arrives pre-typed`
|
|
1180
1205
|
: "pick the capability band this model routes as" }), _jsx(Text, { children: " " }), TIERS.map((tier, index) => (_jsxs(Text, { children: [_jsx(Pointer, { on: overlay.tierAt === index }), overlay.suggestion?.tier === tier ? _jsx(Glyph, { kind: "on" }) : _jsx(Text, { children: " " }), _jsx(Text, { bold: overlay.tierAt === index, children: ` ${padCell(tier, 9)}` }), _jsx(Text, { dimColor: true, children: tier === "frontier" ? "strongest band — integrity shapes" : tier === "mid" ? "capable daily driver" : "fast + cheap" })] }, tier)))] })), overlay.stage === "note" && (_jsxs(_Fragment, { children: [_jsx(Text, { dimColor: true, children: "benchmark provenance (required) \u2014 where does this tier claim come from?" }), _jsx(Text, { children: " " }), _jsxs(Text, { children: [_jsx(Text, { color: INK.brand, children: "> " }), _jsx(Text, { children: clip(overlay.note, bodyW - 6) || "" }), _jsx(Text, { color: INK.brand, children: "\u2588" })] })] }))] }));
|
|
@@ -1220,6 +1245,15 @@ export function FleetApp({ ageMs, agents, initialDenyAdapters, initialDenyModels
|
|
|
1220
1245
|
const selectedSeat = ui.judgeSeat ?? judgeKeepRow;
|
|
1221
1246
|
return (_jsxs(OverlayPanel, { title: "pick \u00B7 judge", width: bodyW, children: [_jsx(Text, { dimColor: true, children: "one seat judges acceptance criteria \u2014 failover stays runtime (GATE-09)" }), _jsx(SearchRow, { filter: ui.filter, active: true }), above > 0 && _jsx(ElisionMark, { count: above, side: "above" }), visible.map((label, index) => (_jsxs(Text, { children: [_jsx(Pointer, { on: start + index === overlay.at }), label === selectedSeat ? _jsx(Glyph, { kind: "on" }) : _jsx(Text, { children: " " }), _jsx(Text, { bold: start + index === overlay.at, children: ` ${clip(label, bodyW - 10)}` })] }, label))), below > 0 && _jsx(ElisionMark, { count: below, side: "below" })] }));
|
|
1222
1247
|
})();
|
|
1248
|
+
const modelVisibilityLine = () => {
|
|
1249
|
+
const hidden = hiddenModelCount();
|
|
1250
|
+
const folded = foldedModelCount();
|
|
1251
|
+
return [
|
|
1252
|
+
...(folded ? [`${folded} same-model gateway id${folded === 1 ? "" : "s"} folded`] : []),
|
|
1253
|
+
...(hidden ? [`${hidden} retired/preview/non-worker hidden — a shows all`] : []),
|
|
1254
|
+
...(ui.showAll ? ["showing retired models — a hides them again"] : []),
|
|
1255
|
+
].join(" · ");
|
|
1256
|
+
};
|
|
1223
1257
|
const listNode = (() => {
|
|
1224
1258
|
if (overlayNode)
|
|
1225
1259
|
return overlayNode;
|
|
@@ -1230,7 +1264,7 @@ export function FleetApp({ ageMs, agents, initialDenyAdapters, initialDenyModels
|
|
|
1230
1264
|
? "All models"
|
|
1231
1265
|
: modelGroups[ui.adapterAt]?.adapter ?? "";
|
|
1232
1266
|
const scopeDenied = ui.adapterAt !== -1 && ui.deny.has(modelGroups[ui.adapterAt]?.adapter ?? "");
|
|
1233
|
-
return (_jsxs(Box, { flexDirection: "column", children: [_jsxs(Text, { children: [_jsx(Text, { bold: true, children: scopeLabel }), _jsx(Text, { dimColor: true, children: ` ${rows.length}` })] }), _jsx(SearchRow, { filter: ui.filter, active: ui.searching, hint: "/ to search" }), scopeDenied && _jsx(Text, { color: INK.warn, children: `${modelGroups[ui.adapterAt]?.adapter} is out of the fleet — Space on its rail row adds it back` }), above > 0 && _jsx(ElisionMark, { count: above, side: "above" }), visible.map((row, index) => renderModelRow(row, ui.focus === "list" && start + index === ui.listAt)), below > 0 && _jsx(ElisionMark, { count: below, side: "below", hint: "/ to search" }), rows.length === 0 && !scopeDenied && _jsx(Text, { dimColor: true, children: " no models match" })] }));
|
|
1267
|
+
return (_jsxs(Box, { flexDirection: "column", children: [_jsxs(Text, { children: [_jsx(Text, { bold: true, children: scopeLabel }), _jsx(Text, { dimColor: true, children: ` ${rows.length}` })] }), _jsx(SearchRow, { filter: ui.filter, active: ui.searching, hint: "/ to search" }), modelVisibilityLine() && _jsx(Text, { dimColor: true, children: ` ${modelVisibilityLine()}` }), scopeDenied && _jsx(Text, { color: INK.warn, children: `${modelGroups[ui.adapterAt]?.adapter} is out of the fleet — Space on its rail row adds it back` }), above > 0 && _jsx(ElisionMark, { count: above, side: "above" }), visible.map((row, index) => renderModelRow(row, ui.focus === "list" && start + index === ui.listAt)), below > 0 && _jsx(ElisionMark, { count: below, side: "below", hint: "/ to search" }), rows.length === 0 && !scopeDenied && _jsx(Text, { dimColor: true, children: " no models match" })] }));
|
|
1234
1268
|
}
|
|
1235
1269
|
if (ui.view === "shapes") {
|
|
1236
1270
|
const rows = shapeList();
|
|
@@ -1266,11 +1300,6 @@ export function FleetApp({ ageMs, agents, initialDenyAdapters, initialDenyModels
|
|
|
1266
1300
|
: "unclassified — Space/Enter classifies; unclassified models are never routed");
|
|
1267
1301
|
}
|
|
1268
1302
|
}
|
|
1269
|
-
const hidden = hiddenModelCount();
|
|
1270
|
-
if (hidden > 0 && lines.length < 2)
|
|
1271
|
-
lines.push(`${hidden} retired/preview/non-worker hidden — a shows all`);
|
|
1272
|
-
if (ui.showAll && lines.length < 2)
|
|
1273
|
-
lines.push("showing retired models — a hides them again");
|
|
1274
1303
|
return lines;
|
|
1275
1304
|
}
|
|
1276
1305
|
if (ui.view === "shapes") {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "tickmarkr",
|
|
3
|
-
"version": "2.
|
|
3
|
+
"version": "2.4.1",
|
|
4
4
|
"description": "Spec in, verified work out.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "MIT",
|
|
@@ -57,5 +57,63 @@
|
|
|
57
57
|
"tsx": "^4.19.0",
|
|
58
58
|
"typescript": "^5.6.0",
|
|
59
59
|
"vitest": "^3.0.0"
|
|
60
|
+
},
|
|
61
|
+
"tickmarkrExport": {
|
|
62
|
+
"publicPaths": {
|
|
63
|
+
"exact": [
|
|
64
|
+
".gitignore",
|
|
65
|
+
".oxlintrc.json",
|
|
66
|
+
"CHANGELOG.md",
|
|
67
|
+
"CODE_OF_CONDUCT.md",
|
|
68
|
+
"CONTRIBUTING.md",
|
|
69
|
+
"FLEET.md",
|
|
70
|
+
"LICENSE",
|
|
71
|
+
"README.md",
|
|
72
|
+
"RELEASING.md",
|
|
73
|
+
"SECURITY.md",
|
|
74
|
+
"package-lock.json",
|
|
75
|
+
"package.json",
|
|
76
|
+
"tickmarkr.spec.md",
|
|
77
|
+
"tsconfig.json",
|
|
78
|
+
"vitest.config.ts",
|
|
79
|
+
".github/pull_request_template.md",
|
|
80
|
+
"scripts/assert-test-file-count.sh",
|
|
81
|
+
"scripts/emit-schema.ts",
|
|
82
|
+
"scripts/probe-rig.mjs",
|
|
83
|
+
"specs/export-selftest.spec.md"
|
|
84
|
+
],
|
|
85
|
+
"prefixes": [
|
|
86
|
+
".github/ISSUE_TEMPLATE/",
|
|
87
|
+
".github/workflows/",
|
|
88
|
+
"assets/",
|
|
89
|
+
"docs/codebase/",
|
|
90
|
+
"fixtures/",
|
|
91
|
+
"schema/",
|
|
92
|
+
"skills/tickmarkr-auto/",
|
|
93
|
+
"skills/tickmarkr-loop/",
|
|
94
|
+
"skills/tickmarkr-overseer/",
|
|
95
|
+
"src/",
|
|
96
|
+
"tests/"
|
|
97
|
+
]
|
|
98
|
+
},
|
|
99
|
+
"excludedPaths": [
|
|
100
|
+
".planning",
|
|
101
|
+
"specs",
|
|
102
|
+
".tickmarkr",
|
|
103
|
+
".overseer",
|
|
104
|
+
".claude",
|
|
105
|
+
"docs",
|
|
106
|
+
"ASSESSMENT-*.md",
|
|
107
|
+
".gitignore",
|
|
108
|
+
"scripts/measure-trailer-width.mjs",
|
|
109
|
+
"scripts/scan-scope-corpus.mjs",
|
|
110
|
+
"scripts/repro-obs96.mjs",
|
|
111
|
+
"CLAUDE.md",
|
|
112
|
+
"scripts/export-public.sh",
|
|
113
|
+
"scripts/verify-export.sh",
|
|
114
|
+
"tests/scripts/verify-export.test.ts",
|
|
115
|
+
".github/workflows/ci.yml",
|
|
116
|
+
"**/*.local.*"
|
|
117
|
+
]
|
|
60
118
|
}
|
|
61
119
|
}
|
|
@@ -165,6 +165,21 @@ overseer holds the mission's judgment. A tier collapse is therefore a context le
|
|
|
165
165
|
**The tell, and you will not notice it from inside:** if you are typing `tickmarkr resume`, or reading a
|
|
166
166
|
journal tail to decide what happens next, or sweeping orphans — you have taken the loop. Hand it back.
|
|
167
167
|
|
|
168
|
+
### Pre-run checklist — install these laws in the brief before `compile` / `plan` / `run`
|
|
169
|
+
|
|
170
|
+
1. **Run the files[]-versus-oracle pin sweep (RULING-222-36).** Read every acceptance item and enumerate
|
|
171
|
+
the PINS its satisfaction must move: implementation files, callers, fixtures, tests, snapshots and
|
|
172
|
+
documentation assertions. Put every pin in that task's `files[]` before compile. A criterion that
|
|
173
|
+
forces a file its scope forbids is a plan defect; repair the plan before dispatch, never spend a
|
|
174
|
+
worker's repair ladder on the scope/test catch-22.
|
|
175
|
+
2. **Install the live-run command boundary (RULING-222-28 §5 and 222-29 §3).** While a run is live, run
|
|
176
|
+
no `vitest` probe of any size — the daemon's `suite-wait` guard counts it. During that same window the
|
|
177
|
+
runner's name never enters a shell argv: not reason text, a heredoc, `pgrep -f`, or a commit subject.
|
|
178
|
+
Write records containing the word through an editor, and wait for run-end before probing.
|
|
179
|
+
3. **Gate a mid-run fix at the base, not at a summary (law 47 / OBS-909).** A fix landed on main while a
|
|
180
|
+
run is live is proved with `tickmarkr verify --base <main>`, never with a suite summary copied from a
|
|
181
|
+
different tree. A release proof runs every CI-ordered step — including lint — before its suite.
|
|
182
|
+
|
|
168
183
|
### What the ORCHESTRATOR does, and what you require of it
|
|
169
184
|
|
|
170
185
|
- **The live surface arrives with the run.** `tickmarkr run` is stdout-silent until run-end by design;
|
|
@@ -358,7 +373,7 @@ The protocol, in both directions:
|
|
|
358
373
|
herdr pane run <partner> "I am at <N>%. Clear me: send /clear to <my-pane>, then point me at <my-handoff>."
|
|
359
374
|
# 3. the PARTNER sends the clear, then VERIFIES before pointing:
|
|
360
375
|
herdr pane run <my-pane> "/clear"
|
|
361
|
-
# read the pane back — a cleared claude session shows an empty prompt and a
|
|
376
|
+
# read the pane back — a cleared claude session shows an empty prompt and a LOWER context percentage
|
|
362
377
|
# 4. and only THEN, as a SEPARATE send, the re-orientation:
|
|
363
378
|
herdr pane run <my-pane> "You were cleared at <N>%. Read <handoff> and <brief>. Same-process clear kept
|
|
364
379
|
every background task alive: retire only your recorded watcher pids, verify twice, kill-by-pid, arm,
|
|
@@ -367,7 +382,9 @@ The protocol, in both directions:
|
|
|
367
382
|
|
|
368
383
|
⚠ **Steps 3 and 4 are two sends, never one.** A pointer batched with the clear lands *during* it and is
|
|
369
384
|
lost with the context it was meant to survive. Verify the clear landed by reading the prompt line
|
|
370
|
-
|
|
385
|
+
and re-reading the banner percentage below its pre-clear value before sending the pointer — the same
|
|
386
|
+
read-back every other send in this skill requires. A working seat or a typed prompt defers the act;
|
|
387
|
+
neither authorises a queued `/clear`.
|
|
371
388
|
⚠ **The partner must not clear itself in the same window.** One supervising tier stays live at all
|
|
372
389
|
times; the seat holding the endgame goes second.
|
|
373
390
|
⚠ **Step 4 must state an EXPECTED-RETURN DEADLINE**, e.g. *"confirm you are back within 10 minutes."*
|
|
@@ -451,7 +468,9 @@ The first argument chooses the closed per-seat tier (`orchestrator-context` or `
|
|
|
451
468
|
and every beat names the second argument as that tier's seat. The watcher beats only after reading a
|
|
452
469
|
rendered percentage, keeps beating on the supervision cadence even when its requested poll is slower,
|
|
453
470
|
continues past WARN to ACT, and records a stand-down on each controlled exit. A killed watcher alone
|
|
454
|
-
leaves its last beat to age into `STALE`.
|
|
471
|
+
leaves its last beat to age into `STALE`. Auto-clear waits for `idle` plus an empty or dim-only ANSI
|
|
472
|
+
prompt line, prints `CONTEXT_ACT_DEFERRED` while that gate is closed, and emits `CONTEXT_CLEARED` only
|
|
473
|
+
after a banner read-back proves the percentage dropped; only then does it send the re-brief.
|
|
455
474
|
**Every handoff's re-arm list ends with the announce step from Setup 0** — inform the surviving
|
|
456
475
|
orchestrator the fresh seat is live — or the next seat re-arms silently beside a tier that still
|
|
457
476
|
believes it is alone.
|
|
@@ -545,6 +564,11 @@ they are left implicit:
|
|
|
545
564
|
that discriminates** — text sitting on `❯` is exactly what will not run — and a delivery report
|
|
546
565
|
is prompt-line state alone, or nothing: a decorative check beside a real one reads as
|
|
547
566
|
corroboration (two greens, one of which was never capable of disagreeing).
|
|
567
|
+
**Law 50 applies to every handoff, including `/clear`: a seat-send to a WORKING seat leaves a DRAFT.**
|
|
568
|
+
Send only when the seat is idle and the ANSI prompt line is empty or dim-only (the Esc/SGR discriminator
|
|
569
|
+
separates an autosuggest ghost from typed input), then read back activity or an ACK; presence is not
|
|
570
|
+
delivery. If a stale draft must be replaced, supersede it explicitly with
|
|
571
|
+
`agent prompt " <-- disregard … ACTUAL: …"` instead of stacking another instruction behind it.
|
|
548
572
|
- **A MESSAGE TO A WORKING SEAT IS A QUEUED MESSAGE, AND THE QUEUE DRAINS ONLY AT TURN BOUNDARIES.**
|
|
549
573
|
Delivery is not arrival: `agent prompt` to a `working` claude seat lands in its queue (`Press up to
|
|
550
574
|
edit queued messages` on the seat's prompt line is the tell) and is READ only when the current turn
|
|
@@ -994,10 +1018,21 @@ orchestrator turn boundary.
|
|
|
994
1018
|
- **RUN BOTH RELEASE PROOFS BEFORE STANDING DOWN.** The first is the exported-tree suite, which
|
|
995
1019
|
maintainers run from the private repo with its `verify:export` script — that tooling is deliberately
|
|
996
1020
|
not part of this package, so the command does not exist in a public checkout; it must install,
|
|
997
|
-
build, and run the suite in the exported tree; then
|
|
1021
|
+
build, lint, and only then run the suite in the exported tree; then
|
|
998
1022
|
`TICKMARKR_E2E=1 npx vitest run tests/e2e/orca-smoke.e2e.test.ts` must exercise the installed-Orca
|
|
999
1023
|
smoke. A skipped smoke is not proof. Record both results before declaring the release complete.
|
|
1000
1024
|
|
|
1025
|
+
- **GRADE PUBLIC CI FROM JOB LOGS AT RUN-END ONLY (law 48).** A badge is not the grade, and
|
|
1026
|
+
`gh run view --job --log` is empty while the run is in progress. At run-end execute
|
|
1027
|
+
`scripts/grade-ci.sh <run-id> <expected-count> <tag>` and require GREEN from both jobs; its
|
|
1028
|
+
UNREADABLE state is a hard stop, not a green default. Derive the expected glob pathspec count with
|
|
1029
|
+
`git ls-files ':(glob)tests/**/*.test.ts'` (including the `.tsx` sibling where present), never a bare
|
|
1030
|
+
`tests/**/*.test.ts`: bare `**` is not a git glob (law 49).
|
|
1031
|
+
|
|
1032
|
+
- **RE-READ THE REGISTRY BEFORE RULING (law 51).** A successful Release run and the first
|
|
1033
|
+
`npm view` are separate observations: registry propagation can briefly return the prior version.
|
|
1034
|
+
Read `npm view tickmarkr version` again and verify provenance before recording the publish verdict.
|
|
1035
|
+
|
|
1001
1036
|
- **REWRITE THE MEMORY INDEX FIRST, and read it back.** Minutes after `2.1.1` hit npm, the index line
|
|
1002
1037
|
a fresh session loads still read *"⛔ 2.1.1 CANNOT ship from run …2011"* — true when written, and by
|
|
1003
1038
|
then the exact opposite of the truth. **The index is what everyone loads and the body is what nobody
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
#!/bin/bash
|
|
2
|
+
# grade-ci.sh <run-id> <expected-count> [tag] — grade both public CI jobs from their JOB LOGS.
|
|
3
|
+
# Grade only at run-end. A missing/in-progress/empty log is UNREADABLE, never evidence of green.
|
|
4
|
+
# Tri-state: exit 0 GREEN, 1 RED, 2 UNREADABLE. UNREADABLE dominates a mixed result.
|
|
5
|
+
set -u
|
|
6
|
+
|
|
7
|
+
run=${1:?run id required}
|
|
8
|
+
expected=${2:?expected tracked test-file count required}
|
|
9
|
+
tag=${3:-$run}
|
|
10
|
+
repo=${TKR_GRADE_CI_REPO:-alzahrani-khalid/tickmarkr}
|
|
11
|
+
out_dir=${TKR_GRADE_CI_DIR:-${TKR_STATE_DIR:-.tickmarkr}/overseer/diag}
|
|
12
|
+
mkdir -p "$out_dir" || { echo "UNREADABLE: cannot create log directory $out_dir"; exit 2; }
|
|
13
|
+
|
|
14
|
+
verdict=0
|
|
15
|
+
mark_unreadable() { verdict=2; }
|
|
16
|
+
mark_red() { [ "$verdict" -eq 0 ] && verdict=1; }
|
|
17
|
+
|
|
18
|
+
jobs=$(gh run view "$run" --repo "$repo" --json jobs \
|
|
19
|
+
--jq '.jobs[] | [.databaseId, .name, .status, (.conclusion // "")] | @tsv') \
|
|
20
|
+
|| { echo "UNREADABLE: job list"; exit 2; }
|
|
21
|
+
[ -n "$jobs" ] || { echo "UNREADABLE: empty job list"; exit 2; }
|
|
22
|
+
|
|
23
|
+
seen_test=0
|
|
24
|
+
seen_macos=0
|
|
25
|
+
while IFS=$'\t' read -r id name status conclusion; do
|
|
26
|
+
case "$name" in
|
|
27
|
+
test) seen_test=1 ;;
|
|
28
|
+
test-macos) seen_macos=1 ;;
|
|
29
|
+
*) continue ;;
|
|
30
|
+
esac
|
|
31
|
+
|
|
32
|
+
echo "$name: status=$status conclusion=${conclusion:-none} job=$id"
|
|
33
|
+
if [ "$status" != "completed" ]; then
|
|
34
|
+
echo "$name: UNREADABLE (job has not reached run-end)"
|
|
35
|
+
mark_unreadable
|
|
36
|
+
continue
|
|
37
|
+
fi
|
|
38
|
+
|
|
39
|
+
log="$out_dir/CI-$tag-$name.log"
|
|
40
|
+
gh run view --repo "$repo" --job "$id" --log > "$log" 2>/dev/null
|
|
41
|
+
if [ ! -s "$log" ]; then
|
|
42
|
+
echo "$name: UNREADABLE (empty log)"
|
|
43
|
+
mark_unreadable
|
|
44
|
+
continue
|
|
45
|
+
fi
|
|
46
|
+
|
|
47
|
+
oracle=$(grep -oE 'COUNT_ORACLE [A-Z]+ expected=[0-9A-Z]+ actual=[0-9A-Z]+' "$log" | tail -1)
|
|
48
|
+
files=$(grep -oE 'Test Files .*' "$log" | sed 's/[[:space:]]*$//')
|
|
49
|
+
passed=$(printf '%s\n' "$files" | grep -oE '[0-9]+ passed' | awk '{s+=$1} END{print s+0}')
|
|
50
|
+
skipped=$(printf '%s\n' "$files" | grep -oE '[0-9]+ skipped' | awk '{s+=$1} END{print s+0}')
|
|
51
|
+
failed=$(printf '%s\n' "$files" | grep -oE '[0-9]+ failed' | awk '{s+=$1} END{print s+0}')
|
|
52
|
+
timedout=$(grep -cE 'Test timed out|Error: Hook timed out' "$log" || true)
|
|
53
|
+
errors=$(grep -oE '##\[error\].*' "$log" | sort | uniq -c | sed 's/^ *//' | tr '\n' ';')
|
|
54
|
+
|
|
55
|
+
echo "$name: oracle=[${oracle:-MISSING}] files=[$(printf '%s' "$files" | tr '\n' '|')] passed=$passed skipped=$skipped failed=$failed timedout=$timedout errors=[$errors]"
|
|
56
|
+
if [ -z "$oracle" ] || [ -z "$files" ]; then
|
|
57
|
+
echo "$name: UNREADABLE"
|
|
58
|
+
mark_unreadable
|
|
59
|
+
elif [ "$oracle" = "COUNT_ORACLE GREEN expected=$expected actual=$expected" ] \
|
|
60
|
+
&& [ "$failed" -eq 0 ] && [ "$timedout" -eq 0 ] \
|
|
61
|
+
&& [ $((passed + skipped)) -eq "$expected" ]; then
|
|
62
|
+
echo "$name: GREEN"
|
|
63
|
+
else
|
|
64
|
+
echo "$name: RED"
|
|
65
|
+
mark_red
|
|
66
|
+
fi
|
|
67
|
+
done <<< "$jobs"
|
|
68
|
+
|
|
69
|
+
if [ "$seen_test" -ne 1 ]; then
|
|
70
|
+
echo "test: UNREADABLE (job missing from run)"
|
|
71
|
+
mark_unreadable
|
|
72
|
+
fi
|
|
73
|
+
if [ "$seen_macos" -ne 1 ]; then
|
|
74
|
+
echo "test-macos: UNREADABLE (job missing from run)"
|
|
75
|
+
mark_unreadable
|
|
76
|
+
fi
|
|
77
|
+
|
|
78
|
+
echo "VERDICT rc=$verdict (0=all GREEN 1=RED 2=UNREADABLE)"
|
|
79
|
+
exit "$verdict"
|