worktrust 0.7.2 → 0.7.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -0
- package/log-session.mjs +10 -79
- package/package.json +2 -1
- package/preserve-lines.mjs +12 -0
- package/stretch-evidence.mjs +149 -0
- package/worktrust.mjs +3 -3
package/README.md
CHANGED
|
@@ -161,6 +161,8 @@ nothing it does not:
|
|
|
161
161
|
| `signals`, `analyzer_version` | the behaviour signals of that stretch as keys and counts from the local counter's rubric (framing, steering, verification, recovery), and the rubric's version; derived on this computer from your own turns, never a word of them (0.7.0) |
|
|
162
162
|
| `verification`, `delivery` | per kind of check run in the stretch (test, typecheck, lint, build, the project's gate, CI) how many ran and how many failed; per delivery step (commit, PR, push, deploy) how many succeeded, and whether a check had passed before the first; named from each command on this computer, never the command itself (0.7.1) |
|
|
163
163
|
| `recovery` | of the tool calls that failed in the stretch, how many a later call of the same kind recovered, the middle time that took, how many were retried unchanged and how many with a different approach; inputs compared on this computer by digest, never kept (0.7.2) |
|
|
164
|
+
| `delegation` | the longest and the middle chain of actions the agent took on its own between your turns (or your stopping it), and how often it stopped to ask you a question or to have a plan approved (0.7.3) |
|
|
165
|
+
| `context`, `routing` | how often the context was compacted and which context commands you used (compact, clear, resume, context, model, memory; never a command of your own), how many models answered and how often the model changed (0.7.4) |
|
|
164
166
|
| `collector_version` | the CLI version that measured it |
|
|
165
167
|
| `seq`, `prev`, `hash` | the line's place, the hash of the line before it, and its own hash |
|
|
166
168
|
|
package/log-session.mjs
CHANGED
|
@@ -135,31 +135,6 @@ const DAY_SECONDS = 86400;
|
|
|
135
135
|
const TOOL_CAP = 1800;
|
|
136
136
|
/** Tools whose result waits for the person: the gap to it is the person's time. Names only; nothing of the call is read. */
|
|
137
137
|
const WAITS_FOR_PERSON = new Set(["AskUserQuestion", "ExitPlanMode"]);
|
|
138
|
-
/**
|
|
139
|
-
* WHAT A TOOL CALL WAS, AS A CATEGORY (0.7.1, verification per stretch; the capability roadmap's second tranche). A shell
|
|
140
|
-
* command is read HERE to name its kind: a check (test, typecheck, lint, build, CI, the project's own gate) or a step of
|
|
141
|
-
* delivery (commit, PR, push, deploy). The command itself never leaves this function; only the kind does, and only into
|
|
142
|
-
* the local archive. A chained command can be several kinds; its one outcome counts for each.
|
|
143
|
-
*/
|
|
144
|
-
const SHELL_TOOLS = new Set(["Bash", "bash", "shell", "run_command", "run_shell_command", "execute_command", "terminal"]);
|
|
145
|
-
const CALL_KINDS = [
|
|
146
|
-
["test", /(^|[\s/;&|])(vitest|jest|pytest|mocha|rspec|phpunit)\b|\b(npm|pnpm|yarn|bun)\s+(run\s+)?test\b|\bgo test\b|\bcargo test\b|\bnode --test\b|\bplaywright test\b|scripts\/test-[\w-]+\.m?[jt]s\b/],
|
|
147
|
-
["typecheck", /\btsc\b|\b(npm|pnpm|yarn|bun)\s+(run\s+)?typecheck\b|\bmypy\b|\bpyright\b/],
|
|
148
|
-
["lint", /\beslint\b|\b(npm|pnpm|yarn|bun)\s+(run\s+)?lint\b|\bruff\b|\bflake8\b|\bclippy\b|\bbiome\s+(check|lint)\b/],
|
|
149
|
-
["build", /\b(npm|pnpm|yarn|bun)\s+(run\s+)?build\b|\bnext build\b|\bvite build\b|\bcargo build\b|\bgo build\b/],
|
|
150
|
-
["gate", /\b(npm|pnpm|yarn|bun)\s+(run\s+)?(gate|verify|check)\b|scripts\/check-[\w-]+\.m?[jt]s\b/],
|
|
151
|
-
["ci", /\bgh\s+(run\s+(watch|view)|pr\s+checks)\b/],
|
|
152
|
-
["commit", /\bgit\s+commit\b/],
|
|
153
|
-
["pr", /\bgh\s+pr\s+(create|merge)\b/],
|
|
154
|
-
["push", /\bgit\s+push\b/],
|
|
155
|
-
["deploy", /\b(vercel(\s+deploy)?\s+--prod|fly\s+deploy|netlify\s+deploy|wrangler\s+deploy)\b/],
|
|
156
|
-
];
|
|
157
|
-
const CHECK_KINDS = ["test", "typecheck", "lint", "build", "gate", "ci"];
|
|
158
|
-
const DELIVERY_KINDS = ["commit", "pr", "push", "deploy"];
|
|
159
|
-
const callKindsOf = (block) => {
|
|
160
|
-
const command = block && SHELL_TOOLS.has(String(block.name)) ? block.input?.command ?? block.input?.CommandLine ?? block.input?.cmd : null;
|
|
161
|
-
return typeof command === "string" ? CALL_KINDS.filter(([, pattern]) => pattern.test(command)).map(([kind]) => kind) : [];
|
|
162
|
-
};
|
|
163
138
|
/** The derived record of a stretch, kept for the local archive only (see payloadFor). */
|
|
164
139
|
const DERIVED = new WeakMap();
|
|
165
140
|
const AGENT_RUNS_MAX = 10000, AGENT_SECONDS_MAX = 2592000;
|
|
@@ -234,10 +209,13 @@ const excluded = (path) => excludes.some((text) => text && path.includes(text));
|
|
|
234
209
|
* Without them this hook still reads Claude Code; the never-matching patterns keep every other client out.
|
|
235
210
|
*/
|
|
236
211
|
const { CODEX_ROLLOUT = /(?!)/, codexIdOf = (file) => String(file).split(/[\\/]/).at(-1).replace(/\.jsonl$/, ""), antigravityIdOf = (file) => String(file).split(/[\\/]/).at(-4), codexLines, codexRolloutFiles = function* () {}, codexCwd = () => null, ANTIGRAVITY_TRANSCRIPT = /(?!)/, antigravityLines, antigravityRoots = () => [], antigravityTranscripts = function* () {}, antigravityContext = () => ({}), rememberAntigravity = () => null } = (await import("./transcript-readers.mjs").catch(() => null)) ?? {};
|
|
212
|
+
// THE STRETCH'S DERIVED RECORD (0.7.4: split from this file): tool calls named by kind, verification, recovery, delegation,
|
|
213
|
+
// context and routing, for the local archive only. A copy without the module still measures and sends exactly as before.
|
|
214
|
+
const { callOf = (block) => ({ id: block.id, kinds: [], family: String(block.name), digest: "" }), deriveStretch = () => ({}) } = (await import("./stretch-evidence.mjs").catch(() => null)) ?? {};
|
|
237
215
|
const { DATABASE_SESSION = /(?!)/, databaseLines = () => null, databaseSessions = function* () {} } = (await import("./session-databases.mjs").catch(() => null)) ?? {};
|
|
238
216
|
const DATABASE_CLIENTS = ["hermes", "goose", "opencode", "openclaw", "cursor", "copilot"]; // the registry's keys, the name each line carries
|
|
239
217
|
const LIVE_SOURCE = new RegExp(`^(codex|antigravity|${DATABASE_CLIENTS.join("|")}):`); // a sweep entry's id → the client it names (every database client, 0.6.16)
|
|
240
|
-
const READER_FILES = ["transcript-readers.mjs", "session-databases.mjs"];
|
|
218
|
+
const READER_FILES = ["transcript-readers.mjs", "session-databases.mjs", "stretch-evidence.mjs"]; // 0.7.4: the derived record travels with the hook
|
|
241
219
|
/** The readers go where the hook and the counter run: copied from beside this file, else (or for a newer counter) from the deployment. */
|
|
242
220
|
async function installReaders(origin, fromNetwork = false) {
|
|
243
221
|
let installed = true;
|
|
@@ -407,7 +385,10 @@ const messageOf = (line, at) => {
|
|
|
407
385
|
const content = line.message?.content ?? null;
|
|
408
386
|
const blocks = Array.isArray(content) ? content : [];
|
|
409
387
|
return {
|
|
410
|
-
|
|
388
|
+
// A COMPACTION'S SUMMARY IS NOT THE PERSON (0.7.4): Claude Code writes it on the user role; the counter always left it
|
|
389
|
+
// out, the hook counted it as a turn and the gap before it as the person's time. It is the loop's own, like a meta line.
|
|
390
|
+
at, type: line.type, usage: line.message?.usage ?? null, model: line.message?.model ?? null, messageId: typeof line.message?.id === "string" ? line.message.id : null, meta: line.isMeta === true || line.isCompactSummary === true, content,
|
|
391
|
+
compacted: line.isCompactSummary === true,
|
|
411
392
|
// A tool's answer comes back on the user role; a line a reader marks `agentic` is an agent run (Copilot's asked → completed).
|
|
412
393
|
kind: line.type === "assistant" ? "assistant" : blocks.some((block) => block && block.type === "tool_result") ? "tool_result" : "user",
|
|
413
394
|
toolUse: line.type === "assistant" && blocks.some((block) => block && block.type === "tool_use"),
|
|
@@ -416,7 +397,7 @@ const messageOf = (line, at) => {
|
|
|
416
397
|
agentic: line.agentic === true, cumulative: line.cumulative === true, // the session's running totals (Hermes, Goose without a ledger)
|
|
417
398
|
// 0.7.1: each tool call's kinds (never its command) and each tool result's outcome, joined on the call's id in `measure`.
|
|
418
399
|
// 0.7.2: and the call's FAMILY (the tool, and its check kinds) and a digest of its input, compared here and never kept.
|
|
419
|
-
calls: line.type === "assistant" ? blocks.filter((block) => block && block.type === "tool_use" && typeof block.id === "string").map(
|
|
400
|
+
calls: line.type === "assistant" ? blocks.filter((block) => block && block.type === "tool_use" && typeof block.id === "string").map(callOf) : [],
|
|
420
401
|
results: blocks.filter((block) => block && block.type === "tool_result" && typeof block.tool_use_id === "string").map((block) => ({ id: block.tool_use_id, failed: block.is_error === true })),
|
|
421
402
|
};
|
|
422
403
|
};
|
|
@@ -495,56 +476,6 @@ function isHumanTurn(content) {
|
|
|
495
476
|
return types.has("text") || types.has("image");
|
|
496
477
|
}
|
|
497
478
|
|
|
498
|
-
/**
|
|
499
|
-
* VERIFICATION PER STRETCH (0.7.1; framework L7, L12): per kind of check, how many ran and how many failed; per step of
|
|
500
|
-
* delivery, how many succeeded; and whether a check had PASSED before the first delivery step. Null where the stretch
|
|
501
|
-
* ran neither, so "no check seen" is never written as "zero checks".
|
|
502
|
-
*/
|
|
503
|
-
function verificationOf(messages) {
|
|
504
|
-
const kinds = new Map();
|
|
505
|
-
for (const message of messages) for (const call of message.calls ?? []) kinds.set(call.id, call.kinds);
|
|
506
|
-
const checks = {}, delivered = {};
|
|
507
|
-
let deliveredAt = null, passedFirst = false;
|
|
508
|
-
for (const message of messages) for (const result of message.results ?? []) {
|
|
509
|
-
for (const kind of kinds.get(result.id) ?? []) {
|
|
510
|
-
if (CHECK_KINDS.includes(kind)) { const tally = (checks[kind] ??= [0, 0]); tally[0] += 1; if (result.failed) tally[1] += 1; else if (deliveredAt === null) passedFirst = true; }
|
|
511
|
-
else if (DELIVERY_KINDS.includes(kind) && !result.failed) { delivered[kind] = (delivered[kind] ?? 0) + 1; if (deliveredAt === null) deliveredAt = message.at; }
|
|
512
|
-
}
|
|
513
|
-
}
|
|
514
|
-
const any = (record) => Object.keys(record).length > 0;
|
|
515
|
-
const recovery = recoveryOf(messages);
|
|
516
|
-
return { ...(any(checks) ? { verification: checks } : {}), ...(any(delivered) ? { delivery: { ...delivered, verified_first: passedFirst } } : {}), ...(recovery ? { recovery } : {}) };
|
|
517
|
-
}
|
|
518
|
-
|
|
519
|
-
/**
|
|
520
|
-
* RECOVERY PER STRETCH (0.7.2; framework L9): of the tool calls that failed, how many were followed by a success of the
|
|
521
|
-
* same family (the same tool, the same kind of check) later in the stretch, the middle time that took, and what came
|
|
522
|
-
* next: the same input again (a blind retry) or a different one (a changed strategy). Inputs are compared by digest on
|
|
523
|
-
* this computer and never kept. Null when nothing failed.
|
|
524
|
-
*/
|
|
525
|
-
function recoveryOf(messages) {
|
|
526
|
-
const calls = [];
|
|
527
|
-
const byId = new Map();
|
|
528
|
-
for (const message of messages) {
|
|
529
|
-
for (const call of message.calls ?? []) { const entry = { ...call, at: message.at, failed: null }; calls.push(entry); byId.set(call.id, entry); }
|
|
530
|
-
for (const result of message.results ?? []) { const call = byId.get(result.id); if (call) call.failed = result.failed; }
|
|
531
|
-
}
|
|
532
|
-
const done = calls.filter((call) => call.failed !== null);
|
|
533
|
-
let failures = 0, recovered = 0, blind = 0, changed = 0;
|
|
534
|
-
const latencies = [];
|
|
535
|
-
done.forEach((call, index) => {
|
|
536
|
-
if (!call.failed) return;
|
|
537
|
-
failures += 1;
|
|
538
|
-
const later = done.slice(index + 1).filter((next) => next.family === call.family);
|
|
539
|
-
if (later.length > 0) { if (later[0].digest === call.digest) blind += 1; else changed += 1; }
|
|
540
|
-
const success = later.find((next) => !next.failed);
|
|
541
|
-
if (success) { recovered += 1; latencies.push(Math.max(0, Math.round((success.at - call.at) / 1000))); }
|
|
542
|
-
});
|
|
543
|
-
if (failures === 0) return null;
|
|
544
|
-
latencies.sort((a, b) => a - b);
|
|
545
|
-
return { failures, recovered, blind_retries: blind, strategy_changed: changed, ...(latencies.length ? { median_seconds: latencies[Math.floor((latencies.length - 1) / 2)] } : {}) };
|
|
546
|
-
}
|
|
547
|
-
|
|
548
479
|
/** One day's messages → the stretch, measured. Null when there is nothing honest to send. */
|
|
549
480
|
function measure(cwd, messages) {
|
|
550
481
|
// MEASURED, AND ONLY MEASURED: every second counted lies between two recorded timestamps, and a
|
|
@@ -575,7 +506,7 @@ function measure(cwd, messages) {
|
|
|
575
506
|
else if (isHumanTurn(message.content) && (prev.kind === "tool_result" || prev.toolUse)) steers += 1;
|
|
576
507
|
}
|
|
577
508
|
const from = firstOf(messages);
|
|
578
|
-
return { cwd, seconds: measured, derived:
|
|
509
|
+
return { cwd, seconds: measured, derived: deriveStretch(messages, { isHumanTurn, INTERRUPTED, textOf }), layers: measured !== null ? timeLayers(messages) : null, ...tokens, cumulative: messages.some((message) => message.cumulative), exchanges, interrupts, steers, model, from, to: messages[messages.length - 1].at, offset: offsetAt(from) };
|
|
579
510
|
}
|
|
580
511
|
|
|
581
512
|
/**
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "worktrust",
|
|
3
|
-
"version": "0.7.
|
|
3
|
+
"version": "0.7.4",
|
|
4
4
|
"description": "Couple this computer to WorkTrust: approve a code in your browser, and every AI app here reports what you built. Metadata only, no dependencies.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
@@ -17,6 +17,7 @@
|
|
|
17
17
|
"preserve.mjs",
|
|
18
18
|
"preserve-summary.mjs",
|
|
19
19
|
"preserve-lines.mjs",
|
|
20
|
+
"stretch-evidence.mjs",
|
|
20
21
|
"README.md",
|
|
21
22
|
"SECURITY.md",
|
|
22
23
|
"LICENSE"
|
package/preserve-lines.mjs
CHANGED
|
@@ -63,6 +63,18 @@ export function archiveLine(entry, behaviour = null, { ids = {}, collector } = {
|
|
|
63
63
|
const { failures, recovered, blind_retries: blind, strategy_changed: changed, median_seconds: median } = entry.recovery;
|
|
64
64
|
if ([failures, recovered, blind, changed].every(whole) && failures > 0 && recovered <= failures && blind + changed <= failures && (median === undefined || whole(median))) line.recovery = { failures, recovered, blind_retries: blind, strategy_changed: changed, ...(median !== undefined ? { median_seconds: median } : {}) };
|
|
65
65
|
}
|
|
66
|
+
// DELEGATION (0.7.3): the longest and middle chain of the agent's own actions between the person's steps, questions, plans.
|
|
67
|
+
if (entry.delegation && typeof entry.delegation === "object") {
|
|
68
|
+
const { chain_max: most, chain_median: middle, questions, plans } = entry.delegation;
|
|
69
|
+
if ([most, middle, questions, plans].every(whole) && most > 0 && middle <= most) line.delegation = { chain_max: most, chain_median: middle, questions, plans };
|
|
70
|
+
}
|
|
71
|
+
// CONTEXT AND ROUTING (0.7.4): compactions and the known context commands; how many models, how many switches.
|
|
72
|
+
const COMMANDS = ["compact", "clear", "resume", "context", "model", "memory"];
|
|
73
|
+
if (entry.context && typeof entry.context === "object" && whole(entry.context.compactions)) {
|
|
74
|
+
const commands = Object.entries(entry.context.commands ?? {}).filter(([name, count]) => COMMANDS.includes(name) && whole(count) && count > 0).sort();
|
|
75
|
+
if (entry.context.compactions > 0 || commands.length > 0) line.context = { compactions: entry.context.compactions, ...(commands.length ? { commands: Object.fromEntries(commands) } : {}) };
|
|
76
|
+
}
|
|
77
|
+
if (entry.routing && typeof entry.routing === "object" && whole(entry.routing.models) && whole(entry.routing.switches) && entry.routing.models > 0) line.routing = { models: entry.routing.models, switches: entry.routing.switches };
|
|
66
78
|
// The behaviour signals of this stretch (0.7.0): keys of the counter's rubric and whole counts, never a word of a turn.
|
|
67
79
|
if (behaviour && behaviour.signals && typeof behaviour.analyzer_version === "string" && /^counter@\d+\.\d+\.\d+$/.test(behaviour.analyzer_version)) {
|
|
68
80
|
const kept = Object.entries(behaviour.signals).filter(([key, count]) => SIGNAL.test(key) && Number.isInteger(count) && count > 0).sort();
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* THE WORK STRETCH'S DERIVED RECORD (CLI 0.7.4; the capability roadmap, docs/CAPABILITY-EVIDENCE-ROADMAP.md). Read by the
|
|
3
|
+
* session hook from one stretch's messages, on this computer: what each tool call WAS as a kind (never its command),
|
|
4
|
+
* verification and delivery, recovery, delegation, context and routing. Everything here goes into the local archive
|
|
5
|
+
* (`preserve --archive`) and nothing into what the hook sends; each field is null where the stretch shows none of it.
|
|
6
|
+
* Split from log-session.mjs so each tranche adds a function here, with its own test, instead of growing the hook.
|
|
7
|
+
*/
|
|
8
|
+
import { createHash } from "node:crypto";
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* WHAT A TOOL CALL WAS, AS A CATEGORY (0.7.1, verification per stretch; the capability roadmap's second tranche). A shell
|
|
12
|
+
* command is read HERE to name its kind: a check (test, typecheck, lint, build, CI, the project's own gate) or a step of
|
|
13
|
+
* delivery (commit, PR, push, deploy). The command itself never leaves this function; only the kind does, and only into
|
|
14
|
+
* the local archive. A chained command can be several kinds; its one outcome counts for each.
|
|
15
|
+
*/
|
|
16
|
+
const SHELL_TOOLS = new Set(["Bash", "bash", "shell", "run_command", "run_shell_command", "execute_command", "terminal"]);
|
|
17
|
+
const CALL_KINDS = [
|
|
18
|
+
["test", /(^|[\s/;&|])(vitest|jest|pytest|mocha|rspec|phpunit)\b|\b(npm|pnpm|yarn|bun)\s+(run\s+)?test\b|\bgo test\b|\bcargo test\b|\bnode --test\b|\bplaywright test\b|scripts\/test-[\w-]+\.m?[jt]s\b/],
|
|
19
|
+
["typecheck", /\btsc\b|\b(npm|pnpm|yarn|bun)\s+(run\s+)?typecheck\b|\bmypy\b|\bpyright\b/],
|
|
20
|
+
["lint", /\beslint\b|\b(npm|pnpm|yarn|bun)\s+(run\s+)?lint\b|\bruff\b|\bflake8\b|\bclippy\b|\bbiome\s+(check|lint)\b/],
|
|
21
|
+
["build", /\b(npm|pnpm|yarn|bun)\s+(run\s+)?build\b|\bnext build\b|\bvite build\b|\bcargo build\b|\bgo build\b/],
|
|
22
|
+
["gate", /\b(npm|pnpm|yarn|bun)\s+(run\s+)?(gate|verify|check)\b|scripts\/check-[\w-]+\.m?[jt]s\b/],
|
|
23
|
+
["ci", /\bgh\s+(run\s+(watch|view)|pr\s+checks)\b/],
|
|
24
|
+
["commit", /\bgit\s+commit\b/],
|
|
25
|
+
["pr", /\bgh\s+pr\s+(create|merge)\b/],
|
|
26
|
+
["push", /\bgit\s+push\b/],
|
|
27
|
+
["deploy", /\b(vercel(\s+deploy)?\s+--prod|fly\s+deploy|netlify\s+deploy|wrangler\s+deploy)\b/],
|
|
28
|
+
];
|
|
29
|
+
const CHECK_KINDS = ["test", "typecheck", "lint", "build", "gate", "ci"];
|
|
30
|
+
const DELIVERY_KINDS = ["commit", "pr", "push", "deploy"];
|
|
31
|
+
export const callKindsOf = (block) => {
|
|
32
|
+
const command = block && SHELL_TOOLS.has(String(block.name)) ? block.input?.command ?? block.input?.CommandLine ?? block.input?.cmd : null;
|
|
33
|
+
return typeof command === "string" ? CALL_KINDS.filter(([, pattern]) => pattern.test(command)).map(([kind]) => kind) : [];
|
|
34
|
+
};
|
|
35
|
+
/** One tool call as the stretch reads it: its id, its kinds, its FAMILY (the tool and its check kinds) and a digest of its input (compared, never kept). */
|
|
36
|
+
export const callOf = (block) => { const kinds = callKindsOf(block); return { id: block.id, kinds, family: `${String(block.name)}|${kinds.join("+")}`, digest: createHash("sha256").update(JSON.stringify(block.input ?? null)).digest("base64url").slice(0, 16) }; };
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* VERIFICATION PER STRETCH (0.7.1; framework L7, L12): per kind of check, how many ran and how many failed; per step of
|
|
40
|
+
* delivery, how many succeeded; and whether a check had PASSED before the first delivery step. Null where the stretch
|
|
41
|
+
* ran neither, so "no check seen" is never written as "zero checks".
|
|
42
|
+
*/
|
|
43
|
+
function verificationOf(messages, helpers) {
|
|
44
|
+
const kinds = new Map();
|
|
45
|
+
for (const message of messages) for (const call of message.calls ?? []) kinds.set(call.id, call.kinds);
|
|
46
|
+
const checks = {}, delivered = {};
|
|
47
|
+
let deliveredAt = null, passedFirst = false;
|
|
48
|
+
for (const message of messages) for (const result of message.results ?? []) {
|
|
49
|
+
for (const kind of kinds.get(result.id) ?? []) {
|
|
50
|
+
if (CHECK_KINDS.includes(kind)) { const tally = (checks[kind] ??= [0, 0]); tally[0] += 1; if (result.failed) tally[1] += 1; else if (deliveredAt === null) passedFirst = true; }
|
|
51
|
+
else if (DELIVERY_KINDS.includes(kind) && !result.failed) { delivered[kind] = (delivered[kind] ?? 0) + 1; if (deliveredAt === null) deliveredAt = message.at; }
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
const any = (record) => Object.keys(record).length > 0;
|
|
55
|
+
const recovery = recoveryOf(messages), delegation = delegationOf(messages, helpers);
|
|
56
|
+
return { ...(any(checks) ? { verification: checks } : {}), ...(any(delivered) ? { delivery: { ...delivered, verified_first: passedFirst } } : {}), ...(recovery ? { recovery } : {}), ...(delegation ? { delegation } : {}) };
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* DELEGATION AND CHECKPOINTS PER STRETCH (0.7.3; framework L3, L5): how many tool actions the agent took on its own
|
|
61
|
+
* between two moments the person stepped in (their own turn, or stopping it), as the longest and the middle chain; and
|
|
62
|
+
* how often the agent stopped for the person on purpose: a question it asked them, a plan it asked them to approve.
|
|
63
|
+
* Null when the stretch made no tool call.
|
|
64
|
+
*/
|
|
65
|
+
function delegationOf(messages, { isHumanTurn, INTERRUPTED, textOf }) {
|
|
66
|
+
const chains = [];
|
|
67
|
+
let chain = 0, calls = 0, questions = 0, plans = 0;
|
|
68
|
+
for (const message of messages) {
|
|
69
|
+
if (message.bridge) continue;
|
|
70
|
+
const stepIn = message.type === "user" && !message.meta && message.kind !== "tool_result" && (isHumanTurn(message.content) || INTERRUPTED.test(textOf(message.content).trim()));
|
|
71
|
+
if (stepIn) { if (chain > 0) chains.push(chain); chain = 0; continue; }
|
|
72
|
+
for (const call of message.calls ?? []) {
|
|
73
|
+
const tool = call.family.split("|")[0];
|
|
74
|
+
calls += 1; chain += 1;
|
|
75
|
+
if (tool === "AskUserQuestion") questions += 1;
|
|
76
|
+
if (tool === "ExitPlanMode") plans += 1;
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
if (chain > 0) chains.push(chain);
|
|
80
|
+
if (calls === 0) return null;
|
|
81
|
+
chains.sort((a, b) => a - b);
|
|
82
|
+
return { chain_max: chains.at(-1), chain_median: chains[Math.floor((chains.length - 1) / 2)], questions, plans };
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/**
|
|
86
|
+
* RECOVERY PER STRETCH (0.7.2; framework L9): of the tool calls that failed, how many were followed by a success of the
|
|
87
|
+
* same family (the same tool, the same kind of check) later in the stretch, the middle time that took, and what came
|
|
88
|
+
* next: the same input again (a blind retry) or a different one (a changed strategy). Inputs are compared by digest on
|
|
89
|
+
* this computer and never kept. Null when nothing failed.
|
|
90
|
+
*/
|
|
91
|
+
function recoveryOf(messages) {
|
|
92
|
+
const calls = [];
|
|
93
|
+
const byId = new Map();
|
|
94
|
+
for (const message of messages) {
|
|
95
|
+
for (const call of message.calls ?? []) { const entry = { ...call, at: message.at, failed: null }; calls.push(entry); byId.set(call.id, entry); }
|
|
96
|
+
for (const result of message.results ?? []) { const call = byId.get(result.id); if (call) call.failed = result.failed; }
|
|
97
|
+
}
|
|
98
|
+
const done = calls.filter((call) => call.failed !== null);
|
|
99
|
+
let failures = 0, recovered = 0, blind = 0, changed = 0;
|
|
100
|
+
const latencies = [];
|
|
101
|
+
done.forEach((call, index) => {
|
|
102
|
+
if (!call.failed) return;
|
|
103
|
+
failures += 1;
|
|
104
|
+
const later = done.slice(index + 1).filter((next) => next.family === call.family);
|
|
105
|
+
if (later.length > 0) { if (later[0].digest === call.digest) blind += 1; else changed += 1; }
|
|
106
|
+
const success = later.find((next) => !next.failed);
|
|
107
|
+
if (success) { recovered += 1; latencies.push(Math.max(0, Math.round((success.at - call.at) / 1000))); }
|
|
108
|
+
});
|
|
109
|
+
if (failures === 0) return null;
|
|
110
|
+
latencies.sort((a, b) => a - b);
|
|
111
|
+
return { failures, recovered, blind_retries: blind, strategy_changed: changed, ...(latencies.length ? { median_seconds: latencies[Math.floor((latencies.length - 1) / 2)] } : {}) };
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
/**
|
|
116
|
+
* CONTEXT AND ROUTING PER STRETCH (0.7.4; framework L6, L10): how often the context was compacted (Claude Code's own
|
|
117
|
+
* summary line), which of the client's context commands the person used (a fixed list: a command of their own could
|
|
118
|
+
* carry a name), how many models answered and how often the model changed between answers. Null where none shows.
|
|
119
|
+
*/
|
|
120
|
+
const CONTEXT_COMMANDS = ["compact", "clear", "resume", "context", "model", "memory"];
|
|
121
|
+
const COMMAND = /<command-name>\/?([a-z-]+)<\/command-name>/;
|
|
122
|
+
function contextOf(messages, { textOf }) {
|
|
123
|
+
let compactions = 0;
|
|
124
|
+
const commands = {};
|
|
125
|
+
for (const message of messages) {
|
|
126
|
+
if (message.bridge) continue;
|
|
127
|
+
if (message.compacted) compactions += 1;
|
|
128
|
+
const name = message.type === "user" ? COMMAND.exec(textOf(message.content))?.[1] : null;
|
|
129
|
+
if (name && CONTEXT_COMMANDS.includes(name)) commands[name] = (commands[name] ?? 0) + 1;
|
|
130
|
+
}
|
|
131
|
+
return compactions > 0 || Object.keys(commands).length > 0 ? { compactions, ...(Object.keys(commands).length ? { commands } : {}) } : null;
|
|
132
|
+
}
|
|
133
|
+
function routingOf(messages) {
|
|
134
|
+
const models = new Set();
|
|
135
|
+
let last = null, switches = 0;
|
|
136
|
+
for (const message of messages) {
|
|
137
|
+
if (message.bridge || message.type !== "assistant" || !message.model || message.model === "<synthetic>") continue;
|
|
138
|
+
models.add(message.model);
|
|
139
|
+
if (last !== null && message.model !== last) switches += 1;
|
|
140
|
+
last = message.model;
|
|
141
|
+
}
|
|
142
|
+
return models.size > 0 ? { models: models.size, switches } : null;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/** The stretch's whole derived record; `helpers` are the hook's own readers of a human turn, so both read it the same way. */
|
|
146
|
+
export function deriveStretch(messages, helpers) {
|
|
147
|
+
const context = contextOf(messages, helpers), routing = routingOf(messages);
|
|
148
|
+
return { ...verificationOf(messages, helpers), ...(context ? { context } : {}), ...(routing ? { routing } : {}) };
|
|
149
|
+
}
|
package/worktrust.mjs
CHANGED
|
@@ -71,14 +71,14 @@ const command = args.find((arg, at) => !arg.startsWith("--") && !(at > 0 && VALU
|
|
|
71
71
|
const flag = (name) => { const at = args.indexOf(`--${name}`); return at >= 0 ? args[at + 1] : undefined; };
|
|
72
72
|
const has = (name) => args.includes(`--${name}`);
|
|
73
73
|
/** This CLI's version, said to the door so the app can tell which computer runs an old one (check-cli-package holds it equal to package.json). */
|
|
74
|
-
const CLI_VERSION = "0.7.
|
|
74
|
+
const CLI_VERSION = "0.7.4";
|
|
75
75
|
const ORIGIN = (flag("origin") ?? process.env.WORKTRUST_ORIGIN ?? "https://app.worktrust.io").replace(/\/$/, "");
|
|
76
76
|
const MCP = flag("url") ?? process.env.WORKTRUST_MCP_URL ?? `${ORIGIN}/api/mcp`;
|
|
77
77
|
const HOME_DIR = join(homedir(), ".worktrust");
|
|
78
78
|
const HERE = dirname(fileURLToPath(import.meta.url));
|
|
79
|
-
const SCRIPTS = ["setup-mcp.mjs", "log-session.mjs", "count-behaviour.mjs", "transcript-readers.mjs", "session-databases.mjs", "config-edits.mjs"];
|
|
79
|
+
const SCRIPTS = ["setup-mcp.mjs", "log-session.mjs", "count-behaviour.mjs", "transcript-readers.mjs", "session-databases.mjs", "config-edits.mjs", "stretch-evidence.mjs"];
|
|
80
80
|
/** The modules the scripts import by name from beside them; a download keeps that name. */
|
|
81
|
-
const IMPORTED = new Set(["transcript-readers.mjs", "session-databases.mjs", "config-edits.mjs"]);
|
|
81
|
+
const IMPORTED = new Set(["transcript-readers.mjs", "session-databases.mjs", "config-edits.mjs", "stretch-evidence.mjs"]);
|
|
82
82
|
const downloaded = (name) => (IMPORTED.has(name) ? name : `${name}.download.mjs`);
|
|
83
83
|
const BUNDLED = SCRIPTS.every((name) => existsSync(join(HERE, name)));
|
|
84
84
|
const PLACEHOLDER = `wt_${"0".repeat(43)}`;
|