worktrust 0.8.0 → 0.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -0
- package/log-session.mjs +7 -3
- package/package.json +1 -1
- package/preserve-lines.mjs +8 -0
- package/stretch-evidence.mjs +41 -3
- package/worktrust.mjs +1 -1
package/README.md
CHANGED
|
@@ -167,6 +167,7 @@ nothing it does not:
|
|
|
167
167
|
| `context`, `routing` | how often the context was compacted and which context commands you used (compact, clear, resume, context, model, memory; never a command of your own), how many models answered and how often the model changed (0.7.4) |
|
|
168
168
|
| `steering` | the moments you stepped in while the agent worked (stopping it, or a message mid-task), how many changed what the agent did next, and how many were followed by a passing check (0.7.5) |
|
|
169
169
|
| `tools`, `complexity` | how many different tools the agent used, and the stretch's task complexity C1 to C5 by the rule complexity/1 (points for layers touched, tools, subagents, duration, a recovered failure, kinds of check and a delivery; the rule is in stretch-evidence.mjs), with its points and the rule's version (0.7.6) |
|
|
170
|
+
| `oversight`, `planning`, `changes` | how often you refused an action or a plan the agent proposed, and how often a guard (the client's safety layer) refused one; how often the agent wrote its to-do list, the most items, and of the last list how many were done (never what they say); how many files the stretch's commits changed and how many of them tests (never a path) (0.8.1) |
|
|
170
171
|
| `collector_version` | the CLI version that measured it |
|
|
171
172
|
| `seq`, `prev`, `hash` | the line's place, the hash of the line before it, and its own hash |
|
|
172
173
|
|
package/log-session.mjs
CHANGED
|
@@ -135,6 +135,8 @@ const DAY_SECONDS = 86400;
|
|
|
135
135
|
const TOOL_CAP = 1800;
|
|
136
136
|
/** Tools whose result waits for the person: the gap to it is the person's time. Names only; nothing of the call is read. */
|
|
137
137
|
const WAITS_FOR_PERSON = new Set(["AskUserQuestion", "ExitPlanMode"]);
|
|
138
|
+
/** A path that holds tests, by the conventions of the common test runners (read here; the path never leaves). */
|
|
139
|
+
const TEST_PATH = /(^|[\\/])(__tests__|tests?|spec|e2e)[\\/]|\.(test|spec)\.[cm]?[jt]sx?$|_test\.(go|py|rb)$|(^|[\\/])test_[^\\/]+\.py$|Tests?\.(java|kt|cs|swift)$/;
|
|
138
140
|
/** The derived record of a stretch, kept for the local archive only (see payloadFor). */
|
|
139
141
|
const DERIVED = new WeakMap();
|
|
140
142
|
const AGENT_RUNS_MAX = 10000, AGENT_SECONDS_MAX = 2592000;
|
|
@@ -211,7 +213,7 @@ const excluded = (path) => excludes.some((text) => text && path.includes(text));
|
|
|
211
213
|
const { CODEX_ROLLOUT = /(?!)/, codexIdOf = (file) => String(file).split(/[\\/]/).at(-1).replace(/\.jsonl$/, ""), antigravityIdOf = (file) => String(file).split(/[\\/]/).at(-4), codexLines, codexRolloutFiles = function* () {}, codexCwd = () => null, ANTIGRAVITY_TRANSCRIPT = /(?!)/, antigravityLines, antigravityRoots = () => [], antigravityTranscripts = function* () {}, antigravityContext = () => ({}), rememberAntigravity = () => null } = (await import("./transcript-readers.mjs").catch(() => null)) ?? {};
|
|
212
214
|
// THE STRETCH'S DERIVED RECORD (0.7.4: split from this file): tool calls named by kind, verification, recovery, delegation,
|
|
213
215
|
// context and routing, for the local archive only. A copy without the module still measures and sends exactly as before.
|
|
214
|
-
const { callOf = (block) => ({ id: block.id, kinds: [], family: String(block.name), digest: "" }), deriveStretch = () => ({}), complexityOf = null } = (await import("./stretch-evidence.mjs").catch(() => null)) ?? {};
|
|
216
|
+
const { callOf = (block) => ({ id: block.id, kinds: [], family: String(block.name), digest: "" }), deriveStretch = () => ({}), complexityOf = null, resultOf = (block) => ({ id: block.tool_use_id, failed: block.is_error === true, refusal: null }) } = (await import("./stretch-evidence.mjs").catch(() => null)) ?? {};
|
|
215
217
|
const { DATABASE_SESSION = /(?!)/, databaseLines = () => null, databaseSessions = function* () {} } = (await import("./session-databases.mjs").catch(() => null)) ?? {};
|
|
216
218
|
const DATABASE_CLIENTS = ["hermes", "goose", "opencode", "openclaw", "cursor", "copilot"]; // the registry's keys, the name each line carries
|
|
217
219
|
const LIVE_SOURCE = new RegExp(`^(codex|antigravity|${DATABASE_CLIENTS.join("|")}):`); // a sweep entry's id → the client it names (every database client, 0.6.16)
|
|
@@ -398,7 +400,7 @@ const messageOf = (line, at) => {
|
|
|
398
400
|
// 0.7.1: each tool call's kinds (never its command) and each tool result's outcome, joined on the call's id in `measure`.
|
|
399
401
|
// 0.7.2: and the call's FAMILY (the tool, and its check kinds) and a digest of its input, compared here and never kept.
|
|
400
402
|
calls: line.type === "assistant" ? blocks.filter((block) => block && block.type === "tool_use" && typeof block.id === "string").map(callOf) : [],
|
|
401
|
-
results: blocks.filter((block) => block && block.type === "tool_result" && typeof block.tool_use_id === "string").map(
|
|
403
|
+
results: blocks.filter((block) => block && block.type === "tool_result" && typeof block.tool_use_id === "string").map(resultOf),
|
|
402
404
|
};
|
|
403
405
|
};
|
|
404
406
|
|
|
@@ -624,7 +626,9 @@ function payloadFor(stretch) {
|
|
|
624
626
|
};
|
|
625
627
|
// WHAT STAYS ON THIS COMPUTER (0.7.1): the stretch's derived record rides beside the payload for the local archive, never in it.
|
|
626
628
|
// 0.7.6: and the stretch's complexity class, read from what the hook knows here (layers, subagents, duration) and the record.
|
|
627
|
-
|
|
629
|
+
// 0.8.1: what the stretch changed, as counts of files (never their paths): how many, and how many of them tests.
|
|
630
|
+
const changed = [...new Set(touched.paths)], tests = changed.filter((path) => TEST_PATH.test(path)).length;
|
|
631
|
+
if (stretch.derived) DERIVED.set(payload, { ...stretch.derived, ...(changed.length > 0 ? { changes: { files: changed.length, test_files: tests } } : {}), ...(complexityOf ? { complexity: complexityOf(stretch.derived, { layers: Math.max(layers.length, layer ? 1 : 0), agentRuns: stretch.agents?.runs ?? 0, agentPeak: stretch.agents?.peak ?? 0, seconds: stretch.seconds }) } : {}) });
|
|
628
632
|
return payload;
|
|
629
633
|
}
|
|
630
634
|
|
package/package.json
CHANGED
package/preserve-lines.mjs
CHANGED
|
@@ -83,6 +83,14 @@ export function archiveLine(entry, behaviour = null, { ids = {}, collector } = {
|
|
|
83
83
|
// TOOL BREADTH AND COMPLEXITY (0.7.6): distinct tools; a class C1–C5 with its points and the rule's version.
|
|
84
84
|
if (whole(entry.tools) && entry.tools > 0) line.tools = entry.tools;
|
|
85
85
|
if (entry.complexity && /^C[1-5]$/.test(entry.complexity.class ?? "") && whole(entry.complexity.points) && /^complexity\/\d+$/.test(entry.complexity.rule ?? "")) line.complexity = { class: entry.complexity.class, points: entry.complexity.points, rule: entry.complexity.rule };
|
|
86
|
+
// OVERSIGHT, PLANNING AND CHANGES (0.8.1): refusals by the person and by a guard; the agent's to-do lists; files changed and test files among them.
|
|
87
|
+
const pick = (record, keys) => (record && typeof record === "object" && keys.every((key) => whole(record[key])) ? Object.fromEntries(keys.map((key) => [key, record[key]])) : null);
|
|
88
|
+
const oversight = pick(entry.oversight, ["refused", "plans_rejected", "guard_denied"]);
|
|
89
|
+
if (oversight) line.oversight = oversight;
|
|
90
|
+
const planning = pick(entry.planning, ["lists", "items_max", "items_last", "done_last"]);
|
|
91
|
+
if (planning && planning.lists > 0 && planning.done_last <= planning.items_last && planning.items_last <= planning.items_max) line.planning = planning;
|
|
92
|
+
const changes = pick(entry.changes, ["files", "test_files"]);
|
|
93
|
+
if (changes && changes.test_files <= changes.files) line.changes = changes;
|
|
86
94
|
// The behaviour signals of this stretch (0.7.0): keys of the counter's rubric and whole counts, never a word of a turn.
|
|
87
95
|
if (behaviour && behaviour.signals && typeof behaviour.analyzer_version === "string" && /^counter@\d+\.\d+\.\d+$/.test(behaviour.analyzer_version)) {
|
|
88
96
|
const kept = Object.entries(behaviour.signals).filter(([key, count]) => SIGNAL.test(key) && Number.isInteger(count) && count > 0).sort();
|
package/stretch-evidence.mjs
CHANGED
|
@@ -32,8 +32,23 @@ export const callKindsOf = (block) => {
|
|
|
32
32
|
const command = block && SHELL_TOOLS.has(String(block.name)) ? block.input?.command ?? block.input?.CommandLine ?? block.input?.cmd : null;
|
|
33
33
|
return typeof command === "string" ? CALL_KINDS.filter(([, pattern]) => pattern.test(command)).map(([kind]) => kind) : [];
|
|
34
34
|
};
|
|
35
|
+
/**
|
|
36
|
+
* A TOOL RESULT AS THE STRETCH READS IT (0.8.1): its call's id, whether it failed, and, read here and never kept, WHO
|
|
37
|
+
* refused it when it was refused: the PERSON (Claude Code's fixed sentence when they decline a tool use or a plan) or a
|
|
38
|
+
* GUARD (the client's own safety layer or auto-mode classifier). Framework L8 and L5.
|
|
39
|
+
*/
|
|
40
|
+
const PERSON_REFUSED = /^\s*(The user doesn't want to (proceed|take this action)|User rejected|The user rejected|The user declined)/i;
|
|
41
|
+
const GUARD_REFUSED = /^\s*Permission for this (command|action) was denied by/i;
|
|
42
|
+
export const resultOf = (block) => {
|
|
43
|
+
const text = typeof block.content === "string" ? block.content : Array.isArray(block.content) ? block.content.map((part) => (part && typeof part.text === "string" ? part.text : "")).join("") : "";
|
|
44
|
+
const failed = block.is_error === true;
|
|
45
|
+
return { id: block.tool_use_id, failed, refusal: failed && PERSON_REFUSED.test(text) ? "person" : failed && GUARD_REFUSED.test(text) ? "guard" : null };
|
|
46
|
+
};
|
|
47
|
+
/** A to-do list the agent wrote (TodoWrite): how many items and how many done, never what they say. */
|
|
48
|
+
const planOf = (block) => (block.name === "TodoWrite" && Array.isArray(block.input?.todos) ? { items: block.input.todos.length, done: block.input.todos.filter((todo) => todo && todo.status === "completed").length } : null);
|
|
49
|
+
|
|
35
50
|
/** One tool call as the stretch reads it: its id, its kinds, its FAMILY (the tool and its check kinds) and a digest of its input (compared, never kept). */
|
|
36
|
-
export const callOf = (block) => { const kinds = callKindsOf(block); return { id: block.id, kinds, family: `${String(block.name)}|${kinds.join("+")}`, digest: createHash("sha256").update(JSON.stringify(block.input ?? null)).digest("base64url").slice(0, 16) }; };
|
|
51
|
+
export const callOf = (block) => { const kinds = callKindsOf(block), plan = planOf(block); return { id: block.id, kinds, family: `${String(block.name)}|${kinds.join("+")}`, digest: createHash("sha256").update(JSON.stringify(block.input ?? null)).digest("base64url").slice(0, 16), ...(plan ? { plan } : {}) }; };
|
|
37
52
|
|
|
38
53
|
/**
|
|
39
54
|
* VERIFICATION PER STRETCH (0.7.1; framework L7, L12): per kind of check, how many ran and how many failed; per step of
|
|
@@ -194,9 +209,32 @@ export function complexityOf(derived, { layers = 0, agentRuns = 0, agentPeak = 0
|
|
|
194
209
|
/** How many different tools the stretch's agent used (framework L11: tool breadth); null without a tool call. */
|
|
195
210
|
const toolsOf = (messages) => { const names = new Set(); for (const message of messages) for (const call of message.calls ?? []) names.add(call.family.split("|")[0]); return names.size > 0 ? names.size : null; };
|
|
196
211
|
|
|
212
|
+
/**
|
|
213
|
+
* OVERSIGHT AND PLANNING PER STRETCH (0.8.1; framework L8, L5, L2). Oversight: how often the PERSON refused an action
|
|
214
|
+
* or a plan the agent proposed, and how often a GUARD (the client's safety layer) refused one. Planning: how often the
|
|
215
|
+
* agent wrote its to-do list, the most items one list held, and of the last list how many were done. Null where none.
|
|
216
|
+
*/
|
|
217
|
+
function oversightOf(messages) {
|
|
218
|
+
let person = 0, plans = 0, guard = 0;
|
|
219
|
+
const tools = new Map();
|
|
220
|
+
for (const message of messages) for (const call of message.calls ?? []) tools.set(call.id, call.family.split("|")[0]);
|
|
221
|
+
for (const message of messages) for (const result of message.results ?? []) {
|
|
222
|
+
if (result.refusal === "person") { if (tools.get(result.id) === "ExitPlanMode") plans += 1; else person += 1; }
|
|
223
|
+
if (result.refusal === "guard") guard += 1;
|
|
224
|
+
}
|
|
225
|
+
return person + plans + guard > 0 ? { refused: person, plans_rejected: plans, guard_denied: guard } : null;
|
|
226
|
+
}
|
|
227
|
+
function planningOf(messages) {
|
|
228
|
+
const lists = [];
|
|
229
|
+
for (const message of messages) for (const call of message.calls ?? []) if (call.plan) lists.push(call.plan);
|
|
230
|
+
if (lists.length === 0) return null;
|
|
231
|
+
const last = lists.at(-1);
|
|
232
|
+
return { lists: lists.length, items_max: Math.max(...lists.map((list) => list.items)), items_last: last.items, done_last: last.done };
|
|
233
|
+
}
|
|
234
|
+
|
|
197
235
|
/** The stretch's whole derived record; `helpers` are the hook's own readers of a human turn, so both read it the same way. */
|
|
198
236
|
export function deriveStretch(messages, helpers) {
|
|
199
237
|
const context = contextOf(messages, helpers), routing = routingOf(messages), steering = steeringOf(messages, helpers);
|
|
200
|
-
const tools = toolsOf(messages);
|
|
201
|
-
return { ...verificationOf(messages, helpers), ...(context ? { context } : {}), ...(routing ? { routing } : {}), ...(steering ? { steering } : {}), ...(tools ? { tools } : {}) };
|
|
238
|
+
const tools = toolsOf(messages), oversight = oversightOf(messages), planning = planningOf(messages);
|
|
239
|
+
return { ...(oversight ? { oversight } : {}), ...(planning ? { planning } : {}), ...verificationOf(messages, helpers), ...(context ? { context } : {}), ...(routing ? { routing } : {}), ...(steering ? { steering } : {}), ...(tools ? { tools } : {}) };
|
|
202
240
|
}
|
package/worktrust.mjs
CHANGED
|
@@ -73,7 +73,7 @@ const command = args.find((arg, at) => !arg.startsWith("--") && !(at > 0 && VALU
|
|
|
73
73
|
const flag = (name) => { const at = args.indexOf(`--${name}`); return at >= 0 ? args[at + 1] : undefined; };
|
|
74
74
|
const has = (name) => args.includes(`--${name}`);
|
|
75
75
|
/** This CLI's version, said to the door so the app can tell which computer runs an old one (check-cli-package holds it equal to package.json). */
|
|
76
|
-
const CLI_VERSION = "0.8.
|
|
76
|
+
const CLI_VERSION = "0.8.1";
|
|
77
77
|
const ORIGIN = (flag("origin") ?? process.env.WORKTRUST_ORIGIN ?? "https://app.worktrust.io").replace(/\/$/, "");
|
|
78
78
|
const MCP = flag("url") ?? process.env.WORKTRUST_MCP_URL ?? `${ORIGIN}/api/mcp`;
|
|
79
79
|
const HOME_DIR = join(homedir(), ".worktrust");
|