@phnx-labs/agents-cli 1.22.52 → 1.22.53
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +146 -0
- package/README.md +1 -1
- package/dist/commands/accounts.js +1 -1
- package/dist/commands/exec.js +16 -9
- package/dist/commands/fleet-capture.js +7 -0
- package/dist/commands/focus.js +2 -0
- package/dist/commands/go.js +2 -1
- package/dist/commands/sessions-inject.js +8 -3
- package/dist/commands/sessions-picker.js +2 -1
- package/dist/commands/sessions.js +30 -20
- package/dist/commands/ssh.js +35 -12
- package/dist/commands/sync.js +44 -0
- package/dist/lib/account-registry.d.ts +15 -5
- package/dist/lib/account-registry.js +150 -50
- package/dist/lib/answer-router.js +2 -1
- package/dist/lib/browser/profiles.d.ts +18 -0
- package/dist/lib/browser/profiles.js +26 -1
- package/dist/lib/browser/registry.d.ts +44 -14
- package/dist/lib/browser/registry.js +141 -45
- package/dist/lib/daemon/runner.js +10 -2
- package/dist/lib/device-config.js +3 -2
- package/dist/lib/devices/config-migration.js +147 -1
- package/dist/lib/devices/device-docs.d.ts +35 -0
- package/dist/lib/devices/device-docs.js +163 -0
- package/dist/lib/devices/discovery-policy.d.ts +14 -2
- package/dist/lib/devices/discovery-policy.js +31 -21
- package/dist/lib/devices/registry.d.ts +11 -5
- package/dist/lib/devices/registry.js +46 -18
- package/dist/lib/exec.d.ts +60 -30
- package/dist/lib/exec.js +65 -27
- package/dist/lib/feed/feed.d.ts +10 -2
- package/dist/lib/feed/feed.js +12 -1
- package/dist/lib/hosts/dispatch.d.ts +4 -3
- package/dist/lib/hosts/dispatch.js +12 -8
- package/dist/lib/hosts/providers/local.d.ts +9 -3
- package/dist/lib/hosts/providers/local.js +23 -12
- package/dist/lib/hosts/reconnect.d.ts +7 -4
- package/dist/lib/hosts/reconnect.js +29 -25
- package/dist/lib/hosts/registry.js +4 -1
- package/dist/lib/hosts/remote-os.js +3 -1
- package/dist/lib/session/active.d.ts +10 -1
- package/dist/lib/session/active.js +7 -1
- package/dist/lib/session/actor-sidecar.d.ts +7 -0
- package/dist/lib/session/actor-sidecar.js +2 -0
- package/dist/lib/session/db.d.ts +1 -1
- package/dist/lib/session/db.js +39 -3
- package/dist/lib/session/discover.js +7 -12
- package/dist/lib/session/live-metadata.js +1 -0
- package/dist/lib/session/pid-registry.d.ts +7 -0
- package/dist/lib/session/prompt.d.ts +15 -0
- package/dist/lib/session/prompt.js +21 -0
- package/dist/lib/session/types.d.ts +17 -0
- package/dist/lib/session/types.js +10 -0
- package/dist/lib/share/worker-template.js +12 -7
- package/dist/lib/state.d.ts +8 -0
- package/dist/lib/state.js +143 -11
- package/dist/lib/terminal/resolve.d.ts +7 -0
- package/dist/lib/terminal/resolve.js +41 -2
- package/dist/lib/traces/insights.d.ts +67 -0
- package/dist/lib/traces/insights.js +178 -0
- package/dist/lib/traces/phenotype.d.ts +67 -0
- package/dist/lib/traces/phenotype.js +437 -0
- package/dist/lib/traces/segments.d.ts +133 -0
- package/dist/lib/traces/segments.js +301 -0
- package/dist/lib/traces/sync.d.ts +33 -0
- package/dist/lib/traces/sync.js +11 -2
- package/dist/lib/types.d.ts +47 -1
- package/dist/lib/watchdog/runner.js +18 -4
- package/package.json +1 -1
|
@@ -0,0 +1,437 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Failure phenotype classifier + outcome taxonomy for the traces insight engine.
|
|
3
|
+
*
|
|
4
|
+
* Both functions are pure: they take a redacted {@link SessionDetail} (the same
|
|
5
|
+
* shape the traces sync writes to `sessions/<id>.json`) and return a decision
|
|
6
|
+
* derived only from the already-derived step/gap/meta signal. They never read
|
|
7
|
+
* raw transcript text and never fabricate a signal that is not in the data.
|
|
8
|
+
*
|
|
9
|
+
* The rubrics are expressed as data-driven tables of conditions, not as
|
|
10
|
+
* if/else-by-name chains. Each table row is a named phenotype/outcome with a
|
|
11
|
+
* declarative predicate; the classifier walks the table in priority order and
|
|
12
|
+
* returns the first match, or the honest lower-confidence default when no
|
|
13
|
+
* high-confidence signal is present.
|
|
14
|
+
*/
|
|
15
|
+
// ---------------------------------------------------------------------------
|
|
16
|
+
// Tool / program category tables
|
|
17
|
+
// ---------------------------------------------------------------------------
|
|
18
|
+
const READ_PLAN_TOOLS = new Set([
|
|
19
|
+
'Read',
|
|
20
|
+
'read_file',
|
|
21
|
+
'grep',
|
|
22
|
+
'list_dir',
|
|
23
|
+
'search',
|
|
24
|
+
'codebase_search',
|
|
25
|
+
'ToolSearch',
|
|
26
|
+
'web_search',
|
|
27
|
+
'web_fetch',
|
|
28
|
+
'WebSearch',
|
|
29
|
+
'FetchURL',
|
|
30
|
+
'TaskCreate',
|
|
31
|
+
'todo_write',
|
|
32
|
+
]);
|
|
33
|
+
const WRITE_EDIT_TOOLS = new Set([
|
|
34
|
+
'Edit',
|
|
35
|
+
'Write',
|
|
36
|
+
'search_replace',
|
|
37
|
+
'write',
|
|
38
|
+
'notebookedit',
|
|
39
|
+
'multiedit',
|
|
40
|
+
]);
|
|
41
|
+
const HUMAN_FACING_TOOLS = new Set(['AskUserQuestion', 'SendMessage', 'wait']);
|
|
42
|
+
const SHELL_TOOLS = new Set([
|
|
43
|
+
'Bash',
|
|
44
|
+
'run_terminal_command',
|
|
45
|
+
'exec_command',
|
|
46
|
+
'exec',
|
|
47
|
+
'shell',
|
|
48
|
+
'Execute',
|
|
49
|
+
]);
|
|
50
|
+
const OUTCOME_SIGNALS = [
|
|
51
|
+
{ type: 'merge', pattern: /\bgh pr merge\b|\bmerged?\b.*\b(PR|pull request|branch)\b|\brebase-?merge\b/i },
|
|
52
|
+
{ type: 'merge', pattern: /\bmerge\b.*\bsucceeded\b|\bsuccessfully\s+merged\b/i },
|
|
53
|
+
{ type: 'test', pattern: /\b(bun test|npm test|yarn test|pnpm test|pytest|jest|vitest|cargo test|go test)\b/i },
|
|
54
|
+
{ type: 'test', pattern: /\btsc\s+--noEmit\b|\blint\b|\btest\.sh\b/i },
|
|
55
|
+
{ type: 'env', pattern: /\bbun install\b|\bnpm install\b|\byarn install\b|\bpnpm install\b|\bpip install\b/i },
|
|
56
|
+
{ type: 'env', pattern: /\bnode_modules\b|\bmissing\b|\bnot found\b|\bpermission denied\b|\bcommand not found\b/i },
|
|
57
|
+
{ type: 'env', pattern: /\bssh.*key\b|\bclone\b.*\bfailed\b/i },
|
|
58
|
+
{ type: 'revert', pattern: /\brevert\b|\bgit checkout\b|\breset\s+--hard\b/i },
|
|
59
|
+
];
|
|
60
|
+
// ---------------------------------------------------------------------------
|
|
61
|
+
// Step helpers
|
|
62
|
+
// ---------------------------------------------------------------------------
|
|
63
|
+
function isToolStep(step) {
|
|
64
|
+
return step.kind === 'tool';
|
|
65
|
+
}
|
|
66
|
+
function substantiveSteps(session) {
|
|
67
|
+
return session.steps.filter((s) => isToolStep(s) && !HUMAN_FACING_TOOLS.has(s.tool ?? s.lane));
|
|
68
|
+
}
|
|
69
|
+
function firstStepOrdinalOf(session, predicate) {
|
|
70
|
+
for (const step of session.steps) {
|
|
71
|
+
if (predicate(step))
|
|
72
|
+
return step.ordinal;
|
|
73
|
+
}
|
|
74
|
+
return undefined;
|
|
75
|
+
}
|
|
76
|
+
function lastStepOrdinalOf(session, predicate) {
|
|
77
|
+
let last;
|
|
78
|
+
for (const step of session.steps) {
|
|
79
|
+
if (predicate(step))
|
|
80
|
+
last = step.ordinal;
|
|
81
|
+
}
|
|
82
|
+
return last;
|
|
83
|
+
}
|
|
84
|
+
function stepText(step) {
|
|
85
|
+
return `${step.label ?? ''} ${step.detail ?? ''}`.trim().toLowerCase();
|
|
86
|
+
}
|
|
87
|
+
function isShellStep(step) {
|
|
88
|
+
return SHELL_TOOLS.has(step.tool ?? step.lane);
|
|
89
|
+
}
|
|
90
|
+
function hasSignal(session, type, toolFilter) {
|
|
91
|
+
for (const step of session.steps) {
|
|
92
|
+
if (toolFilter && !toolFilter.has(step.tool ?? step.lane))
|
|
93
|
+
continue;
|
|
94
|
+
const text = stepText(step);
|
|
95
|
+
for (const signal of OUTCOME_SIGNALS) {
|
|
96
|
+
if (signal.type === type && signal.pattern.test(text))
|
|
97
|
+
return true;
|
|
98
|
+
}
|
|
99
|
+
}
|
|
100
|
+
return false;
|
|
101
|
+
}
|
|
102
|
+
function signalCounts(session) {
|
|
103
|
+
const counts = { merge: 0, test: 0, env: 0, revert: 0 };
|
|
104
|
+
for (const step of session.steps) {
|
|
105
|
+
const text = stepText(step);
|
|
106
|
+
for (const signal of OUTCOME_SIGNALS) {
|
|
107
|
+
// Merge/test signals must come from an actual shell/exec step, otherwise
|
|
108
|
+
// a grep pattern or read_file detail false-positives as a landing signal.
|
|
109
|
+
if ((signal.type === 'merge' || signal.type === 'test') && !isShellStep(step))
|
|
110
|
+
continue;
|
|
111
|
+
if (signal.pattern.test(text))
|
|
112
|
+
counts[signal.type]++;
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
return counts;
|
|
116
|
+
}
|
|
117
|
+
// ---------------------------------------------------------------------------
|
|
118
|
+
// Phenotype predicates
|
|
119
|
+
// ---------------------------------------------------------------------------
|
|
120
|
+
/**
|
|
121
|
+
* Failure-to-act: the agent stalled or produced no meaningful tool use.
|
|
122
|
+
*
|
|
123
|
+
* Rubric:
|
|
124
|
+
* - zero tool calls and ≤2 turns, or
|
|
125
|
+
* - all substantive steps are `thinking`/human-facing, or
|
|
126
|
+
* - the session ended on a human-facing ask with no follow-up tool work.
|
|
127
|
+
*/
|
|
128
|
+
function isFailureToAct(session) {
|
|
129
|
+
if (session.meta.tools === 0 && session.meta.turns <= 2)
|
|
130
|
+
return true;
|
|
131
|
+
const substantive = substantiveSteps(session);
|
|
132
|
+
return substantive.length === 0;
|
|
133
|
+
}
|
|
134
|
+
function reasonFailureToAct(session) {
|
|
135
|
+
if (session.meta.tools === 0 && session.meta.turns <= 2) {
|
|
136
|
+
return `no tool use (${session.meta.tools} tools, ${session.meta.turns} turns)`;
|
|
137
|
+
}
|
|
138
|
+
return 'no substantive tool steps';
|
|
139
|
+
}
|
|
140
|
+
/**
|
|
141
|
+
* Out-of-order: a write/edit happened before any read/plan of the target.
|
|
142
|
+
*
|
|
143
|
+
* Rubric: the first substantive step is a write/edit and no earlier read/plan
|
|
144
|
+
* or search step exists.
|
|
145
|
+
*/
|
|
146
|
+
function isOutOfOrder(session) {
|
|
147
|
+
const firstWrite = firstStepOrdinalOf(session, (s) => WRITE_EDIT_TOOLS.has(s.tool ?? s.lane));
|
|
148
|
+
if (firstWrite === undefined)
|
|
149
|
+
return false;
|
|
150
|
+
const firstReadPlan = firstStepOrdinalOf(session, (s) => READ_PLAN_TOOLS.has(s.tool ?? s.lane));
|
|
151
|
+
if (firstReadPlan === undefined)
|
|
152
|
+
return true;
|
|
153
|
+
return firstWrite < firstReadPlan;
|
|
154
|
+
}
|
|
155
|
+
function reasonOutOfOrder(session) {
|
|
156
|
+
const tool = session.steps.find((s) => WRITE_EDIT_TOOLS.has(s.tool ?? s.lane));
|
|
157
|
+
return `write/edit step ${tool?.tool ?? tool?.lane ?? ''} preceded any read/plan`;
|
|
158
|
+
}
|
|
159
|
+
/**
|
|
160
|
+
* False-termination: the session stopped with an unresolved error.
|
|
161
|
+
*
|
|
162
|
+
* Rubric:
|
|
163
|
+
* - meta outcome is `errored`, and
|
|
164
|
+
* - at least one step outcome is `error`, and
|
|
165
|
+
* - the last non-thinking step is an error or is followed by no successful recovery.
|
|
166
|
+
*/
|
|
167
|
+
function isFalseTermination(session) {
|
|
168
|
+
if (session.meta.outcome !== 'errored')
|
|
169
|
+
return false;
|
|
170
|
+
if (session.meta.errorCount === 0)
|
|
171
|
+
return false;
|
|
172
|
+
const substantive = substantiveSteps(session);
|
|
173
|
+
if (substantive.length === 0)
|
|
174
|
+
return true;
|
|
175
|
+
const last = substantive[substantive.length - 1];
|
|
176
|
+
if (last.outcome === 'error')
|
|
177
|
+
return true;
|
|
178
|
+
const lastErrorOrdinal = lastStepOrdinalOf(session, (s) => s.outcome === 'error');
|
|
179
|
+
if (lastErrorOrdinal === undefined)
|
|
180
|
+
return false;
|
|
181
|
+
const recoveryAfter = substantive.some((s) => s.ordinal > lastErrorOrdinal && s.outcome === 'ok' && !HUMAN_FACING_TOOLS.has(s.tool ?? s.lane));
|
|
182
|
+
return !recoveryAfter;
|
|
183
|
+
}
|
|
184
|
+
function reasonFalseTermination(session) {
|
|
185
|
+
const substantive = substantiveSteps(session);
|
|
186
|
+
if (substantive.length === 0)
|
|
187
|
+
return 'errored with no substantive steps';
|
|
188
|
+
const last = substantive[substantive.length - 1];
|
|
189
|
+
if (last.outcome === 'error') {
|
|
190
|
+
return `last substantive step ${last.tool ?? last.lane} ended in error`;
|
|
191
|
+
}
|
|
192
|
+
return 'errored with no successful recovery after the last error';
|
|
193
|
+
}
|
|
194
|
+
/**
|
|
195
|
+
* Premature-completion: declared done while tests were failing or no verification
|
|
196
|
+
* step ran for an engineering task.
|
|
197
|
+
*
|
|
198
|
+
* Rubric:
|
|
199
|
+
* - meta outcome is `completed`, and
|
|
200
|
+
* - the session performed write/edit work, and
|
|
201
|
+
* - either errors were present or no test/build/lint verification step ran.
|
|
202
|
+
*/
|
|
203
|
+
function isPrematureCompletion(session) {
|
|
204
|
+
if (session.meta.outcome !== 'completed')
|
|
205
|
+
return false;
|
|
206
|
+
const didWriteEdit = session.steps.some((s) => WRITE_EDIT_TOOLS.has(s.tool ?? s.lane));
|
|
207
|
+
if (!didWriteEdit)
|
|
208
|
+
return false;
|
|
209
|
+
if (session.meta.errorCount > 0)
|
|
210
|
+
return true;
|
|
211
|
+
const verified = session.steps.some((s) => {
|
|
212
|
+
if (!SHELL_TOOLS.has(s.tool ?? s.lane))
|
|
213
|
+
return false;
|
|
214
|
+
const text = stepText(s);
|
|
215
|
+
return /\b(bun test|npm test|yarn test|pnpm test|pytest|jest|vitest|cargo test|go test|tsc\s+--noEmit|lint|build|verify)\b/.test(text);
|
|
216
|
+
});
|
|
217
|
+
return !verified;
|
|
218
|
+
}
|
|
219
|
+
function reasonPrematureCompletion(session) {
|
|
220
|
+
if (session.meta.errorCount > 0) {
|
|
221
|
+
return `declared completed with ${session.meta.errorCount} unresolved error(s)`;
|
|
222
|
+
}
|
|
223
|
+
return 'engineering work completed without a test/build/lint verification step';
|
|
224
|
+
}
|
|
225
|
+
/** Ordered rubric: the first matching phenotype wins. */
|
|
226
|
+
const PHENOTYPE_RULES = [
|
|
227
|
+
{ key: 'failure-to-act', score: isFailureToAct, reason: reasonFailureToAct },
|
|
228
|
+
{ key: 'out-of-order', score: isOutOfOrder, reason: reasonOutOfOrder },
|
|
229
|
+
{ key: 'false-termination', score: isFalseTermination, reason: reasonFalseTermination },
|
|
230
|
+
{ key: 'premature-completion', score: isPrematureCompletion, reason: reasonPrematureCompletion },
|
|
231
|
+
];
|
|
232
|
+
// ---------------------------------------------------------------------------
|
|
233
|
+
// Outcome predicates
|
|
234
|
+
// ---------------------------------------------------------------------------
|
|
235
|
+
/** High-confidence merge when explicit merge command/signal exists and session completed. */
|
|
236
|
+
function isMerged(session) {
|
|
237
|
+
// Only a shell/exec step can produce a real merge signal; grep patterns and
|
|
238
|
+
// read_file details must not count.
|
|
239
|
+
return session.meta.outcome === 'completed' && hasSignal(session, 'merge', SHELL_TOOLS);
|
|
240
|
+
}
|
|
241
|
+
function mergedReason(session) {
|
|
242
|
+
return 'explicit merge signal present and session completed';
|
|
243
|
+
}
|
|
244
|
+
/**
|
|
245
|
+
* Tests-green when a test/build/lint step returned ok, or when the session
|
|
246
|
+
* completed cleanly with test steps whose individual outcomes are unknown
|
|
247
|
+
* (e.g. Grok-derived sessions that record spanMs=0 and never pair results).
|
|
248
|
+
*/
|
|
249
|
+
function isTestsGreen(session) {
|
|
250
|
+
if (session.meta.outcome !== 'completed')
|
|
251
|
+
return false;
|
|
252
|
+
const testSteps = session.steps.filter((s) => {
|
|
253
|
+
if (!isShellStep(s))
|
|
254
|
+
return false;
|
|
255
|
+
const text = stepText(s);
|
|
256
|
+
return OUTCOME_SIGNALS.some((sig) => sig.type === 'test' && sig.pattern.test(text));
|
|
257
|
+
});
|
|
258
|
+
if (testSteps.length === 0)
|
|
259
|
+
return false;
|
|
260
|
+
const lastTest = testSteps[testSteps.length - 1];
|
|
261
|
+
if (lastTest.outcome === 'ok')
|
|
262
|
+
return true;
|
|
263
|
+
// Infer from a clean completion when step-level outcomes are not available.
|
|
264
|
+
return session.meta.errorCount === 0;
|
|
265
|
+
}
|
|
266
|
+
function testsGreenConfidence(session) {
|
|
267
|
+
const testSteps = session.steps.filter((s) => {
|
|
268
|
+
if (!isShellStep(s))
|
|
269
|
+
return false;
|
|
270
|
+
const text = stepText(s);
|
|
271
|
+
return OUTCOME_SIGNALS.some((sig) => sig.type === 'test' && sig.pattern.test(text));
|
|
272
|
+
});
|
|
273
|
+
const lastTest = testSteps[testSteps.length - 1];
|
|
274
|
+
return lastTest?.outcome === 'ok' ? 'high' : 'medium';
|
|
275
|
+
}
|
|
276
|
+
function testsGreenReason(session) {
|
|
277
|
+
const confidence = testsGreenConfidence(session);
|
|
278
|
+
return confidence === 'high'
|
|
279
|
+
? 'last test/build/lint step returned ok'
|
|
280
|
+
: 'completed cleanly with test steps (step outcomes unavailable)';
|
|
281
|
+
}
|
|
282
|
+
/** Human-takeover when the final tool action asks the human. */
|
|
283
|
+
function isHumanTakeover(session) {
|
|
284
|
+
const tools = session.steps.filter((s) => isToolStep(s));
|
|
285
|
+
if (tools.length === 0)
|
|
286
|
+
return false;
|
|
287
|
+
const last = tools[tools.length - 1];
|
|
288
|
+
return HUMAN_FACING_TOOLS.has(last.tool ?? last.lane);
|
|
289
|
+
}
|
|
290
|
+
function humanTakeoverReason(session) {
|
|
291
|
+
const tools = session.steps.filter((s) => isToolStep(s));
|
|
292
|
+
const last = tools[tools.length - 1];
|
|
293
|
+
return `final tool step is ${last.tool ?? last.lane}`;
|
|
294
|
+
}
|
|
295
|
+
/**
|
|
296
|
+
* Invalid-env when setup/environment failures dominate the session.
|
|
297
|
+
*
|
|
298
|
+
* Rubric:
|
|
299
|
+
* - ≥2 env errors and they make up at least half of all errors, or
|
|
300
|
+
* - ≥3 env steps with ≥50% error rate in an errored session, or
|
|
301
|
+
* - the first substantive error is an env/setup error and no recovery follows it.
|
|
302
|
+
*/
|
|
303
|
+
function isInvalidEnv(session) {
|
|
304
|
+
const envSteps = session.steps.filter((s) => {
|
|
305
|
+
const text = stepText(s);
|
|
306
|
+
return OUTCOME_SIGNALS.some((sig) => sig.type === 'env' && sig.pattern.test(text));
|
|
307
|
+
});
|
|
308
|
+
const envErrors = envSteps.filter((s) => s.outcome === 'error').length;
|
|
309
|
+
const totalErrors = session.meta.errorCount;
|
|
310
|
+
if (totalErrors > 0 && envErrors >= 2 && envErrors / totalErrors >= 0.5)
|
|
311
|
+
return true;
|
|
312
|
+
if (session.meta.outcome === 'errored' && envSteps.length >= 3 && envErrors / envSteps.length >= 0.5)
|
|
313
|
+
return true;
|
|
314
|
+
const substantive = substantiveSteps(session);
|
|
315
|
+
const firstError = substantive.find((s) => s.outcome === 'error');
|
|
316
|
+
if (session.meta.outcome === 'errored' &&
|
|
317
|
+
firstError &&
|
|
318
|
+
OUTCOME_SIGNALS.some((sig) => sig.type === 'env' && sig.pattern.test(stepText(firstError)))) {
|
|
319
|
+
const firstErrorIndex = substantive.indexOf(firstError);
|
|
320
|
+
const after = substantive.slice(firstErrorIndex + 1);
|
|
321
|
+
const recovered = after.some((s) => s.outcome === 'ok' && !HUMAN_FACING_TOOLS.has(s.tool ?? s.lane));
|
|
322
|
+
if (!recovered)
|
|
323
|
+
return true;
|
|
324
|
+
}
|
|
325
|
+
return false;
|
|
326
|
+
}
|
|
327
|
+
function invalidEnvReason(session) {
|
|
328
|
+
const envSteps = session.steps.filter((s) => {
|
|
329
|
+
const text = stepText(s);
|
|
330
|
+
return OUTCOME_SIGNALS.some((sig) => sig.type === 'env' && sig.pattern.test(text));
|
|
331
|
+
});
|
|
332
|
+
const envErrors = envSteps.filter((s) => s.outcome === 'error').length;
|
|
333
|
+
const totalErrors = session.meta.errorCount;
|
|
334
|
+
if (totalErrors > 0 && envErrors >= 2 && envErrors / totalErrors >= 0.5) {
|
|
335
|
+
return `${envErrors}/${totalErrors} errors are environment/setup related`;
|
|
336
|
+
}
|
|
337
|
+
return 'environment/setup failures dominated the early session and blocked recovery';
|
|
338
|
+
}
|
|
339
|
+
/** Abandoned when the session errored and stalled without recovery. */
|
|
340
|
+
function isAbandoned(session) {
|
|
341
|
+
if (session.meta.outcome !== 'errored')
|
|
342
|
+
return false;
|
|
343
|
+
const hasStall = session.gaps.some((g) => g.durationMs >= 120_000);
|
|
344
|
+
const substantive = substantiveSteps(session);
|
|
345
|
+
if (substantive.length === 0)
|
|
346
|
+
return hasStall || session.meta.errorCount > 0;
|
|
347
|
+
const lastErrorOrdinal = lastStepOrdinalOf(session, (s) => s.outcome === 'error');
|
|
348
|
+
if (lastErrorOrdinal === undefined)
|
|
349
|
+
return false;
|
|
350
|
+
const recoveryAfter = substantive.some((s) => s.ordinal > lastErrorOrdinal && s.outcome === 'ok' && !HUMAN_FACING_TOOLS.has(s.tool ?? s.lane));
|
|
351
|
+
return !recoveryAfter || hasStall;
|
|
352
|
+
}
|
|
353
|
+
function abandonedReason(session) {
|
|
354
|
+
const hasStall = session.gaps.some((g) => g.durationMs >= 120_000);
|
|
355
|
+
return hasStall ? 'errored with a long stall and no recovery' : 'errored with no successful recovery';
|
|
356
|
+
}
|
|
357
|
+
/** Partial: progress was made but no landing or test signal is present. */
|
|
358
|
+
function isPartial(_session) {
|
|
359
|
+
return true;
|
|
360
|
+
}
|
|
361
|
+
function partialReason(session) {
|
|
362
|
+
const counts = signalCounts(session);
|
|
363
|
+
if (session.meta.outcome === 'completed') {
|
|
364
|
+
if (counts.test > 0)
|
|
365
|
+
return 'tests ran but final test step did not return ok';
|
|
366
|
+
if (counts.merge > 0)
|
|
367
|
+
return 'merge-related activity but no confirmed merge';
|
|
368
|
+
return 'completed without a landing or test signal';
|
|
369
|
+
}
|
|
370
|
+
return 'incomplete work with no higher-confidence outcome signal';
|
|
371
|
+
}
|
|
372
|
+
function ruleConfidence(rule, session) {
|
|
373
|
+
return typeof rule.confidence === 'function' ? rule.confidence(session) : rule.confidence;
|
|
374
|
+
}
|
|
375
|
+
/**
|
|
376
|
+
* Ordered rubric: the first matching outcome wins. `partial` is the
|
|
377
|
+
* lower-confidence default when no high/medium-confidence signal is present.
|
|
378
|
+
*/
|
|
379
|
+
const OUTCOME_RULES = [
|
|
380
|
+
{ key: 'merged', confidence: 'high', score: isMerged, reason: mergedReason },
|
|
381
|
+
{ key: 'tests-green', confidence: testsGreenConfidence, score: isTestsGreen, reason: testsGreenReason },
|
|
382
|
+
{ key: 'human-takeover', confidence: 'medium', score: isHumanTakeover, reason: humanTakeoverReason },
|
|
383
|
+
{ key: 'invalid-env', confidence: 'medium', score: isInvalidEnv, reason: invalidEnvReason },
|
|
384
|
+
{ key: 'abandoned', confidence: 'medium', score: isAbandoned, reason: abandonedReason },
|
|
385
|
+
{ key: 'partial', confidence: 'low', score: isPartial, reason: partialReason },
|
|
386
|
+
];
|
|
387
|
+
// ---------------------------------------------------------------------------
|
|
388
|
+
// Public API
|
|
389
|
+
// ---------------------------------------------------------------------------
|
|
390
|
+
/**
|
|
391
|
+
* Classify the failure phenotype of a session from its derived trajectory.
|
|
392
|
+
*
|
|
393
|
+
* Definitions (from agent-failure research):
|
|
394
|
+
* - `false-termination` — stopped with an unresolved error.
|
|
395
|
+
* - `premature-completion` — declared done while tests were failing or no
|
|
396
|
+
* verification step ran for the engineering work.
|
|
397
|
+
* - `out-of-order` — a write/edit step occurred before any read/plan of the
|
|
398
|
+
* target.
|
|
399
|
+
* - `failure-to-act` — stalled or produced no meaningful tool use.
|
|
400
|
+
*
|
|
401
|
+
* Returns `null` when none of the failure phenotypes apply.
|
|
402
|
+
*/
|
|
403
|
+
export function classifyPhenotype(session) {
|
|
404
|
+
return classifyPhenotypeDetailed(session).phenotype;
|
|
405
|
+
}
|
|
406
|
+
/** Detailed phenotype result with a human-readable reason. */
|
|
407
|
+
export function classifyPhenotypeDetailed(session) {
|
|
408
|
+
for (const rule of PHENOTYPE_RULES) {
|
|
409
|
+
if (rule.score(session)) {
|
|
410
|
+
return { phenotype: rule.key, reason: rule.reason(session) };
|
|
411
|
+
}
|
|
412
|
+
}
|
|
413
|
+
return { phenotype: null, reason: 'no failure phenotype matched' };
|
|
414
|
+
}
|
|
415
|
+
/**
|
|
416
|
+
* Derive the coarse outcome of a session from its end state + tool signals.
|
|
417
|
+
*
|
|
418
|
+
* `merged` and `tests-green` are high-confidence only when an explicit signal
|
|
419
|
+
* is present in the derived steps. When that signal is genuinely not in the
|
|
420
|
+
* data, the function returns the honest lower-confidence value (`partial` for
|
|
421
|
+
* completed work without a landing signal, `abandoned` for errored/unresolved
|
|
422
|
+
* work, `invalid-env` for setup-dominant failures, `human-takeover` when the
|
|
423
|
+
* session ends on a human-facing ask).
|
|
424
|
+
*/
|
|
425
|
+
export function deriveOutcome(session) {
|
|
426
|
+
return deriveOutcomeDetailed(session).outcome;
|
|
427
|
+
}
|
|
428
|
+
/** Detailed outcome result with confidence and a human-readable reason. */
|
|
429
|
+
export function deriveOutcomeDetailed(session) {
|
|
430
|
+
for (const rule of OUTCOME_RULES) {
|
|
431
|
+
if (rule.score(session)) {
|
|
432
|
+
return { outcome: rule.key, confidence: ruleConfidence(rule, session), reason: rule.reason(session) };
|
|
433
|
+
}
|
|
434
|
+
}
|
|
435
|
+
// `partial` is the exhaustive default; this line is unreachable but keeps TS happy.
|
|
436
|
+
return { outcome: 'partial', confidence: 'low', reason: 'no outcome signal matched' };
|
|
437
|
+
}
|
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Group-by dimensions + time-to-first-tool latency for the traces insight
|
|
3
|
+
* engine (the console's Issue bar).
|
|
4
|
+
*
|
|
5
|
+
* Pure functions, no I/O. Classifiers are DATA-DRIVEN TABLES — a new task type
|
|
6
|
+
* or timing bucket is a row, not a new if/else-by-name arm. The integrator
|
|
7
|
+
* (`insights.ts` / `sync.ts`) tags each session; this file does not write the
|
|
8
|
+
* shard.
|
|
9
|
+
*
|
|
10
|
+
* Input is the redacted `SessionDetail` from `agents traces sync` plus the
|
|
11
|
+
* SyncRow fields the detail strips (prompt, gitBranch, files). Structural —
|
|
12
|
+
* `SessionDetail` is a valid `SegmentSession`.
|
|
13
|
+
*/
|
|
14
|
+
export declare const TASK_TYPES: readonly ["bugfix", "feature", "refactor", "test", "chore", "other"];
|
|
15
|
+
export type TaskType = (typeof TASK_TYPES)[number];
|
|
16
|
+
export declare const FAILURE_TIMINGS: readonly ["early", "mid", "late"];
|
|
17
|
+
export type FailureTiming = (typeof FAILURE_TIMINGS)[number];
|
|
18
|
+
/** The #1 group-by axis: compare the pair, never the model alone. */
|
|
19
|
+
export interface SegmentAgent {
|
|
20
|
+
model: string;
|
|
21
|
+
harness: string;
|
|
22
|
+
}
|
|
23
|
+
export interface SegmentFile {
|
|
24
|
+
path: string;
|
|
25
|
+
/** Write/create vs Edit/patch. */
|
|
26
|
+
action: 'new' | 'edit';
|
|
27
|
+
}
|
|
28
|
+
export interface SegmentStep {
|
|
29
|
+
startMs: number;
|
|
30
|
+
durationMs?: number;
|
|
31
|
+
outcome?: 'ok' | 'error' | 'unknown' | string;
|
|
32
|
+
tool?: string;
|
|
33
|
+
kind?: 'tool' | 'thinking' | string;
|
|
34
|
+
label?: string;
|
|
35
|
+
}
|
|
36
|
+
/**
|
|
37
|
+
* Session shape the classifiers read. `SessionDetail` (sessions/<id>.json)
|
|
38
|
+
* satisfies this; callers may also pass SyncRow fields (`prompt`, `gitBranch`,
|
|
39
|
+
* `agent`, `model`, `files`) alongside steps.
|
|
40
|
+
*/
|
|
41
|
+
export interface SegmentSession {
|
|
42
|
+
id?: string;
|
|
43
|
+
prompt?: string | null;
|
|
44
|
+
gitBranch?: string | null;
|
|
45
|
+
topic?: string | null;
|
|
46
|
+
label?: string | null;
|
|
47
|
+
mode?: string | null;
|
|
48
|
+
agent?: string | null;
|
|
49
|
+
harness?: string | null;
|
|
50
|
+
model?: string | null;
|
|
51
|
+
spanMs?: number | null;
|
|
52
|
+
files?: readonly SegmentFile[] | null;
|
|
53
|
+
meta?: {
|
|
54
|
+
agent?: string | null;
|
|
55
|
+
model?: string | null;
|
|
56
|
+
spanMs?: number | null;
|
|
57
|
+
repo?: string | null;
|
|
58
|
+
tools?: number | null;
|
|
59
|
+
} | null;
|
|
60
|
+
steps?: readonly SegmentStep[] | null;
|
|
61
|
+
whereItWentWrong?: string | null;
|
|
62
|
+
}
|
|
63
|
+
export interface Percentiles {
|
|
64
|
+
p50: number;
|
|
65
|
+
p90: number;
|
|
66
|
+
p99: number;
|
|
67
|
+
max: number;
|
|
68
|
+
}
|
|
69
|
+
/** Time-to-first-tool latency — `steps[0].startMs` over tool-using sessions. */
|
|
70
|
+
export interface LatencyInsight {
|
|
71
|
+
firstToolMs: Percentiles;
|
|
72
|
+
}
|
|
73
|
+
export interface SegmentDimensions {
|
|
74
|
+
agent: SegmentAgent;
|
|
75
|
+
taskType: TaskType;
|
|
76
|
+
failureTiming: FailureTiming | null;
|
|
77
|
+
}
|
|
78
|
+
/**
|
|
79
|
+
* Tool name → file action. Lowercased. Write-family is a new file; Edit-family
|
|
80
|
+
* is a patch. Anything else is ignored for diff-shape.
|
|
81
|
+
*/
|
|
82
|
+
export declare const FILE_ACTION_TOOLS: ReadonlyArray<{
|
|
83
|
+
action: SegmentFile['action'];
|
|
84
|
+
tools: readonly string[];
|
|
85
|
+
}>;
|
|
86
|
+
type DiffShape = {
|
|
87
|
+
files: readonly SegmentFile[];
|
|
88
|
+
testOnly: boolean;
|
|
89
|
+
newOnly: boolean;
|
|
90
|
+
choreOnly: boolean;
|
|
91
|
+
};
|
|
92
|
+
/**
|
|
93
|
+
* Task-type rules, first match wins. Prompt/branch keywords outrank diff shape
|
|
94
|
+
* except where `shape` is set (test-only files, chore-only files, all-new files).
|
|
95
|
+
*
|
|
96
|
+
* Order is the product priority: test-only work is not a "feature"; a `fix/`
|
|
97
|
+
* branch is a bugfix even if files are new; chore lockfile edits are not features.
|
|
98
|
+
*/
|
|
99
|
+
export declare const TASK_TYPE_RULES: ReadonlyArray<{
|
|
100
|
+
type: Exclude<TaskType, 'other'>;
|
|
101
|
+
patterns: readonly RegExp[];
|
|
102
|
+
shape?: keyof Pick<DiffShape, 'testOnly' | 'newOnly' | 'choreOnly'>;
|
|
103
|
+
}>;
|
|
104
|
+
/**
|
|
105
|
+
* Normalized position of the first failing step inside the session span.
|
|
106
|
+
* Early failures cascade — highest-signal bucket. Exclusive upper bound.
|
|
107
|
+
*/
|
|
108
|
+
export declare const FAILURE_TIMING_BUCKETS: ReadonlyArray<{
|
|
109
|
+
timing: FailureTiming;
|
|
110
|
+
maxExclusive: number;
|
|
111
|
+
}>;
|
|
112
|
+
/** Model × harness unit — the #1 group-by axis. */
|
|
113
|
+
export declare function deriveAgent(session: SegmentSession): SegmentAgent;
|
|
114
|
+
/**
|
|
115
|
+
* Classify the session's work from the opening prompt + diff shape (files
|
|
116
|
+
* touched, new vs edit, test-only). First matching table row wins.
|
|
117
|
+
*/
|
|
118
|
+
export declare function classifyTaskType(session: SegmentSession): TaskType;
|
|
119
|
+
/**
|
|
120
|
+
* Bucket the FIRST failing step by its normalized position in the session
|
|
121
|
+
* span (`firstFailure.startMs / spanMs`). Sessions with no failing step
|
|
122
|
+
* return null — they do not belong on the failure-timing axis.
|
|
123
|
+
*/
|
|
124
|
+
export declare function failureTiming(session: SegmentSession): FailureTiming | null;
|
|
125
|
+
/**
|
|
126
|
+
* Time-to-first-tool latency from `steps[0].startMs` across sessions that
|
|
127
|
+
* recorded at least one step. Nearest-rank percentiles (same formula as the
|
|
128
|
+
* traces index stats) so p99 on `/tmp/traces-real` lands at ~128s.
|
|
129
|
+
*/
|
|
130
|
+
export declare function computeLatency(sessions: readonly SegmentSession[]): LatencyInsight;
|
|
131
|
+
/** All three group-by dimensions for one session. */
|
|
132
|
+
export declare function deriveDimensions(session: SegmentSession): SegmentDimensions;
|
|
133
|
+
export {};
|