@phnx-labs/agents-cli 1.22.52 → 1.22.53
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +146 -0
- package/README.md +1 -1
- package/dist/commands/accounts.js +1 -1
- package/dist/commands/exec.js +16 -9
- package/dist/commands/fleet-capture.js +7 -0
- package/dist/commands/focus.js +2 -0
- package/dist/commands/go.js +2 -1
- package/dist/commands/sessions-inject.js +8 -3
- package/dist/commands/sessions-picker.js +2 -1
- package/dist/commands/sessions.js +30 -20
- package/dist/commands/ssh.js +35 -12
- package/dist/commands/sync.js +44 -0
- package/dist/lib/account-registry.d.ts +15 -5
- package/dist/lib/account-registry.js +150 -50
- package/dist/lib/answer-router.js +2 -1
- package/dist/lib/browser/profiles.d.ts +18 -0
- package/dist/lib/browser/profiles.js +26 -1
- package/dist/lib/browser/registry.d.ts +44 -14
- package/dist/lib/browser/registry.js +141 -45
- package/dist/lib/daemon/runner.js +10 -2
- package/dist/lib/device-config.js +3 -2
- package/dist/lib/devices/config-migration.js +147 -1
- package/dist/lib/devices/device-docs.d.ts +35 -0
- package/dist/lib/devices/device-docs.js +163 -0
- package/dist/lib/devices/discovery-policy.d.ts +14 -2
- package/dist/lib/devices/discovery-policy.js +31 -21
- package/dist/lib/devices/registry.d.ts +11 -5
- package/dist/lib/devices/registry.js +46 -18
- package/dist/lib/exec.d.ts +60 -30
- package/dist/lib/exec.js +65 -27
- package/dist/lib/feed/feed.d.ts +10 -2
- package/dist/lib/feed/feed.js +12 -1
- package/dist/lib/hosts/dispatch.d.ts +4 -3
- package/dist/lib/hosts/dispatch.js +12 -8
- package/dist/lib/hosts/providers/local.d.ts +9 -3
- package/dist/lib/hosts/providers/local.js +23 -12
- package/dist/lib/hosts/reconnect.d.ts +7 -4
- package/dist/lib/hosts/reconnect.js +29 -25
- package/dist/lib/hosts/registry.js +4 -1
- package/dist/lib/hosts/remote-os.js +3 -1
- package/dist/lib/session/active.d.ts +10 -1
- package/dist/lib/session/active.js +7 -1
- package/dist/lib/session/actor-sidecar.d.ts +7 -0
- package/dist/lib/session/actor-sidecar.js +2 -0
- package/dist/lib/session/db.d.ts +1 -1
- package/dist/lib/session/db.js +39 -3
- package/dist/lib/session/discover.js +7 -12
- package/dist/lib/session/live-metadata.js +1 -0
- package/dist/lib/session/pid-registry.d.ts +7 -0
- package/dist/lib/session/prompt.d.ts +15 -0
- package/dist/lib/session/prompt.js +21 -0
- package/dist/lib/session/types.d.ts +17 -0
- package/dist/lib/session/types.js +10 -0
- package/dist/lib/share/worker-template.js +12 -7
- package/dist/lib/state.d.ts +8 -0
- package/dist/lib/state.js +143 -11
- package/dist/lib/terminal/resolve.d.ts +7 -0
- package/dist/lib/terminal/resolve.js +41 -2
- package/dist/lib/traces/insights.d.ts +67 -0
- package/dist/lib/traces/insights.js +178 -0
- package/dist/lib/traces/phenotype.d.ts +67 -0
- package/dist/lib/traces/phenotype.js +437 -0
- package/dist/lib/traces/segments.d.ts +133 -0
- package/dist/lib/traces/segments.js +301 -0
- package/dist/lib/traces/sync.d.ts +33 -0
- package/dist/lib/traces/sync.js +11 -2
- package/dist/lib/types.d.ts +47 -1
- package/dist/lib/watchdog/runner.js +18 -4
- package/package.json +1 -1
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Cross-session failure clustering + time-wasted attribution for the traces
|
|
3
|
+
* insight engine (PHNX-3141) — the piece that turns per-tool error counts
|
|
4
|
+
* into "here is your #1 systemic problem and what it cost."
|
|
5
|
+
*
|
|
6
|
+
* Pure and SQL-shaped: it consumes the same `SyncRow[]` + `tool_calls` rows
|
|
7
|
+
* `buildIndexShard` already loads (no re-parsing of transcripts), so cost
|
|
8
|
+
* stays proportional to this sync's row count, never the full corpus.
|
|
9
|
+
*
|
|
10
|
+
* `tool_calls` rows carry `ordinal`/`timestamp` per call within a session
|
|
11
|
+
* (`db.ts`'s `idx_tool_calls_session ON tool_calls(session_id, ordinal)`), which
|
|
12
|
+
* is enough to reconstruct per-session call order and inter-call gaps without a
|
|
13
|
+
* full `SessionTrajectory` — that is what makes this incremental at scale.
|
|
14
|
+
*
|
|
15
|
+
* Scope note: `FailureSignature` does not yet carry a `phenotype`
|
|
16
|
+
* (false-termination / out-of-order / …, `phenotype.ts`) — classifying that
|
|
17
|
+
* needs the full derived trajectory (turns, ordered steps, gaps), which is
|
|
18
|
+
* only ever materialized per-session during upload, not cached the way
|
|
19
|
+
* `InsightFacets` is. Folding it in is a real, scoped follow-up (see
|
|
20
|
+
* `cli/AGENTS.md`), not a silent omission.
|
|
21
|
+
*/
|
|
22
|
+
import { type TraceFailureCause } from './classify.js';
|
|
23
|
+
import { type LatencyInsight } from './segments.js';
|
|
24
|
+
import { type SyncRow, type ToolCallRow, type TracesIndexShard } from './sync.js';
|
|
25
|
+
export interface FailureSignature {
|
|
26
|
+
tool: string;
|
|
27
|
+
cause: TraceFailureCause;
|
|
28
|
+
/** Normalized error text — volatile tokens (ids, counts, countdowns) stripped so instances fold together. */
|
|
29
|
+
key: string;
|
|
30
|
+
}
|
|
31
|
+
export interface FailurePattern {
|
|
32
|
+
/** Stable hash of the signature — deep-linkable, unaffected by row order. */
|
|
33
|
+
id: string;
|
|
34
|
+
label: string;
|
|
35
|
+
signature: FailureSignature;
|
|
36
|
+
/** Distinct sessions this pattern occurred in. */
|
|
37
|
+
sessions: number;
|
|
38
|
+
/** Total failing calls matching this signature. */
|
|
39
|
+
occurrences: number;
|
|
40
|
+
/** Estimated ms of retry/stall time attributable to this pattern (see attribution rule below). */
|
|
41
|
+
wastedMs: number;
|
|
42
|
+
/** Bounded example session ids for drill-down. */
|
|
43
|
+
exampleSessionIds: string[];
|
|
44
|
+
/** Movement vs the same pattern id in the previous shard. */
|
|
45
|
+
drift: 'up' | 'flat' | 'down';
|
|
46
|
+
}
|
|
47
|
+
export interface ComputedInsights {
|
|
48
|
+
/** Top-K patterns ranked by wastedMs (impact) — a rare 1-occurrence/8h loop still surfaces. */
|
|
49
|
+
failurePatterns: FailurePattern[];
|
|
50
|
+
/** Sum of wastedMs across every cluster found this sync, not just the top-K rows above. */
|
|
51
|
+
wastedMsTotal: number;
|
|
52
|
+
latency: LatencyInsight;
|
|
53
|
+
}
|
|
54
|
+
/** Strip volatile tokens from a failure's evidence text so repeat instances hash identically. */
|
|
55
|
+
export declare function normalizeErrorKey(desc: string, raw: string | null): string;
|
|
56
|
+
/**
|
|
57
|
+
* Cluster failed tool calls into ranked patterns and estimate the wasted time
|
|
58
|
+
* behind each, plus device-wide time-to-first-tool latency.
|
|
59
|
+
*
|
|
60
|
+
* wastedMs attribution: the gap between a failed call and the NEXT call in the
|
|
61
|
+
* same session counts as wasted when either (a) the next call repeats the same
|
|
62
|
+
* signature (a retry loop) or (b) the gap itself is a stall (≥60s) — an idle
|
|
63
|
+
* gap unrelated to a nearby failure is never counted. This is an estimate, not
|
|
64
|
+
* ground truth (a stall could be legitimate user think-time); it is not
|
|
65
|
+
* inflated by folding in ordinary processing time between unrelated calls.
|
|
66
|
+
*/
|
|
67
|
+
export declare function computeInsights(rows: readonly SyncRow[], calls: readonly ToolCallRow[], prevShard?: TracesIndexShard | null): ComputedInsights;
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Cross-session failure clustering + time-wasted attribution for the traces
|
|
3
|
+
* insight engine (PHNX-3141) — the piece that turns per-tool error counts
|
|
4
|
+
* into "here is your #1 systemic problem and what it cost."
|
|
5
|
+
*
|
|
6
|
+
* Pure and SQL-shaped: it consumes the same `SyncRow[]` + `tool_calls` rows
|
|
7
|
+
* `buildIndexShard` already loads (no re-parsing of transcripts), so cost
|
|
8
|
+
* stays proportional to this sync's row count, never the full corpus.
|
|
9
|
+
*
|
|
10
|
+
* `tool_calls` rows carry `ordinal`/`timestamp` per call within a session
|
|
11
|
+
* (`db.ts`'s `idx_tool_calls_session ON tool_calls(session_id, ordinal)`), which
|
|
12
|
+
* is enough to reconstruct per-session call order and inter-call gaps without a
|
|
13
|
+
* full `SessionTrajectory` — that is what makes this incremental at scale.
|
|
14
|
+
*
|
|
15
|
+
* Scope note: `FailureSignature` does not yet carry a `phenotype`
|
|
16
|
+
* (false-termination / out-of-order / …, `phenotype.ts`) — classifying that
|
|
17
|
+
* needs the full derived trajectory (turns, ordered steps, gaps), which is
|
|
18
|
+
* only ever materialized per-session during upload, not cached the way
|
|
19
|
+
* `InsightFacets` is. Folding it in is a real, scoped follow-up (see
|
|
20
|
+
* `cli/AGENTS.md`), not a silent omission.
|
|
21
|
+
*/
|
|
22
|
+
import { classifyCause } from './classify.js';
|
|
23
|
+
import { computeLatency } from './segments.js';
|
|
24
|
+
import { failureDescription } from './sync.js';
|
|
25
|
+
// ---------------------------------------------------------------------------
|
|
26
|
+
// Tunables
|
|
27
|
+
// ---------------------------------------------------------------------------
|
|
28
|
+
/** Bounded shard size — patterns, not sessions, so 738 or 738k render identically. */
|
|
29
|
+
const TOP_K_PATTERNS = 25;
|
|
30
|
+
const MAX_EXAMPLE_SESSIONS = 5;
|
|
31
|
+
/** A gap this long right after a failure reads as an idle stall, not think-time (matches sync.ts's own "stalled Xm" threshold). */
|
|
32
|
+
const STALL_MS = 60_000;
|
|
33
|
+
// ---------------------------------------------------------------------------
|
|
34
|
+
// Signature normalization — fold volatile per-instance text together
|
|
35
|
+
// ---------------------------------------------------------------------------
|
|
36
|
+
const VOLATILE_TOKEN_PATTERNS = [
|
|
37
|
+
{ pattern: /\bfor user [\w.-]+/gi, replacement: 'for user _' },
|
|
38
|
+
{ pattern: /\btry again in [\w.]+s?\b/gi, replacement: 'try again in _s' },
|
|
39
|
+
{ pattern: /\b[0-9a-f]{7,40}\b/gi, replacement: '_sha_' },
|
|
40
|
+
{ pattern: /\b[\w.-]+@[\w.-]+\.\w+\b/gi, replacement: '_email_' },
|
|
41
|
+
{ pattern: /\b\d+\b/g, replacement: '_n_' },
|
|
42
|
+
];
|
|
43
|
+
/** Strip volatile tokens from a failure's evidence text so repeat instances hash identically. */
|
|
44
|
+
export function normalizeErrorKey(desc, raw) {
|
|
45
|
+
let text = (raw && raw.trim().length > 0 ? raw : desc).toLowerCase();
|
|
46
|
+
for (const { pattern, replacement } of VOLATILE_TOKEN_PATTERNS) {
|
|
47
|
+
text = text.replace(pattern, replacement);
|
|
48
|
+
}
|
|
49
|
+
return text.replace(/\s+/g, ' ').trim().slice(0, 160);
|
|
50
|
+
}
|
|
51
|
+
function hashSignature(tool, cause, key) {
|
|
52
|
+
const input = `${tool} ${cause} ${key}`;
|
|
53
|
+
let hash = 5381;
|
|
54
|
+
for (let i = 0; i < input.length; i++) {
|
|
55
|
+
hash = ((hash << 5) + hash + input.charCodeAt(i)) >>> 0;
|
|
56
|
+
}
|
|
57
|
+
return hash.toString(16).padStart(8, '0');
|
|
58
|
+
}
|
|
59
|
+
// ---------------------------------------------------------------------------
|
|
60
|
+
// Label rules — a table, not an if/else-by-name chain (matches segments.ts's TASK_TYPE_RULES)
|
|
61
|
+
// ---------------------------------------------------------------------------
|
|
62
|
+
const LABEL_RULES = [
|
|
63
|
+
{ pattern: /rate limit/i, label: 'rate limit back-off loop' },
|
|
64
|
+
{ pattern: /permission denied/i, label: 'permission denied' },
|
|
65
|
+
{ pattern: /not found|no such file/i, label: 'missing resource' },
|
|
66
|
+
{ pattern: /timed? ?out/i, label: 'timeout' },
|
|
67
|
+
{ pattern: /econnrefused|connection refused|network/i, label: 'network error' },
|
|
68
|
+
{ pattern: /conflict|diverged/i, label: 'git conflict' },
|
|
69
|
+
];
|
|
70
|
+
function labelFor(tool, cause, key) {
|
|
71
|
+
if (cause === 'guard')
|
|
72
|
+
return `${tool}: git guard denial`;
|
|
73
|
+
if (cause === 'hook')
|
|
74
|
+
return `${tool}: hook denial`;
|
|
75
|
+
const rule = LABEL_RULES.find((row) => row.pattern.test(key));
|
|
76
|
+
return `${tool}: ${rule ? rule.label : key.slice(0, 48)}`;
|
|
77
|
+
}
|
|
78
|
+
// ---------------------------------------------------------------------------
|
|
79
|
+
// Public API
|
|
80
|
+
// ---------------------------------------------------------------------------
|
|
81
|
+
/**
|
|
82
|
+
* Cluster failed tool calls into ranked patterns and estimate the wasted time
|
|
83
|
+
* behind each, plus device-wide time-to-first-tool latency.
|
|
84
|
+
*
|
|
85
|
+
* wastedMs attribution: the gap between a failed call and the NEXT call in the
|
|
86
|
+
* same session counts as wasted when either (a) the next call repeats the same
|
|
87
|
+
* signature (a retry loop) or (b) the gap itself is a stall (≥60s) — an idle
|
|
88
|
+
* gap unrelated to a nearby failure is never counted. This is an estimate, not
|
|
89
|
+
* ground truth (a stall could be legitimate user think-time); it is not
|
|
90
|
+
* inflated by folding in ordinary processing time between unrelated calls.
|
|
91
|
+
*/
|
|
92
|
+
export function computeInsights(rows, calls, prevShard) {
|
|
93
|
+
const bySession = new Map();
|
|
94
|
+
for (const call of calls) {
|
|
95
|
+
const list = bySession.get(call.session_id);
|
|
96
|
+
if (list)
|
|
97
|
+
list.push(call);
|
|
98
|
+
else
|
|
99
|
+
bySession.set(call.session_id, [call]);
|
|
100
|
+
}
|
|
101
|
+
const groups = new Map();
|
|
102
|
+
for (const [sessionId, sessionCalls] of bySession) {
|
|
103
|
+
const ordered = [...sessionCalls].sort((a, b) => a.ordinal - b.ordinal);
|
|
104
|
+
for (let i = 0; i < ordered.length; i++) {
|
|
105
|
+
const call = ordered[i];
|
|
106
|
+
if (call.outcome !== 'error')
|
|
107
|
+
continue;
|
|
108
|
+
const cause = classifyCause(call);
|
|
109
|
+
const key = normalizeErrorKey(failureDescription(call, cause), call.error);
|
|
110
|
+
const groupKey = `${call.tool} ${cause} ${key}`;
|
|
111
|
+
let group = groups.get(groupKey);
|
|
112
|
+
if (!group) {
|
|
113
|
+
group = { tool: call.tool, cause, key, sessions: new Set(), occurrences: 0, wastedMs: 0, examples: [] };
|
|
114
|
+
groups.set(groupKey, group);
|
|
115
|
+
}
|
|
116
|
+
group.occurrences++;
|
|
117
|
+
group.sessions.add(sessionId);
|
|
118
|
+
if (group.examples.length < MAX_EXAMPLE_SESSIONS && !group.examples.includes(sessionId)) {
|
|
119
|
+
group.examples.push(sessionId);
|
|
120
|
+
}
|
|
121
|
+
const next = ordered[i + 1];
|
|
122
|
+
if (!next)
|
|
123
|
+
continue;
|
|
124
|
+
const gapMs = Date.parse(next.timestamp) - Date.parse(call.timestamp);
|
|
125
|
+
if (!Number.isFinite(gapMs) || gapMs <= 0)
|
|
126
|
+
continue;
|
|
127
|
+
const nextIsSameFailure = next.outcome === 'error' &&
|
|
128
|
+
next.tool === call.tool &&
|
|
129
|
+
classifyCause(next) === cause &&
|
|
130
|
+
normalizeErrorKey(failureDescription(next, cause), next.error) === key;
|
|
131
|
+
if (nextIsSameFailure || gapMs >= STALL_MS) {
|
|
132
|
+
group.wastedMs += gapMs;
|
|
133
|
+
}
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
const prevById = new Map((prevShard?.failurePatterns ?? []).map((p) => [p.id, p]));
|
|
137
|
+
const allPatterns = [...groups.values()].map((group) => {
|
|
138
|
+
const id = hashSignature(group.tool, group.cause, group.key);
|
|
139
|
+
const prev = prevById.get(id);
|
|
140
|
+
const drift = !prev
|
|
141
|
+
? 'up'
|
|
142
|
+
: group.occurrences > prev.occurrences
|
|
143
|
+
? 'up'
|
|
144
|
+
: group.occurrences < prev.occurrences
|
|
145
|
+
? 'down'
|
|
146
|
+
: 'flat';
|
|
147
|
+
return {
|
|
148
|
+
id,
|
|
149
|
+
label: labelFor(group.tool, group.cause, group.key),
|
|
150
|
+
signature: { tool: group.tool, cause: group.cause, key: group.key },
|
|
151
|
+
sessions: group.sessions.size,
|
|
152
|
+
occurrences: group.occurrences,
|
|
153
|
+
wastedMs: group.wastedMs,
|
|
154
|
+
exampleSessionIds: group.examples,
|
|
155
|
+
drift,
|
|
156
|
+
};
|
|
157
|
+
});
|
|
158
|
+
const wastedMsTotal = allPatterns.reduce((sum, p) => sum + p.wastedMs, 0);
|
|
159
|
+
const failurePatterns = [...allPatterns]
|
|
160
|
+
.sort((a, b) => b.wastedMs - a.wastedMs || b.occurrences - a.occurrences || a.id.localeCompare(b.id))
|
|
161
|
+
.slice(0, TOP_K_PATTERNS);
|
|
162
|
+
const latency = computeLatency(firstToolSegments(rows, bySession));
|
|
163
|
+
return { failurePatterns, wastedMsTotal, latency };
|
|
164
|
+
}
|
|
165
|
+
/** Synthesize one-step SegmentSessions carrying only the time-to-first-tool offset, for computeLatency() reuse. */
|
|
166
|
+
function firstToolSegments(rows, bySession) {
|
|
167
|
+
return rows.flatMap((row) => {
|
|
168
|
+
const sessionCalls = bySession.get(row.id);
|
|
169
|
+
if (!sessionCalls || sessionCalls.length === 0)
|
|
170
|
+
return [];
|
|
171
|
+
const first = sessionCalls.reduce((min, call) => (call.ordinal < min.ordinal ? call : min));
|
|
172
|
+
const sessionStartMs = Date.parse(row.timestamp);
|
|
173
|
+
const firstCallMs = Date.parse(first.timestamp);
|
|
174
|
+
if (!Number.isFinite(sessionStartMs) || !Number.isFinite(firstCallMs))
|
|
175
|
+
return [];
|
|
176
|
+
return [{ steps: [{ startMs: Math.max(0, firstCallMs - sessionStartMs) }] }];
|
|
177
|
+
});
|
|
178
|
+
}
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Failure phenotype classifier + outcome taxonomy for the traces insight engine.
|
|
3
|
+
*
|
|
4
|
+
* Both functions are pure: they take a redacted {@link SessionDetail} (the same
|
|
5
|
+
* shape the traces sync writes to `sessions/<id>.json`) and return a decision
|
|
6
|
+
* derived only from the already-derived step/gap/meta signal. They never read
|
|
7
|
+
* raw transcript text and never fabricate a signal that is not in the data.
|
|
8
|
+
*
|
|
9
|
+
* The rubrics are expressed as data-driven tables of conditions, not as
|
|
10
|
+
* if/else-by-name chains. Each table row is a named phenotype/outcome with a
|
|
11
|
+
* declarative predicate; the classifier walks the table in priority order and
|
|
12
|
+
* returns the first match, or the honest lower-confidence default when no
|
|
13
|
+
* high-confidence signal is present.
|
|
14
|
+
*/
|
|
15
|
+
import type { SessionDetail } from './sync.js';
|
|
16
|
+
/** A failure mode detectable from the derived trajectory of a session. */
|
|
17
|
+
export type FailurePhenotype = 'false-termination' | 'premature-completion' | 'out-of-order' | 'failure-to-act';
|
|
18
|
+
/**
|
|
19
|
+
* The coarse outcome of a session's work.
|
|
20
|
+
*
|
|
21
|
+
* - `merged` : explicit PR/branch merge signal in the steps (high confidence).
|
|
22
|
+
* - `tests-green` : explicit test command returned ok with no later failure (high confidence).
|
|
23
|
+
* - `partial` : progress was made but no landing/test signal is present (low confidence default).
|
|
24
|
+
* - `abandoned` : errored, stalled, and unresolved.
|
|
25
|
+
* - `human-takeover`: the final substantive action was a human-facing ask/wait.
|
|
26
|
+
* - `invalid-env` : environment/setup failures dominated the session.
|
|
27
|
+
*/
|
|
28
|
+
export type TraceOutcome = 'merged' | 'tests-green' | 'partial' | 'abandoned' | 'human-takeover' | 'invalid-env';
|
|
29
|
+
export interface PhenotypeResult {
|
|
30
|
+
phenotype: FailurePhenotype | null;
|
|
31
|
+
reason: string;
|
|
32
|
+
}
|
|
33
|
+
export interface OutcomeResult {
|
|
34
|
+
outcome: TraceOutcome;
|
|
35
|
+
confidence: 'high' | 'medium' | 'low';
|
|
36
|
+
reason: string;
|
|
37
|
+
}
|
|
38
|
+
/**
|
|
39
|
+
* Classify the failure phenotype of a session from its derived trajectory.
|
|
40
|
+
*
|
|
41
|
+
* Definitions (from agent-failure research):
|
|
42
|
+
* - `false-termination` — stopped with an unresolved error.
|
|
43
|
+
* - `premature-completion` — declared done while tests were failing or no
|
|
44
|
+
* verification step ran for the engineering work.
|
|
45
|
+
* - `out-of-order` — a write/edit step occurred before any read/plan of the
|
|
46
|
+
* target.
|
|
47
|
+
* - `failure-to-act` — stalled or produced no meaningful tool use.
|
|
48
|
+
*
|
|
49
|
+
* Returns `null` when none of the failure phenotypes apply.
|
|
50
|
+
*/
|
|
51
|
+
export declare function classifyPhenotype(session: SessionDetail): FailurePhenotype | null;
|
|
52
|
+
/** Detailed phenotype result with a human-readable reason. */
|
|
53
|
+
export declare function classifyPhenotypeDetailed(session: SessionDetail): PhenotypeResult;
|
|
54
|
+
/**
|
|
55
|
+
* Derive the coarse outcome of a session from its end state + tool signals.
|
|
56
|
+
*
|
|
57
|
+
* `merged` and `tests-green` are high-confidence only when an explicit signal
|
|
58
|
+
* is present in the derived steps. When that signal is genuinely not in the
|
|
59
|
+
* data, the function returns the honest lower-confidence value (`partial` for
|
|
60
|
+
* completed work without a landing signal, `abandoned` for errored/unresolved
|
|
61
|
+
* work, `invalid-env` for setup-dominant failures, `human-takeover` when the
|
|
62
|
+
* session ends on a human-facing ask).
|
|
63
|
+
*/
|
|
64
|
+
export declare function deriveOutcome(session: SessionDetail): TraceOutcome;
|
|
65
|
+
/** Detailed outcome result with confidence and a human-readable reason. */
|
|
66
|
+
export declare function deriveOutcomeDetailed(session: SessionDetail): OutcomeResult;
|
|
67
|
+
export type { SessionDetail } from './sync.js';
|