@phnx-labs/agents-cli 1.22.60 → 1.22.61
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +49 -0
- package/dist/cli/command-registry.d.ts +1 -0
- package/dist/cli/command-registry.js +2 -0
- package/dist/commands/browser.js +9 -4
- package/dist/commands/doctor.js +1 -1
- package/dist/commands/exec.js +35 -1
- package/dist/commands/harness-hooks.d.ts +55 -0
- package/dist/commands/harness-hooks.js +104 -0
- package/dist/commands/harness-wizard.d.ts +33 -14
- package/dist/commands/harness-wizard.js +53 -23
- package/dist/commands/harness.d.ts +14 -0
- package/dist/commands/harness.js +86 -5
- package/dist/commands/reminders.d.ts +9 -0
- package/dist/commands/reminders.js +49 -0
- package/dist/commands/run-account-picker.d.ts +14 -0
- package/dist/commands/run-account-picker.js +13 -0
- package/dist/commands/teams.d.ts +1 -1
- package/dist/commands/teams.js +9 -3
- package/dist/index.js +9 -0
- package/dist/lib/accounting/rotate.d.ts +63 -0
- package/dist/lib/accounting/rotate.js +229 -13
- package/dist/lib/browser/drivers/local.d.ts +11 -0
- package/dist/lib/browser/drivers/local.js +26 -0
- package/dist/lib/browser/profiles.js +8 -6
- package/dist/lib/browser/service.d.ts +12 -8
- package/dist/lib/browser/service.js +38 -10
- package/dist/lib/claude-statusline.d.ts +14 -1
- package/dist/lib/claude-statusline.js +27 -2
- package/dist/lib/daemon/runner.js +17 -2
- package/dist/lib/devices/doctor-findings.d.ts +1 -1
- package/dist/lib/devices/doctor-findings.js +22 -4
- package/dist/lib/doctor-diff.d.ts +21 -5
- package/dist/lib/doctor-diff.js +242 -76
- package/dist/lib/feed/events.d.ts +1 -1
- package/dist/lib/feed/events.js +25 -16
- package/dist/lib/github/gh-overload.d.ts +58 -0
- package/dist/lib/github/gh-overload.js +246 -0
- package/dist/lib/github/rest.d.ts +64 -0
- package/dist/lib/github/rest.js +111 -0
- package/dist/lib/harness-connection-test.d.ts +57 -0
- package/dist/lib/harness-connection-test.js +80 -0
- package/dist/lib/heal.js +8 -3
- package/dist/lib/installations/shims.d.ts +22 -0
- package/dist/lib/installations/shims.js +104 -0
- package/dist/lib/linear-project-counts.js +8 -0
- package/dist/lib/linear-rate-limit.d.ts +26 -0
- package/dist/lib/linear-rate-limit.js +163 -0
- package/dist/lib/mcp.d.ts +9 -0
- package/dist/lib/mcp.js +37 -1
- package/dist/lib/open-url.js +5 -3
- package/dist/lib/permissions.d.ts +28 -0
- package/dist/lib/permissions.js +156 -1
- package/dist/lib/refresh.js +9 -1
- package/dist/lib/reminders.d.ts +29 -0
- package/dist/lib/reminders.js +88 -0
- package/dist/lib/resource-content-diff.d.ts +33 -0
- package/dist/lib/resource-content-diff.js +103 -0
- package/dist/lib/rules/compile.d.ts +7 -0
- package/dist/lib/rules/compile.js +7 -1
- package/dist/lib/session/active.d.ts +41 -4
- package/dist/lib/session/active.js +58 -7
- package/dist/lib/session/host-link.d.ts +22 -0
- package/dist/lib/session/host-link.js +40 -4
- package/dist/lib/session/trajectory.d.ts +42 -0
- package/dist/lib/session/trajectory.js +46 -27
- package/dist/lib/ssh-exec.d.ts +30 -0
- package/dist/lib/ssh-exec.js +37 -5
- package/dist/lib/startup/command-registry.js +1 -1
- package/dist/lib/subagents-registry.d.ts +18 -0
- package/dist/lib/subagents-registry.js +79 -0
- package/dist/lib/teams/agents.d.ts +12 -0
- package/dist/lib/teams/agents.js +51 -0
- package/dist/lib/traces/schema2-build.d.ts +85 -0
- package/dist/lib/traces/schema2-build.js +637 -0
- package/dist/lib/traces/schema2-danger.d.ts +36 -0
- package/dist/lib/traces/schema2-danger.js +185 -0
- package/dist/lib/traces/schema2.d.ts +149 -0
- package/dist/lib/traces/schema2.js +20 -0
- package/dist/lib/traces/sync.d.ts +93 -0
- package/dist/lib/traces/sync.js +75 -22
- package/dist/lib/traces/worker-template.js +5 -0
- package/dist/lib/uninstall.js +10 -1
- package/dist/lib/workflows.d.ts +11 -0
- package/dist/lib/workflows.js +67 -8
- package/package.json +1 -1
|
@@ -0,0 +1,637 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* schema2-build — the PRODUCER's per-tool mappers + `buildSessionDetailV2`
|
|
3
|
+
* (PHNX-3442 step 2, increments 2-4).
|
|
4
|
+
*
|
|
5
|
+
* Populates the `SessionStepV2` discriminated union (schema2.ts) from the parsed
|
|
6
|
+
* session events, reusing the SAME infrastructure the schema-1 path already uses:
|
|
7
|
+
*
|
|
8
|
+
* - the callId pairing loop (`pairSteps` in session/trajectory.ts) — so a step's
|
|
9
|
+
* (use event, result event) triple is recovered without a duplicate loop;
|
|
10
|
+
* - bash unwrap/tokenize/classify (`session/bash-command.ts`) + the effective
|
|
11
|
+
* program resolver (`effectiveProgram`);
|
|
12
|
+
* - the meta / whereItWentWrong / surfacedToolFailures / active-time helpers
|
|
13
|
+
* factored out of sync.ts (`buildDetailMeta`, `buildWhereItWentWrong`, …).
|
|
14
|
+
*
|
|
15
|
+
* The command/patch/output PARSING lives here; the worker stores the shard
|
|
16
|
+
* opaquely and the console reads the union directly and never reparses (spec §5).
|
|
17
|
+
*
|
|
18
|
+
* category / risk / categoryMetrics are DELIBERATELY omitted from the schema-2
|
|
19
|
+
* detail: the shipped consumer (`decodeSessionDetail` → coerceCategory/Risk/Metrics)
|
|
20
|
+
* backfills them to the same neutral defaults it uses for schema-1, so computing
|
|
21
|
+
* them here would be inventing session-level signal this step does not own.
|
|
22
|
+
*/
|
|
23
|
+
import { createHash } from 'node:crypto';
|
|
24
|
+
import { redactSecrets } from '../redact.js';
|
|
25
|
+
import { classifyBashCommand, tokenizeBash, unwrapCommand, } from '../session/bash-command.js';
|
|
26
|
+
import { computeSummaryStats } from '../session/render.js';
|
|
27
|
+
import { extractShellPrograms } from '../session/shell-programs.js';
|
|
28
|
+
import { effectiveProgram, eventTimestampsMs, pairSteps, } from '../session/trajectory.js';
|
|
29
|
+
import { classifyActionDanger } from './schema2-danger.js';
|
|
30
|
+
import { activeMsFromTrajectory, buildDetailMeta, buildWhereItWentWrong, } from './sync.js';
|
|
31
|
+
// ---------------------------------------------------------------------------
|
|
32
|
+
// Small value helpers
|
|
33
|
+
// ---------------------------------------------------------------------------
|
|
34
|
+
/** Cap on a single preview's characters — the same 500-ish bound parse.ts uses. */
|
|
35
|
+
const PREVIEW_MAX = 2000;
|
|
36
|
+
function shortHash(text) {
|
|
37
|
+
return createHash('sha256').update(text).digest('hex').slice(0, 16);
|
|
38
|
+
}
|
|
39
|
+
/**
|
|
40
|
+
* A bounded, redacted preview of text. `truncated` and `originalBytes` are honest —
|
|
41
|
+
* the console needs to know whether it is seeing the whole thing (spec: the UI must
|
|
42
|
+
* distinguish complete from truncated output). `originalBytes` is the UTF-8 byte
|
|
43
|
+
* length of the FULL text, before clipping.
|
|
44
|
+
*/
|
|
45
|
+
function textPreview(raw, redact, knownSecrets) {
|
|
46
|
+
if (raw === undefined || raw === null || raw.length === 0)
|
|
47
|
+
return undefined;
|
|
48
|
+
const originalBytes = Buffer.byteLength(raw, 'utf8');
|
|
49
|
+
const clipped = raw.length > PREVIEW_MAX ? raw.slice(0, PREVIEW_MAX) : raw;
|
|
50
|
+
const text = redact ? redactSecrets(clipped, knownSecrets) : clipped;
|
|
51
|
+
return { text, truncated: raw.length > PREVIEW_MAX, originalBytes };
|
|
52
|
+
}
|
|
53
|
+
function stringArg(args, ...keys) {
|
|
54
|
+
if (!args)
|
|
55
|
+
return undefined;
|
|
56
|
+
for (const key of keys) {
|
|
57
|
+
const v = args[key];
|
|
58
|
+
if (typeof v === 'string' && v.length > 0)
|
|
59
|
+
return v;
|
|
60
|
+
}
|
|
61
|
+
return undefined;
|
|
62
|
+
}
|
|
63
|
+
function numberArg(args, ...keys) {
|
|
64
|
+
if (!args)
|
|
65
|
+
return undefined;
|
|
66
|
+
for (const key of keys) {
|
|
67
|
+
const v = args[key];
|
|
68
|
+
if (typeof v === 'number' && Number.isFinite(v))
|
|
69
|
+
return v;
|
|
70
|
+
}
|
|
71
|
+
return undefined;
|
|
72
|
+
}
|
|
73
|
+
/** The result event's ExecutionResult (exit/status/error codes + a combined output preview). */
|
|
74
|
+
function resultOf(resultEvent, redact, knownSecrets) {
|
|
75
|
+
const result = {};
|
|
76
|
+
if (!resultEvent)
|
|
77
|
+
return result;
|
|
78
|
+
if (typeof resultEvent.exitCode === 'number')
|
|
79
|
+
result.exitCode = resultEvent.exitCode;
|
|
80
|
+
if (typeof resultEvent.statusCode === 'number')
|
|
81
|
+
result.statusCode = resultEvent.statusCode;
|
|
82
|
+
if (typeof resultEvent.errorCode === 'string')
|
|
83
|
+
result.errorCode = resultEvent.errorCode;
|
|
84
|
+
const combined = textPreview(resultEvent.output ?? resultEvent.content, redact, knownSecrets);
|
|
85
|
+
if (combined)
|
|
86
|
+
result.combined = combined;
|
|
87
|
+
return result;
|
|
88
|
+
}
|
|
89
|
+
function stepOutcome(step) {
|
|
90
|
+
const o = step.outcome;
|
|
91
|
+
if (o === 'ok' || o === 'error' || o === 'unknown')
|
|
92
|
+
return o;
|
|
93
|
+
return 'unknown';
|
|
94
|
+
}
|
|
95
|
+
/**
|
|
96
|
+
* `at-least` when the output was truncated by the parser's per-event cap, else
|
|
97
|
+
* `exact`. The parser caps `output` centrally (parse.ts `maxToolOutputChars`), so a
|
|
98
|
+
* result whose text hit the cap under-counts — the count is a floor, not the truth.
|
|
99
|
+
*/
|
|
100
|
+
function countLines(resultEvent) {
|
|
101
|
+
if (!resultEvent)
|
|
102
|
+
return undefined;
|
|
103
|
+
const text = resultEvent.output ?? resultEvent.content;
|
|
104
|
+
if (typeof text !== 'string' || text.length === 0)
|
|
105
|
+
return undefined;
|
|
106
|
+
const lines = text.split('\n');
|
|
107
|
+
// A trailing newline yields a final empty element — don't count it as a line.
|
|
108
|
+
const value = lines.length > 0 && lines[lines.length - 1] === '' ? lines.length - 1 : lines.length;
|
|
109
|
+
// The parser truncates long tool output; we can't see the original length here,
|
|
110
|
+
// so a preview that fills the cap is treated as a floor. PREVIEW-independent:
|
|
111
|
+
// parse.ts already clipped, so the safest signal is whether the text looks cut.
|
|
112
|
+
const truncated = text.length >= PREVIEW_MAX;
|
|
113
|
+
return { value, relation: truncated ? 'at-least' : 'exact' };
|
|
114
|
+
}
|
|
115
|
+
// ---------------------------------------------------------------------------
|
|
116
|
+
// Bash unwrapping — extend unwrapCommand for the shell-exec wrappers it misses
|
|
117
|
+
// ---------------------------------------------------------------------------
|
|
118
|
+
/**
|
|
119
|
+
* `unwrapCommand` (bash-command.ts) strips VAR=/sudo/cd&&/npx/loops/subshells but
|
|
120
|
+
* NOT an interpreter wrapper like `/bin/zsh -lc "…"`, `bash -lc '…'`, or `sh -c …`
|
|
121
|
+
* — the exact shape the managed runner wraps every command in. Peel that first,
|
|
122
|
+
* then hand the inner payload to the existing unwrapper so all the wrappers it DOES
|
|
123
|
+
* know still apply. One extra rule, at the source, not a fork of unwrapCommand.
|
|
124
|
+
*/
|
|
125
|
+
export function unwrapShellExec(command) {
|
|
126
|
+
const s = command.trim();
|
|
127
|
+
// <interpreter> [flags] -c|-lc "PAYLOAD" — interpreter is bash/zsh/sh/dash/ksh,
|
|
128
|
+
// possibly a full path; the -c flag may be clustered with login/interactive
|
|
129
|
+
// flags (`-lc`, `-ic`). The payload is the last quoted argument.
|
|
130
|
+
const m = s.match(/^(?:\S*\/)?(?:bash|zsh|sh|dash|ksh)\s+(?:-[a-zA-Z]*c[a-zA-Z]*)\s+(['"])([\s\S]*)\1\s*$/);
|
|
131
|
+
if (m)
|
|
132
|
+
return unwrapShellExec(m[2]);
|
|
133
|
+
return unwrapCommand(s);
|
|
134
|
+
}
|
|
135
|
+
/**
|
|
136
|
+
* Map a classifier `BashCategory` (the rich vcs|build-test|install|… taxonomy) to
|
|
137
|
+
* the coarse schema-2 `BashCategory` (build|test|git|network|other). `build-test`
|
|
138
|
+
* needs the argv/subcommand to decide build vs test — `bun test` is test, `bun
|
|
139
|
+
* build` is build — so this takes the tokenized argv too.
|
|
140
|
+
*/
|
|
141
|
+
export function mapBashCategory(cat, argv) {
|
|
142
|
+
switch (cat) {
|
|
143
|
+
case 'vcs':
|
|
144
|
+
return 'git';
|
|
145
|
+
case 'remote':
|
|
146
|
+
case 'http':
|
|
147
|
+
return 'network';
|
|
148
|
+
case 'build-test': {
|
|
149
|
+
const lower = argv.map((t) => t.toLowerCase());
|
|
150
|
+
const looksTest = lower.some((t) => t === 'test' || t === 't' || /vitest|jest|pytest|mocha/.test(t) ||
|
|
151
|
+
t === '--test' || /(^|:)test(:|$)/.test(t));
|
|
152
|
+
return looksTest ? 'test' : 'build';
|
|
153
|
+
}
|
|
154
|
+
default:
|
|
155
|
+
return 'other';
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
/**
|
|
159
|
+
* Whether a segment's argv is COMPLETE — i.e. no dynamic node (command
|
|
160
|
+
* substitution, process substitution, arithmetic/param expansion, glob) could
|
|
161
|
+
* change what actually ran. Reuses the shell parser's occurrence walk indirectly:
|
|
162
|
+
* a segment whose reconstructed programs from `extractShellPrograms` are all static
|
|
163
|
+
* is complete. We approximate with the parser's diagnostics + a substitution scan,
|
|
164
|
+
* because the schema-2 argv is the tokenizeBash split (shlex), which cannot itself
|
|
165
|
+
* report expansion.
|
|
166
|
+
*/
|
|
167
|
+
function argvComplete(source) {
|
|
168
|
+
// A command/process substitution or an unexpanded var/glob means the literal
|
|
169
|
+
// argv we tokenized is not the whole story.
|
|
170
|
+
if (/\$\(|\$\{|`|<\(|\)\s*$/.test(source) && /\$\(|\$\{|`|<\(/.test(source))
|
|
171
|
+
return false;
|
|
172
|
+
if (/\$[A-Za-z_]/.test(source))
|
|
173
|
+
return false; // a bare $VAR expansion
|
|
174
|
+
if (/[*?]/.test(source) && !/['"][^'"]*[*?]/.test(source))
|
|
175
|
+
return false; // an unquoted glob
|
|
176
|
+
return true;
|
|
177
|
+
}
|
|
178
|
+
/** Build the per-segment BashAction list for a bash command. */
|
|
179
|
+
export function buildBashActions(unwrapped) {
|
|
180
|
+
const segments = tokenizeBash(unwrapped);
|
|
181
|
+
const actions = [];
|
|
182
|
+
// Recover each segment's raw source text for `source`/argvComplete: tokenizeBash
|
|
183
|
+
// drops the operators, so re-derive display source from the argv join (redaction
|
|
184
|
+
// is applied by the caller on the whole command; the per-action source is the
|
|
185
|
+
// already-tokenized argv, which carries no secrets the command didn't).
|
|
186
|
+
segments.forEach((argv, i) => {
|
|
187
|
+
if (argv.length === 0)
|
|
188
|
+
return;
|
|
189
|
+
const source = argv.join(' ');
|
|
190
|
+
const info = classifyBashCommand(source);
|
|
191
|
+
const prog = effectiveProgram(source);
|
|
192
|
+
const complete = argvComplete(source);
|
|
193
|
+
const categories = [mapBashCategory(info.category, argv)];
|
|
194
|
+
const verdict = classifyActionDanger(argv, complete);
|
|
195
|
+
const action = {
|
|
196
|
+
ordinal: i + 1,
|
|
197
|
+
source,
|
|
198
|
+
argv,
|
|
199
|
+
argvComplete: complete,
|
|
200
|
+
program: prog ?? info.tool,
|
|
201
|
+
categories,
|
|
202
|
+
danger: verdict.danger,
|
|
203
|
+
};
|
|
204
|
+
if (verdict.destructiveOperation)
|
|
205
|
+
action.destructiveOperation = verdict.destructiveOperation;
|
|
206
|
+
actions.push(action);
|
|
207
|
+
});
|
|
208
|
+
return actions;
|
|
209
|
+
}
|
|
210
|
+
function bashExecution(step, useEvent, resultEvent, ctx) {
|
|
211
|
+
const rawCommand = useEvent.command ?? stringArg(useEvent.args, 'command', 'cmd', 'script') ?? '';
|
|
212
|
+
const unwrapped = unwrapShellExec(rawCommand);
|
|
213
|
+
const command = ctx.redact ? redactSecrets(rawCommand, ctx.knownSecrets) : rawCommand;
|
|
214
|
+
const unwrappedCommand = ctx.redact ? redactSecrets(unwrapped, ctx.knownSecrets) : unwrapped;
|
|
215
|
+
const { diagnostics } = extractShellPrograms(unwrapped);
|
|
216
|
+
const actions = buildBashActions(unwrapped);
|
|
217
|
+
// parseStatus: `parsed` when we tokenized ≥1 segment and the parser had no
|
|
218
|
+
// diagnostics; `partial` when we got segments but the parser flagged something;
|
|
219
|
+
// `unparseable` when we recovered no segments at all from a non-empty command.
|
|
220
|
+
let parseStatus;
|
|
221
|
+
if (actions.length === 0)
|
|
222
|
+
parseStatus = rawCommand.trim().length === 0 ? 'parsed' : 'unparseable';
|
|
223
|
+
else if (diagnostics.length > 0)
|
|
224
|
+
parseStatus = 'partial';
|
|
225
|
+
else
|
|
226
|
+
parseStatus = 'parsed';
|
|
227
|
+
return {
|
|
228
|
+
...executionBase(step, 'execution'),
|
|
229
|
+
executionType: 'bash',
|
|
230
|
+
tool: step.tool ?? 'Bash',
|
|
231
|
+
result: resultOf(resultEvent, ctx.redact, ctx.knownSecrets),
|
|
232
|
+
command,
|
|
233
|
+
unwrappedCommand,
|
|
234
|
+
parseStatus,
|
|
235
|
+
parseDiagnostics: diagnostics,
|
|
236
|
+
actions,
|
|
237
|
+
};
|
|
238
|
+
}
|
|
239
|
+
function readExecution(step, useEvent, resultEvent, ctx) {
|
|
240
|
+
const file = stringArg(useEvent.args, 'file_path', 'path', 'notebook_path', 'filePath') ?? useEvent.path ?? '';
|
|
241
|
+
const exec = {
|
|
242
|
+
...executionBase(step, 'execution'),
|
|
243
|
+
executionType: 'read',
|
|
244
|
+
tool: step.tool ?? 'Read',
|
|
245
|
+
result: resultOf(resultEvent, ctx.redact, ctx.knownSecrets),
|
|
246
|
+
file: ctx.redact ? redactSecrets(file, ctx.knownSecrets) : file,
|
|
247
|
+
};
|
|
248
|
+
const offset = numberArg(useEvent.args, 'offset');
|
|
249
|
+
const limit = numberArg(useEvent.args, 'limit');
|
|
250
|
+
if (offset !== undefined)
|
|
251
|
+
exec.offset = offset;
|
|
252
|
+
if (limit !== undefined)
|
|
253
|
+
exec.limit = limit;
|
|
254
|
+
const returnedLines = countLines(resultEvent);
|
|
255
|
+
if (returnedLines)
|
|
256
|
+
exec.returnedLines = returnedLines;
|
|
257
|
+
return exec;
|
|
258
|
+
}
|
|
259
|
+
function grepExecution(step, useEvent, resultEvent, ctx) {
|
|
260
|
+
const query = stringArg(useEvent.args, 'pattern', 'query', 'q') ?? '';
|
|
261
|
+
const grepPath = stringArg(useEvent.args, 'path');
|
|
262
|
+
const glob = stringArg(useEvent.args, 'glob');
|
|
263
|
+
const outputModeRaw = stringArg(useEvent.args, 'output_mode');
|
|
264
|
+
const outputMode = outputModeRaw === 'content' || outputModeRaw === 'files' || outputModeRaw === 'count'
|
|
265
|
+
? (outputModeRaw === 'files' ? 'files' : outputModeRaw)
|
|
266
|
+
: outputModeRaw
|
|
267
|
+
? 'unknown'
|
|
268
|
+
: undefined;
|
|
269
|
+
const exec = {
|
|
270
|
+
...executionBase(step, 'execution'),
|
|
271
|
+
executionType: 'grep',
|
|
272
|
+
tool: step.tool ?? 'Grep',
|
|
273
|
+
result: resultOf(resultEvent, ctx.redact, ctx.knownSecrets),
|
|
274
|
+
query: ctx.redact ? redactSecrets(query, ctx.knownSecrets) : query,
|
|
275
|
+
};
|
|
276
|
+
if (grepPath)
|
|
277
|
+
exec.path = ctx.redact ? redactSecrets(grepPath, ctx.knownSecrets) : grepPath;
|
|
278
|
+
if (glob)
|
|
279
|
+
exec.glob = glob;
|
|
280
|
+
if (outputMode)
|
|
281
|
+
exec.outputMode = outputMode;
|
|
282
|
+
const hits = countLines(resultEvent);
|
|
283
|
+
if (hits)
|
|
284
|
+
exec.hits = hits;
|
|
285
|
+
return exec;
|
|
286
|
+
}
|
|
287
|
+
/**
|
|
288
|
+
* Build a FileMutation from an Edit (`old_string`/`new_string`) or Write
|
|
289
|
+
* (`content`). The single hunk's added/removed line counts come from the string
|
|
290
|
+
* diff; beforeHash/afterHash are content fingerprints so the cross-step revert
|
|
291
|
+
* ledger can match a later edit that restores an earlier one.
|
|
292
|
+
*/
|
|
293
|
+
function fileMutationFromEdit(useEvent) {
|
|
294
|
+
const path = stringArg(useEvent.args, 'file_path', 'path', 'filePath') ?? useEvent.path;
|
|
295
|
+
if (!path)
|
|
296
|
+
return null;
|
|
297
|
+
const oldStr = stringArg(useEvent.args, 'old_string', 'old_str');
|
|
298
|
+
const newStr = stringArg(useEvent.args, 'new_string', 'new_str');
|
|
299
|
+
if (oldStr === undefined && newStr === undefined) {
|
|
300
|
+
// No diff strings — record the mutation with an empty hunk list rather than
|
|
301
|
+
// fabricate line counts.
|
|
302
|
+
return { path, operation: 'update', hunks: [] };
|
|
303
|
+
}
|
|
304
|
+
const before = oldStr ?? '';
|
|
305
|
+
const after = newStr ?? '';
|
|
306
|
+
const removedLines = countStringLines(before);
|
|
307
|
+
const addedLines = countStringLines(after);
|
|
308
|
+
const hunk = {
|
|
309
|
+
id: 'h1',
|
|
310
|
+
addedLines,
|
|
311
|
+
removedLines,
|
|
312
|
+
beforeHash: shortHash(before),
|
|
313
|
+
afterHash: shortHash(after),
|
|
314
|
+
};
|
|
315
|
+
return { path, operation: 'update', hunks: [hunk] };
|
|
316
|
+
}
|
|
317
|
+
function fileMutationFromWrite(useEvent) {
|
|
318
|
+
const path = stringArg(useEvent.args, 'file_path', 'path', 'filePath') ?? useEvent.path;
|
|
319
|
+
if (!path)
|
|
320
|
+
return null;
|
|
321
|
+
const content = stringArg(useEvent.args, 'content', 'contents');
|
|
322
|
+
if (content === undefined) {
|
|
323
|
+
return { path, operation: 'overwrite', hunks: [] };
|
|
324
|
+
}
|
|
325
|
+
const addedLines = countStringLines(content);
|
|
326
|
+
const hunk = {
|
|
327
|
+
id: 'h1',
|
|
328
|
+
addedLines,
|
|
329
|
+
removedLines: 0,
|
|
330
|
+
afterHash: shortHash(content),
|
|
331
|
+
};
|
|
332
|
+
// A Write with no prior-content evidence: `overwrite` when the file may have
|
|
333
|
+
// existed. We cannot tell create vs overwrite from the event, so `overwrite`
|
|
334
|
+
// (the conservative "may have clobbered") — never invent `create`.
|
|
335
|
+
return { path, operation: 'overwrite', hunks: [hunk] };
|
|
336
|
+
}
|
|
337
|
+
function countStringLines(text) {
|
|
338
|
+
if (text.length === 0)
|
|
339
|
+
return 0;
|
|
340
|
+
const lines = text.split('\n');
|
|
341
|
+
return lines.length > 0 && lines[lines.length - 1] === '' ? lines.length - 1 : lines.length;
|
|
342
|
+
}
|
|
343
|
+
function editExecution(step, useEvent, resultEvent, ctx) {
|
|
344
|
+
const mutation = fileMutationFromEdit(useEvent);
|
|
345
|
+
const files = mutation ? [redactMutationPath(mutation, ctx)] : [];
|
|
346
|
+
return {
|
|
347
|
+
...executionBase(step, 'execution'),
|
|
348
|
+
executionType: 'edit',
|
|
349
|
+
tool: step.tool ?? 'Edit',
|
|
350
|
+
result: resultOf(resultEvent, ctx.redact, ctx.knownSecrets),
|
|
351
|
+
files,
|
|
352
|
+
// reverts[] is populated in a cross-step pass over all mutations (see below).
|
|
353
|
+
reverts: [],
|
|
354
|
+
};
|
|
355
|
+
}
|
|
356
|
+
function writeExecution(step, useEvent, resultEvent, ctx) {
|
|
357
|
+
const mutation = fileMutationFromWrite(useEvent);
|
|
358
|
+
const files = mutation ? [redactMutationPath(mutation, ctx)] : [];
|
|
359
|
+
return {
|
|
360
|
+
...executionBase(step, 'execution'),
|
|
361
|
+
executionType: 'write',
|
|
362
|
+
tool: step.tool ?? 'Write',
|
|
363
|
+
result: resultOf(resultEvent, ctx.redact, ctx.knownSecrets),
|
|
364
|
+
files,
|
|
365
|
+
reverts: [],
|
|
366
|
+
};
|
|
367
|
+
}
|
|
368
|
+
function redactMutationPath(m, ctx) {
|
|
369
|
+
if (!ctx.redact)
|
|
370
|
+
return m;
|
|
371
|
+
return { ...m, path: redactSecrets(m.path, ctx.knownSecrets) };
|
|
372
|
+
}
|
|
373
|
+
function genericExecution(step, useEvent, resultEvent, ctx) {
|
|
374
|
+
const inputText = stringArg(useEvent.args, 'command', 'cmd', 'file_path', 'path', 'query', 'pattern', 'url', 'description', 'prompt') ?? (useEvent.args ? JSON.stringify(useEvent.args).slice(0, PREVIEW_MAX) : undefined);
|
|
375
|
+
const exec = {
|
|
376
|
+
...executionBase(step, 'execution'),
|
|
377
|
+
executionType: 'generic',
|
|
378
|
+
tool: step.tool ?? 'unknown',
|
|
379
|
+
result: resultOf(resultEvent, ctx.redact, ctx.knownSecrets),
|
|
380
|
+
};
|
|
381
|
+
const input = textPreview(inputText, ctx.redact, ctx.knownSecrets);
|
|
382
|
+
if (input)
|
|
383
|
+
exec.input = input;
|
|
384
|
+
return exec;
|
|
385
|
+
}
|
|
386
|
+
/**
|
|
387
|
+
* A hook firing → HookExecution. parse.ts emits `type: 'hook'` with `hookName`,
|
|
388
|
+
* `hookEvent`, and `success` (boolean). It does NOT preserve the raw
|
|
389
|
+
* blocked-vs-error distinction (both land as `success: false`), so `decision` is
|
|
390
|
+
* conservative: `allowed` on success, else `unknown` — never a confident `blocked`
|
|
391
|
+
* we cannot prove. `phase` is derived from the lifecycle event name.
|
|
392
|
+
*
|
|
393
|
+
* Hooks are NOT tool_use events, so `pairSteps` never draws them; the build draws
|
|
394
|
+
* them separately from the raw event stream and merges by startMs (see below).
|
|
395
|
+
*
|
|
396
|
+
* TODO(PHNX-3442): (1) if parse.ts is extended to preserve the raw `hook_blocked`
|
|
397
|
+
* vs `hook_error` attachment type, map those to `blocked`/`error` here instead of
|
|
398
|
+
* the conservative `unknown`. (2) parse.ts emits NO permission events today (its
|
|
399
|
+
* `permission-mode` lines are skipped, parse.ts:569), so `PermissionExecution` is
|
|
400
|
+
* never produced — wire it once permission events are parsed.
|
|
401
|
+
*/
|
|
402
|
+
function hookExecution(event, ordinal, startMs) {
|
|
403
|
+
const hookEvent = event.hookEvent;
|
|
404
|
+
const phase = hookEvent === undefined
|
|
405
|
+
? 'other'
|
|
406
|
+
: /^Pre/i.test(hookEvent)
|
|
407
|
+
? 'pre'
|
|
408
|
+
: /^Post/i.test(hookEvent)
|
|
409
|
+
? 'post'
|
|
410
|
+
: /Session/i.test(hookEvent)
|
|
411
|
+
? 'session'
|
|
412
|
+
: 'other';
|
|
413
|
+
const decision = event.success === true ? 'allowed' : 'unknown';
|
|
414
|
+
const exec = {
|
|
415
|
+
kind: 'execution',
|
|
416
|
+
lane: 'hook',
|
|
417
|
+
ordinal,
|
|
418
|
+
startMs,
|
|
419
|
+
durationMs: 0,
|
|
420
|
+
durationEstimated: true,
|
|
421
|
+
outcome: event.success === true ? 'ok' : 'unknown',
|
|
422
|
+
label: event.hookName ?? 'hook',
|
|
423
|
+
executionType: 'hook',
|
|
424
|
+
phase,
|
|
425
|
+
decision,
|
|
426
|
+
result: {},
|
|
427
|
+
};
|
|
428
|
+
if (event.hookName)
|
|
429
|
+
exec.hookName = event.hookName;
|
|
430
|
+
if (hookEvent)
|
|
431
|
+
exec.hookEvent = hookEvent;
|
|
432
|
+
return exec;
|
|
433
|
+
}
|
|
434
|
+
/** Shared base fields for any execution step, mapped from the drawn trajectory step. */
|
|
435
|
+
function executionBase(step, kind) {
|
|
436
|
+
const base = {
|
|
437
|
+
kind,
|
|
438
|
+
ordinal: step.ordinal,
|
|
439
|
+
startMs: step.startMs,
|
|
440
|
+
durationMs: step.durationMs,
|
|
441
|
+
durationEstimated: step.durationEstimated,
|
|
442
|
+
outcome: stepOutcome(step),
|
|
443
|
+
label: step.label,
|
|
444
|
+
lane: step.lane,
|
|
445
|
+
};
|
|
446
|
+
const withCall = step.callId ? { ...base, callId: step.callId } : base;
|
|
447
|
+
return withCall;
|
|
448
|
+
}
|
|
449
|
+
function thinkingStep(step) {
|
|
450
|
+
const outcome = step.outcome === 'error' ? 'unknown' : (step.outcome === 'ok' ? 'ok' : 'unknown');
|
|
451
|
+
return {
|
|
452
|
+
kind: 'thinking',
|
|
453
|
+
lane: 'think',
|
|
454
|
+
ordinal: step.ordinal,
|
|
455
|
+
startMs: step.startMs,
|
|
456
|
+
durationMs: step.durationMs,
|
|
457
|
+
durationEstimated: step.durationEstimated,
|
|
458
|
+
outcome,
|
|
459
|
+
label: step.label,
|
|
460
|
+
};
|
|
461
|
+
}
|
|
462
|
+
// ---------------------------------------------------------------------------
|
|
463
|
+
// Cross-step revert ledger
|
|
464
|
+
// ---------------------------------------------------------------------------
|
|
465
|
+
/**
|
|
466
|
+
* Detect when a later edit/write REVERTS an earlier one on the same path+hunk, by
|
|
467
|
+
* content fingerprint: a hunk B reverts hunk A when they touch the same path and
|
|
468
|
+
* B's afterHash equals A's beforeHash AND B's beforeHash equals A's afterHash — i.e.
|
|
469
|
+
* B put the content back exactly the way A found it. Stamps `revertedByStep` on the
|
|
470
|
+
* reverted hunk + mutation, and appends a RevertLink to the reverting step's
|
|
471
|
+
* `reverts[]`.
|
|
472
|
+
*
|
|
473
|
+
* Conservative: only an EXACT hash round-trip counts. A partial/overlapping change
|
|
474
|
+
* is left un-linked (empty reverts[]) rather than guessed — the spec's "do not fake
|
|
475
|
+
* reverts" bar. This walks the already-built mutation steps in order; a mutation
|
|
476
|
+
* with an empty hunk list (no diff strings were available) never participates.
|
|
477
|
+
*/
|
|
478
|
+
function applyRevertLedger(steps) {
|
|
479
|
+
// Earlier hunks, most-recent-first per (path), so a revert matches the latest
|
|
480
|
+
// un-reverted change to that path.
|
|
481
|
+
const earlier = [];
|
|
482
|
+
for (const step of steps) {
|
|
483
|
+
if (step.kind !== 'execution')
|
|
484
|
+
continue;
|
|
485
|
+
if (step.executionType !== 'edit' && step.executionType !== 'write')
|
|
486
|
+
continue;
|
|
487
|
+
const mut = step;
|
|
488
|
+
for (const file of mut.files) {
|
|
489
|
+
for (const hunk of file.hunks) {
|
|
490
|
+
// Does this hunk revert any earlier un-reverted hunk on the same path?
|
|
491
|
+
if (hunk.beforeHash && hunk.afterHash) {
|
|
492
|
+
const match = earlier.find((e) => e.path === file.path &&
|
|
493
|
+
e.hunk.revertedByStep === undefined &&
|
|
494
|
+
e.hunk.afterHash !== undefined &&
|
|
495
|
+
e.hunk.beforeHash !== undefined &&
|
|
496
|
+
e.hunk.afterHash === hunk.beforeHash &&
|
|
497
|
+
e.hunk.beforeHash === hunk.afterHash);
|
|
498
|
+
if (match) {
|
|
499
|
+
match.hunk.revertedByStep = mut.ordinal;
|
|
500
|
+
// Stamp the mutation too when all its hunks are now reverted.
|
|
501
|
+
const parentFile = match.step.files.find((f) => f.path === match.path);
|
|
502
|
+
if (parentFile && parentFile.hunks.every((h) => h.revertedByStep !== undefined)) {
|
|
503
|
+
parentFile.revertedByStep = mut.ordinal;
|
|
504
|
+
}
|
|
505
|
+
mut.reverts.push({
|
|
506
|
+
revertedStep: match.ordinal,
|
|
507
|
+
path: file.path,
|
|
508
|
+
revertedHunkIds: [match.hunk.id],
|
|
509
|
+
});
|
|
510
|
+
}
|
|
511
|
+
}
|
|
512
|
+
earlier.push({ step: mut, ordinal: mut.ordinal, path: file.path, hunk });
|
|
513
|
+
}
|
|
514
|
+
}
|
|
515
|
+
}
|
|
516
|
+
}
|
|
517
|
+
// ---------------------------------------------------------------------------
|
|
518
|
+
// buildSessionDetailV2
|
|
519
|
+
// ---------------------------------------------------------------------------
|
|
520
|
+
const TOOL_KINDS = {
|
|
521
|
+
bash: new Set(['bash', 'exec', 'execute', 'exec_command', 'run_command', 'run_shell_command', 'shell']),
|
|
522
|
+
read: new Set(['read', 'view', 'cat_file']),
|
|
523
|
+
grep: new Set(['grep', 'search', 'codebase_search', 'grep_search']),
|
|
524
|
+
edit: new Set(['edit', 'multiedit', 'str_replace', 'apply_patch', 'replace_file_content']),
|
|
525
|
+
write: new Set(['write', 'create_file', 'write_file']),
|
|
526
|
+
};
|
|
527
|
+
function toolExecutionKind(tool) {
|
|
528
|
+
const t = tool.toLowerCase();
|
|
529
|
+
for (const [kind, names] of Object.entries(TOOL_KINDS)) {
|
|
530
|
+
if (names.has(t))
|
|
531
|
+
return kind;
|
|
532
|
+
}
|
|
533
|
+
return 'generic';
|
|
534
|
+
}
|
|
535
|
+
/**
|
|
536
|
+
* Build the schema-2 per-session detail from a pre-built trajectory and its raw
|
|
537
|
+
* events. The trajectory supplies meta/gaps/whereItWentWrong/surfacedToolFailures
|
|
538
|
+
* (via the shared sync.ts helpers) and the truncation count; the raw events supply
|
|
539
|
+
* the per-tool detail the schema-1 flat step could not carry.
|
|
540
|
+
*
|
|
541
|
+
* Re-pairs the events with `pairSteps` (the SAME loop buildTrajectory ran) to
|
|
542
|
+
* recover each step's (use event, result event) triple, then dispatches per tool.
|
|
543
|
+
*/
|
|
544
|
+
export function buildSessionDetailV2(traj, events, options = {}) {
|
|
545
|
+
const redact = options.redact !== false;
|
|
546
|
+
const knownSecrets = options.knownSecrets;
|
|
547
|
+
const ctx = { redact, knownSecrets };
|
|
548
|
+
const stats = computeSummaryStats(events);
|
|
549
|
+
const firstTs = stats.firstTs;
|
|
550
|
+
const eventMs = eventTimestampsMs(events);
|
|
551
|
+
const drafts = pairSteps(events, eventMs, firstTs, redact, knownSecrets);
|
|
552
|
+
// Apply the SAME cap the trajectory used, so the two step lists line up and the
|
|
553
|
+
// truncation count is honest. `traj.steps` is already capped + ordinal-numbered.
|
|
554
|
+
const cappedDrafts = drafts.slice(0, traj.steps.length);
|
|
555
|
+
const steps = [];
|
|
556
|
+
for (let i = 0; i < cappedDrafts.length; i++) {
|
|
557
|
+
const draft = cappedDrafts[i];
|
|
558
|
+
// The trajectory step is the AUTHORITATIVE one: buildTrajectory resolved its
|
|
559
|
+
// outcome, duration, durationEstimated, and exitCode after pairing. The fresh
|
|
560
|
+
// draft from this re-pair only supplies the event indices (use/result); its
|
|
561
|
+
// own `step` still carries the pre-resolution placeholders. So read base fields
|
|
562
|
+
// from `traj.steps[i]` and use the draft solely for the event triple.
|
|
563
|
+
const step = traj.steps[i];
|
|
564
|
+
// Guard the positional pairing. buildTrajectory and this builder both derive
|
|
565
|
+
// their drafts from the SAME events via the SAME deterministic `pairSteps`, so
|
|
566
|
+
// draft[i] must describe the same step as traj.steps[i]. If a future change
|
|
567
|
+
// makes the two call sites diverge (e.g. one filters events, the other does
|
|
568
|
+
// not), fail loud here rather than silently emit a shard whose danger flags,
|
|
569
|
+
// timestamps, and paths belong to the wrong step.
|
|
570
|
+
if (draft.step.kind !== step.kind || draft.step.callId !== step.callId) {
|
|
571
|
+
throw new Error(`schema-2 step/draft misalignment at ordinal ${step.ordinal}: ` +
|
|
572
|
+
`draft(kind=${draft.step.kind},callId=${draft.step.callId ?? '∅'}) vs ` +
|
|
573
|
+
`step(kind=${step.kind},callId=${step.callId ?? '∅'})`);
|
|
574
|
+
}
|
|
575
|
+
if (step.kind === 'thinking') {
|
|
576
|
+
steps.push(thinkingStep(step));
|
|
577
|
+
continue;
|
|
578
|
+
}
|
|
579
|
+
const useEvent = events[draft.eventIndex];
|
|
580
|
+
const resultEvent = draft.resultEventIndex !== undefined ? events[draft.resultEventIndex] : undefined;
|
|
581
|
+
const tool = step.tool ?? 'unknown';
|
|
582
|
+
const kind = toolExecutionKind(tool);
|
|
583
|
+
switch (kind) {
|
|
584
|
+
case 'bash':
|
|
585
|
+
steps.push(bashExecution(step, useEvent, resultEvent, ctx));
|
|
586
|
+
break;
|
|
587
|
+
case 'read':
|
|
588
|
+
steps.push(readExecution(step, useEvent, resultEvent, ctx));
|
|
589
|
+
break;
|
|
590
|
+
case 'grep':
|
|
591
|
+
steps.push(grepExecution(step, useEvent, resultEvent, ctx));
|
|
592
|
+
break;
|
|
593
|
+
case 'edit':
|
|
594
|
+
steps.push(editExecution(step, useEvent, resultEvent, ctx));
|
|
595
|
+
break;
|
|
596
|
+
case 'write':
|
|
597
|
+
steps.push(writeExecution(step, useEvent, resultEvent, ctx));
|
|
598
|
+
break;
|
|
599
|
+
default:
|
|
600
|
+
steps.push(genericExecution(step, useEvent, resultEvent, ctx));
|
|
601
|
+
break;
|
|
602
|
+
}
|
|
603
|
+
}
|
|
604
|
+
// Draw hook firings (parse.ts `type:'hook'`) as first-class HookExecution steps.
|
|
605
|
+
// They are not tool_use events, so pairSteps never drew them; merge them into the
|
|
606
|
+
// step stream by startMs and re-number ordinals so the ordering stays truthful.
|
|
607
|
+
// A session with no hook events leaves `steps` and its ordinals untouched.
|
|
608
|
+
const hookSteps = [];
|
|
609
|
+
for (let i = 0; i < events.length; i++) {
|
|
610
|
+
const e = events[i];
|
|
611
|
+
if (e.type !== 'hook')
|
|
612
|
+
continue;
|
|
613
|
+
const startMs = Number.isNaN(eventMs[i]) ? 0 : Math.max(0, eventMs[i] - firstTs);
|
|
614
|
+
hookSteps.push(hookExecution(e, 0, startMs));
|
|
615
|
+
}
|
|
616
|
+
let merged = steps;
|
|
617
|
+
if (hookSteps.length > 0) {
|
|
618
|
+
merged = [...steps, ...hookSteps].sort((a, b) => a.startMs - b.startMs);
|
|
619
|
+
merged.forEach((step, idx) => { step.ordinal = idx + 1; });
|
|
620
|
+
}
|
|
621
|
+
applyRevertLedger(merged);
|
|
622
|
+
const s = traj.session;
|
|
623
|
+
return {
|
|
624
|
+
schema: 2,
|
|
625
|
+
id: s.id,
|
|
626
|
+
meta: buildDetailMeta(traj),
|
|
627
|
+
steps: merged,
|
|
628
|
+
gaps: traj.gaps,
|
|
629
|
+
truncatedSteps: traj.truncatedSteps,
|
|
630
|
+
whereItWentWrong: buildWhereItWentWrong(traj),
|
|
631
|
+
surfacedToolFailures: traj.steps
|
|
632
|
+
.filter((step) => step.outcome === 'error')
|
|
633
|
+
.map((step) => ({ tool: step.tool, label: step.label, detail: step.detail })),
|
|
634
|
+
};
|
|
635
|
+
}
|
|
636
|
+
// re-export for callers/tests
|
|
637
|
+
export { activeMsFromTrajectory };
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Danger classifier for a single shell action's argv (PHNX-3442, producer side).
|
|
3
|
+
*
|
|
4
|
+
* Safety-sensitive: the schema-2 `BashAction.danger` drives the console's
|
|
5
|
+
* destructive-operation surfacing and risk scoring. So this is CONSERVATIVE by
|
|
6
|
+
* construction — it defaults to `normal` and only escalates on CLEAR structural
|
|
7
|
+
* evidence in the tokenized argv, never on a substring of raw command text. The
|
|
8
|
+
* argv it reads is one tokenizeBash segment (see `tokenizeBash` in
|
|
9
|
+
* `session/bash-command.ts`): the executable at argv[0] and its already-split
|
|
10
|
+
* arguments, so a flag like `-rf` is a whole token, not a substring hunt.
|
|
11
|
+
*
|
|
12
|
+
* The three levels mirror the shipped consumer union (`BashDanger`):
|
|
13
|
+
* - DESTRUCTIVE — irrecoverable data loss / history rewrite / force
|
|
14
|
+
* overwrite of an important path. Requires a WHERE-less
|
|
15
|
+
* DELETE, a recursive/force delete, a hard reset, etc.
|
|
16
|
+
* - potentially-destructive — plain `rm`, soft/mixed `git reset`, `mv` over a
|
|
17
|
+
* path, plain `kill` — recoverable-ish but worth a flag.
|
|
18
|
+
* - normal — everything else.
|
|
19
|
+
*
|
|
20
|
+
* `destructiveOperation` is a short stable label (never raw text) naming WHY the
|
|
21
|
+
* action was flagged, so the console can group by operation without re-parsing.
|
|
22
|
+
*/
|
|
23
|
+
import type { BashDanger } from './schema2.js';
|
|
24
|
+
export interface DangerVerdict {
|
|
25
|
+
danger: BashDanger;
|
|
26
|
+
/** Short stable label naming the operation, e.g. `recursive-delete`. Omitted for `normal`. */
|
|
27
|
+
destructiveOperation?: string;
|
|
28
|
+
}
|
|
29
|
+
/**
|
|
30
|
+
* Classify one tokenized shell action. `argvComplete` is false when a dynamic node
|
|
31
|
+
* (command substitution / glob / var expansion) kept the argv incomplete; when a
|
|
32
|
+
* DESTRUCTIVE signal depends on a token that could have been mangled by expansion
|
|
33
|
+
* we DO still flag it (a `rm -rf $DIR` is destructive regardless of what `$DIR`
|
|
34
|
+
* expands to), because the operation itself is the danger, not its target.
|
|
35
|
+
*/
|
|
36
|
+
export declare function classifyActionDanger(argv: string[], _argvComplete?: boolean): DangerVerdict;
|