@bivy/bivy 0.9.1-staging.136 → 0.10.0-staging.138

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/bin/acp-shim.mjs CHANGED
@@ -125,6 +125,74 @@ let pendingModel = null;
125
125
  // right ACP permission request with a concrete optionId.
126
126
  const permissionRequests = new Map();
127
127
 
128
+ // --- tool-call field normalization -------------------------------------------
129
+ // ACP's `tool_call`/`tool_call_update` carries a free-text `title` (whatever
130
+ // prose the agent chose) AND a small fixed `kind` enum (read/edit/delete/move/
131
+ // search/execute/think/fetch/other) that matches the node's tool taxonomy
132
+ // (src/runtime/tool-call-map.ts) far better than prose does. It also often
133
+ // splits the substantive data across three places — `rawInput` (frequently
134
+ // empty on the *first* tool_call notification for some agents, opencode
135
+ // included), `locations` (paths the call touches), and `content` (diff/text
136
+ // blocks, usually only populated by a later tool_call_update) — so a naive
137
+ // single-notification read sees "no real information". Accumulate everything
138
+ // we've learned about a call across its whole lifecycle, keyed by toolCallId.
139
+ const toolCallState = new Map();
140
+
141
+ // The subset of ACP kinds that line up 1:1 with a bucket tool-call-map.ts
142
+ // already recognizes by name; kinds outside this set (delete/move/think/other)
143
+ // have no equivalent normalized rendering yet, so fall back to the agent's own
144
+ // title/kind for display rather than inventing a bucket for them.
145
+ const KIND_TOOL_NAME = { read: "read", edit: "edit", execute: "execute", search: "search", fetch: "fetch" };
146
+
147
+ function mergeToolCallState(toolCallId, u) {
148
+ const prev = toolCallState.get(toolCallId) || {};
149
+ // `content` is normally a ContentBlock[], but some agents send a single block
150
+ // object (the existing `textOf` helper already tolerates both shapes) — wrap
151
+ // it so downstream array-walkers (diffContentFields) see it either way.
152
+ const content = u.content == null ? undefined : Array.isArray(u.content) ? u.content : [u.content];
153
+ const next = {
154
+ kind: u.kind ?? prev.kind,
155
+ title: u.title ?? prev.title,
156
+ rawInput: u.rawInput && typeof u.rawInput === "object" && Object.keys(u.rawInput).length ? u.rawInput : prev.rawInput,
157
+ locations: Array.isArray(u.locations) && u.locations.length ? u.locations : prev.locations,
158
+ content: content && content.length ? content : prev.content,
159
+ };
160
+ toolCallState.set(toolCallId, next);
161
+ return next;
162
+ }
163
+
164
+ /** The ACP "diff" content block, if the call carries one — the shape opencode
165
+ * (and most ACP agents) use to report an edit's before/after text. */
166
+ function diffContentFields(content) {
167
+ for (const c of content || []) {
168
+ if (c && c.type === "diff") return { path: c.path, oldText: c.oldText, newText: c.newText };
169
+ }
170
+ return {};
171
+ }
172
+
173
+ /** Merge everything accumulated about a call into one `input` object shaped the
174
+ * way tool-call-map.ts's key scan expects (path/command/old_string/new_string/…),
175
+ * so a call whose `rawInput` was sparse still classifies once its diff/locations
176
+ * arrive. `rawInput` (the underlying tool's own arguments) wins on key conflicts
177
+ * since it's the most literal source. */
178
+ function toolCallInput(state) {
179
+ const input = { ...(state.rawInput || {}) };
180
+ const diff = diffContentFields(state.content);
181
+ if (input.path == null && diff.path != null) input.path = diff.path;
182
+ if (input.old_string == null && diff.oldText != null) input.old_string = diff.oldText;
183
+ if (input.new_string == null && diff.newText != null) input.new_string = diff.newText;
184
+ if (input.path == null && state.locations?.[0]?.path != null) input.path = state.locations[0].path;
185
+ return input;
186
+ }
187
+
188
+ /** Prefer ACP's structured `kind` (maps straight onto the node's taxonomy) over
189
+ * the agent's free-text `title` — a title like "Edit `src/index.ts`" defeats
190
+ * both the node's bucket classifier and the client's own name-based heuristic,
191
+ * which both expect short tool-name-like tokens, not prose. */
192
+ function toolCallName(state) {
193
+ return (state.kind && KIND_TOOL_NAME[state.kind]) || state.title || state.kind || "tool";
194
+ }
195
+
128
196
  async function ensureInitialized() {
129
197
  if (initialized) return;
130
198
  // Bounded: a binary that accepts the launch args but never speaks ACP would
@@ -191,14 +259,27 @@ function onSessionUpdate(params) {
191
259
  // An auto-run tool (no permission requested) — surface it so the transcript
192
260
  // shows the action; the result arrives via tool_call_update.
193
261
  const toolCallId = String(u.toolCallId ?? u.id ?? "");
194
- bivy({ type: "tool.call", toolCallId, name: String(u.title || u.kind || "tool"), input: u.rawInput ?? u.input ?? {} });
262
+ const state = mergeToolCallState(toolCallId, u);
263
+ bivy({ type: "tool.call", toolCallId, name: toolCallName(state), input: toolCallInput(state) });
195
264
  break;
196
265
  }
197
266
  case "tool_call_update": {
198
267
  const toolCallId = String(u.toolCallId ?? u.id ?? "");
268
+ const state = mergeToolCallState(toolCallId, u);
199
269
  const status = String(u.status || "");
200
270
  if (status === "completed" || status === "failed") {
201
- bivy({ type: "tool.result", toolCallId, name: String(u.title || "tool"), result: textOf(u.content) || status });
271
+ toolCallState.delete(toolCallId);
272
+ bivy({
273
+ type: "tool.result",
274
+ toolCallId,
275
+ name: toolCallName(state),
276
+ result: textOf(u.content) || status,
277
+ isError: status === "failed",
278
+ });
279
+ } else {
280
+ // Still running: forward the fuller name/input as it fills in so a live
281
+ // tool card isn't stuck with the sparse initial notification.
282
+ bivy({ type: "tool.update", toolCallId, name: toolCallName(state), input: toolCallInput(state) });
202
283
  }
203
284
  break;
204
285
  }
@@ -222,7 +303,8 @@ async function onAgentRequest(id, method, params) {
222
303
  const toolCallId = String(tc.toolCallId ?? tc.id ?? `perm-${id}`);
223
304
  const options = Array.isArray(params?.options) ? params.options : [];
224
305
  permissionRequests.set(toolCallId, { requestId: id, options });
225
- bivy({ type: "tool.call", toolCallId, name: String(tc.title || tc.kind || "tool"), input: tc.rawInput ?? tc.input ?? {} });
306
+ const state = mergeToolCallState(toolCallId, tc);
307
+ bivy({ type: "tool.call", toolCallId, name: toolCallName(state), input: toolCallInput(state) });
226
308
  return;
227
309
  }
228
310
  case "fs/read_text_file": {
@@ -77,10 +77,25 @@ class TurnAccumulator {
77
77
  ended = false;
78
78
  reasoning = "";
79
79
  usageSnapshot;
80
- toolUses = [];
81
80
  toolResults = [];
81
+ // Ordered content blocks (text + tool_use, interleaved exactly as they
82
+ // streamed) for the assistant turn. A prior version tracked `text` and tool
83
+ // uses separately and always emitted the whole turn's text ahead of every
84
+ // tool call on finish() — collapsing a turn like "Let me check." → tool →
85
+ // "Now editing." → tool into one merged text block followed by both tools.
86
+ // That read as interim messages "disappearing"/bundling at the end once
87
+ // history reconciled against it. `textFlushed` is the prefix of `text`
88
+ // already sealed into `content` as its own block.
89
+ content = [];
90
+ textFlushed = "";
82
91
  out = [];
83
92
  details = new Map();
93
+ flushPendingText() {
94
+ const pending = this.text.slice(this.textFlushed.length);
95
+ this.textFlushed = this.text;
96
+ if (pending)
97
+ this.content.push({ type: "text", text: pending });
98
+ }
84
99
  constructor(toolContext) {
85
100
  this.toolContext = toolContext;
86
101
  }
@@ -123,7 +138,8 @@ class TurnAccumulator {
123
138
  const detail = mapToolCall(name, input, this.toolContext);
124
139
  if (detail && id)
125
140
  this.details.set(id, detail);
126
- this.toolUses.push({ type: "tool_use", id, name, input: input ?? {}, ...(detail ? { detail } : {}) });
141
+ this.flushPendingText();
142
+ this.content.push({ type: "tool_use", id, name, input: input ?? {}, ...(detail ? { detail } : {}) });
127
143
  events.push({ type: "tool_call", toolName: name, input, toolCallId: id, ...(detail ? { detail } : {}) });
128
144
  }
129
145
  addToolResult(toolUseId, name, content, events, isError = false) {
@@ -140,14 +156,16 @@ class TurnAccumulator {
140
156
  if (this.ended)
141
157
  return;
142
158
  this.ended = true;
159
+ // Whether this turn ever used a tool — checked BEFORE flushing trailing
160
+ // text, so a tool-free turn (content still empty at this point) keeps the
161
+ // plain-text message shape it always had instead of gaining a pointless
162
+ // single-text-block wrapper.
163
+ const hadTools = this.content.length > 0 || this.toolResults.length > 0;
143
164
  const message = { role: "assistant", content: this.text };
144
- if (this.toolUses.length || this.toolResults.length) {
145
- const content = [];
146
- if (this.text)
147
- content.push({ type: "text", text: this.text });
148
- content.push(...this.toolUses);
149
- if (content.length)
150
- this.out.push({ role: "assistant", content });
165
+ if (hadTools) {
166
+ this.flushPendingText();
167
+ if (this.content.length)
168
+ this.out.push({ role: "assistant", content: this.content });
151
169
  if (this.toolResults.length)
152
170
  this.out.push({ role: "user", content: this.toolResults });
153
171
  }
@@ -8,6 +8,7 @@ import { bivySessionEnv } from "./session-env.js";
8
8
  import { mergeAgentCommands } from "./slash-commands.js";
9
9
  import { withExactCapabilitySurface } from "./types.js";
10
10
  import { extractTokenUsage } from "./cli-parsers.js";
11
+ import { mapToolCall, mapToolResult } from "./tool-call-map.js";
11
12
  /** A protocol `usage` message → UsageSnapshot (reuses the CLI token-key scan). */
12
13
  function parseProtocolUsage(raw) {
13
14
  if (!raw || typeof raw !== "object")
@@ -170,11 +171,23 @@ class ProtocolSession {
170
171
  reasoningText = "";
171
172
  stderrOutput = "";
172
173
  lastUsage;
173
- // Accumulate the current turn's tool calls/results so getMessages() keeps them
174
- // in history — re-opening a session then shows what the agent actually did, not
175
- // just its final text. Cleared at the start/end of each turn.
176
- turnToolUses = [];
174
+ // Accumulate the current turn's content blocks (text + tool calls, in the order
175
+ // they actually happened) and tool results so getMessages() keeps them in
176
+ // history — re-opening a session then shows what the agent actually did,
177
+ // interleaved exactly as it happened, not just its final text. Cleared at the
178
+ // start/end of each turn. `turnTextFlushed` is the prefix of `assistantText`
179
+ // already sealed into `turnContent` as a text block — each tool call flushes
180
+ // the text since the last flush before appending its own block, so a turn like
181
+ // "Let me check." → tool → "Now editing." → tool persists as
182
+ // [text, tool_use, text, tool_use] instead of collapsing into one text block
183
+ // followed by every tool (which is what re-flattened on reconcile and read as
184
+ // interim messages "disappearing"/bundling at the end of the turn).
185
+ turnContent = [];
186
+ turnTextFlushed = "";
177
187
  turnToolResults = [];
188
+ // toolCallId -> the node's normalized classification, so a later tool.result
189
+ // (or tool.update) can attach/refresh `detail` on the already-pushed block.
190
+ toolDetailsByCallId = new Map();
178
191
  constructor(runtimeOptions, cwd, capabilitiesRef, toolInterceptor,
179
192
  // When resuming, the agent's own session ref (e.g. a Codex thread id). Passed
180
193
  // back to the shim via the generic session.resume primitive so it reconnects
@@ -389,6 +402,15 @@ class ProtocolSession {
389
402
  this.handleMessage(msg);
390
403
  }
391
404
  }
405
+ /** Seal the assistant text streamed since the last flush as a text block in
406
+ * `turnContent`, ahead of a tool call (or at turn end) — see turnContent's
407
+ * doc comment for why this preserves interleaving on reconcile. */
408
+ flushPendingTurnText() {
409
+ const pending = this.assistantText.slice(this.turnTextFlushed.length);
410
+ this.turnTextFlushed = this.assistantText;
411
+ if (pending)
412
+ this.turnContent.push({ type: "text", text: pending });
413
+ }
392
414
  handleMessage(msg) {
393
415
  this.emitter.emit("protocol-message", msg);
394
416
  const replyTo = typeof msg.replyTo === "string" ? msg.replyTo : "";
@@ -458,19 +480,22 @@ class ProtocolSession {
458
480
  return;
459
481
  }
460
482
  if (type === "session.done") {
483
+ // Whether this turn ever used a tool — checked BEFORE flushing trailing
484
+ // text, so a tool-free turn (turnContent still empty at this point) keeps
485
+ // the plain-text message shape it always had instead of gaining a
486
+ // pointless single-text-block wrapper.
487
+ const hadTools = this.turnContent.length > 0 || this.turnToolResults.length > 0;
461
488
  const message = { role: "assistant", content: this.assistantText };
462
- // Persist the assistant turn. When the turn used tools, store content blocks
463
- // (text + tool_use) plus a trailing user message carrying the tool_result
464
- // blocks, matched by tool_use_id — the same shape the PWA renders from live
465
- // streaming, so a re-opened transcript looks identical to what was on screen.
466
- // A tool-free turn keeps the plain-text form it always used.
467
- if (this.turnToolUses.length || this.turnToolResults.length) {
468
- const assistantContent = [];
469
- if (this.assistantText)
470
- assistantContent.push({ type: "text", text: this.assistantText });
471
- assistantContent.push(...this.turnToolUses);
472
- if (assistantContent.length)
473
- this.messages.push({ role: "assistant", content: assistantContent, timestamp: Date.now() });
489
+ // Persist the assistant turn. When the turn used tools, store the ordered
490
+ // content blocks (text/tool_use interleaved exactly as they streamed — see
491
+ // turnContent's doc comment) plus a trailing user message carrying the
492
+ // tool_result blocks, matched by tool_use_id — the same shape the PWA
493
+ // renders from live streaming, so a re-opened transcript looks identical to
494
+ // what was on screen. A tool-free turn keeps the plain-text form it always used.
495
+ if (hadTools) {
496
+ this.flushPendingTurnText();
497
+ if (this.turnContent.length)
498
+ this.messages.push({ role: "assistant", content: this.turnContent, timestamp: Date.now() });
474
499
  if (this.turnToolResults.length)
475
500
  this.messages.push({ role: "user", content: this.turnToolResults, timestamp: Date.now() });
476
501
  }
@@ -481,32 +506,38 @@ class ProtocolSession {
481
506
  this.streaming = false;
482
507
  this.assistantText = "";
483
508
  this.reasoningText = "";
484
- this.turnToolUses = [];
509
+ this.turnContent = [];
510
+ this.turnTextFlushed = "";
485
511
  this.turnToolResults = [];
512
+ this.toolDetailsByCallId.clear();
486
513
  this.emit({ type: "agent_end" });
487
514
  return;
488
515
  }
489
516
  if (type === "session.error") {
490
517
  this.streaming = false;
491
518
  this.reasoningText = "";
492
- this.turnToolUses = [];
519
+ this.turnContent = [];
520
+ this.turnTextFlushed = "";
493
521
  this.turnToolResults = [];
522
+ this.toolDetailsByCallId.clear();
494
523
  this.emit({ type: "session.error", error: String(msg.error || "Protocol agent error") });
495
524
  this.emit({ type: "agent_end" });
496
525
  return;
497
526
  }
498
527
  if (type === "tool.call") {
499
- this.turnToolUses.push({
500
- type: "tool_use",
501
- id: String(msg.toolCallId || msg.id || ""),
502
- name: String(msg.name || "tool"),
503
- input: msg.input ?? {},
504
- });
528
+ const toolCallId = String(msg.toolCallId || msg.id || "");
529
+ const toolName = String(msg.name || "tool");
530
+ const detail = mapToolCall(toolName, msg.input, { provider: this.runtimeOptions.id || "acp", protocol: "protocol" });
531
+ if (detail)
532
+ this.toolDetailsByCallId.set(toolCallId, detail);
533
+ this.flushPendingTurnText();
534
+ this.turnContent.push({ type: "tool_use", id: toolCallId, name: toolName, input: msg.input ?? {}, ...(detail ? { detail } : {}) });
505
535
  }
506
536
  if (type === "tool.call" && this.capabilitiesRef.toolInterception && this.toolInterceptor) {
507
537
  const toolCallId = String(msg.toolCallId || "");
508
538
  const toolName = String(msg.name || "tool");
509
- this.emit({ type: "tool_call", toolName, input: msg.input, toolCallId });
539
+ const detail = this.toolDetailsByCallId.get(toolCallId);
540
+ this.emit({ type: "tool_call", toolName, input: msg.input, toolCallId, ...(detail ? { detail } : {}) });
510
541
  const decision = await this.toolInterceptor({ sessionId: this.id, toolName, input: msg.input });
511
542
  try {
512
543
  this.write({ id: randomUUID(), type: "tool.decision", sessionId: this.id, toolCallId, decision: decision?.block ? "deny" : "allow", reason: decision?.reason });
@@ -518,14 +549,41 @@ class ProtocolSession {
518
549
  }
519
550
  return;
520
551
  }
552
+ if (type === "tool.update") {
553
+ // A tool call's structured data can arrive progressively (e.g. opencode's
554
+ // ACP shim often has an empty `rawInput` on the initial notification and
555
+ // fills in `content`/`locations` on a later update) — refresh the
556
+ // already-pushed turnContent block in place (never append a duplicate) and
557
+ // push a live update so an open tool card fills in without waiting for the
558
+ // turn to end and history to reconcile.
559
+ const toolCallId = String(msg.toolCallId || "");
560
+ const toolName = String(msg.name || "tool");
561
+ const detail = mapToolCall(toolName, msg.input, { provider: this.runtimeOptions.id || "acp", protocol: "protocol" });
562
+ if (detail)
563
+ this.toolDetailsByCallId.set(toolCallId, detail);
564
+ const block = this.turnContent.find((b) => b.type === "tool_use" && b.id === toolCallId);
565
+ if (block) {
566
+ block.name = toolName;
567
+ block.input = msg.input ?? {};
568
+ if (detail)
569
+ block.detail = detail;
570
+ }
571
+ this.emit({ type: "tool_execution_update", toolName, input: msg.input, toolCallId, ...(detail ? { detail } : {}) });
572
+ return;
573
+ }
521
574
  if (type === "tool.result") {
575
+ const toolCallId = String(msg.toolCallId || msg.tool_use_id || msg.id || "");
522
576
  const result = msg.result ?? msg.output ?? msg.content ?? msg.text ?? msg.summary ?? "";
577
+ const isError = Boolean(msg.isError || msg.is_error);
523
578
  this.turnToolResults.push({
524
579
  type: "tool_result",
525
- tool_use_id: String(msg.toolCallId || msg.tool_use_id || msg.id || ""),
580
+ tool_use_id: toolCallId,
526
581
  content: result,
582
+ ...(isError ? { is_error: true } : {}),
527
583
  });
528
- this.emit({ type: "tool_result", toolName: String(msg.name || "tool"), toolCallId: String(msg.toolCallId || msg.tool_use_id || msg.id || ""), result });
584
+ const priorDetail = this.toolDetailsByCallId.get(toolCallId);
585
+ const detail = priorDetail ? { ...priorDetail, result: mapToolResult(result, isError) } : undefined;
586
+ this.emit({ type: "tool_result", toolName: String(msg.name || "tool"), toolCallId, result, ...(detail ? { detail } : {}) });
529
587
  return;
530
588
  }
531
589
  this.emit({ type, ...msg });
@@ -625,8 +683,10 @@ class ProtocolSession {
625
683
  this.assistantText = "";
626
684
  this.reasoningText = "";
627
685
  this.stderrOutput = "";
628
- this.turnToolUses = [];
686
+ this.turnContent = [];
687
+ this.turnTextFlushed = "";
629
688
  this.turnToolResults = [];
689
+ this.toolDetailsByCallId.clear();
630
690
  await this.command("chat.send", {
631
691
  sessionId: this.id,
632
692
  runtimeSessionRef: this.runtimeSessionRef,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@bivy/bivy",
3
- "version": "0.9.1-staging.136",
3
+ "version": "0.10.0-staging.138",
4
4
  "type": "module",
5
5
  "license": "AGPL-3.0-only",
6
6
  "description": "Run coding agents on machines you own. Open-source, self-hostable agent workspace.",