viber-channel 0.8.1 → 0.8.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -37,10 +37,14 @@ export type BridgeToolName = (typeof BRIDGE_TOOL_NAMES)[number];
37
37
  /** Outbound tools whose SUCCESS suppresses the bridge's final auto-post. */
38
38
  const OUTBOUND_TOOLS: ReadonlySet<string> = new Set(["send_message", "message_agent"]);
39
39
 
40
- const TOOL_DEFS = [
40
+ // Exported so tests can assert the send_message/message_agent framing on THIS
41
+ // (codex/gemma) surface independently of the Claude ListTools surface — a future
42
+ // edit must not fix one runtime while breaking the other (#390 codex review).
43
+ export const TOOL_DEFS = [
41
44
  {
42
45
  name: "send_message",
43
46
  description:
47
+ "Reply to the USER. This is the ONLY way to reach the user — they do NOT see your runtime/stdout, so never answer there. " +
44
48
  "Post your reply into the current Viber conversation. `text` is the conversational reply — " +
45
49
  "read aloud AND shown as Markdown, so keep it short and natural but use light Markdown (short " +
46
50
  "lists, numbered steps, **bold**) for readability; put actions/questions here, visibly. " +
@@ -77,7 +81,7 @@ const TOOL_DEFS = [
77
81
  {
78
82
  name: "message_agent",
79
83
  description:
80
- "Send a direct message to another agent. Pass the target agent's instance id " +
84
+ "Send a direct message to ANOTHER agent — NOT for replying to the user (use send_message for that). Pass the target agent's instance id " +
81
85
  "(from list_agents) and the text. Opens or reuses a private 1:1 DM and posts your message; " +
82
86
  "the agent's reply arrives back on this channel. Targets are normally same-project; an " +
83
87
  "ORCHESTRATOR instance (#307) can also message agents of the owner's other projects. " +
@@ -0,0 +1,43 @@
1
+ // The MCP `instructions` blob served to Claude sessions on the viber-channel
2
+ // server. Extracted into a PURE function so it can be asserted in tests without
3
+ // booting the MCP server (#390 step-03).
4
+ //
5
+ // #369 truncation fix: Claude Code truncates long MCP `instructions`
6
+ // ("…[truncated]") and this blob was ~3 KB → the TAIL was dropped before the
7
+ // agent saw it (that hid the agent's name #328, and threatened the #332 pointer
8
+ // + exit-intent). So: CRITICAL lines are FRONT-LOADED (identity → core reply →
9
+ // actions/questions → anti-deadlock → capabilities → exit) and the nice-to-have
10
+ // style notes live at the tail where a cut is harmless.
11
+
12
+ /**
13
+ * Build the MCP `instructions` string.
14
+ *
15
+ * @param identityLine the per-session agent identity line (from
16
+ * agentIdentityFromEnv()) — "" for JP's own session (no explicit-name flag),
17
+ * in which case it is filtered out and the rest of the blob still applies.
18
+ */
19
+ export function buildChannelInstructions(identityLine: string): string {
20
+ return [
21
+ identityLine,
22
+ // --- critical: front-loaded so truncation can never drop them ---
23
+ // #390: route the REPLY unambiguously. The <channel> header carries TWO
24
+ // source= attributes — the first is the MCP server name (always
25
+ // "viber-channel"), the LAST is meta.source, the real discriminant:
26
+ // "microphone" = voice, "conversation" = typed UI text (both = the USER),
27
+ // "agent-dm" = another agent. Naming the real values (not the server name)
28
+ // avoids the old imprecise "source=viber-channel" cue.
29
+ 'To reply to the USER, ALWAYS use send_message — NEVER write your answer to the terminal / your runtime stdout (the user does not watch it; there may be many terminals). A <channel> event whose metadata source (the last source= attribute) is "microphone" (voice) or "conversation" (typed text) is the user talking to you → reply with send_message. source="agent-dm" is ANOTHER agent → reply with message_agent. send_message = the user; message_agent = other agents ONLY, never the user.',
30
+ "Your send_message `text` is BOTH read aloud (TTS) AND shown as Markdown — make it natural aloud AND easy to read; lead with the answer, no preamble.",
31
+ "Put ACTIONS and QUESTIONS in `text`, visibly — never bury them in the artifact. Use light Markdown (short bullet/numbered lists, **bold**) to stay readable; a short spoken list is fine.",
32
+ // #343 anti-deadlock (hands-free: the user is NOT watching the terminal).
33
+ "CRITICAL — while this channel is active, NEVER block on a terminal prompt or AskUserQuestion: the user is hands-free and cannot see the terminal, so it deadlocks. Put EVERY question or choice in send_message `text` and take the answer from the next user message (voice or typed).",
34
+ // #332 discoverability pointer (the how-to lives in the capabilities tool).
35
+ "To orchestrate or spawn OTHER agents (Codex, Claude, Gemma) on this machine, call the `capabilities` tool for how — only when relevant.",
36
+ "On exit intent (bye, au revoir, stop) call stop_conversation(), speak a brief farewell, and stop.",
37
+ // --- nice-to-have style notes (safe near the tail) ---
38
+ "The `artifact` (format: markdown default, or code/json/html) is for HEAVY/LONG content (big code, large tables, long analyses, JSON, file lists); keep `text` a brief spoken summary that points to it ('details on the side'). Scripts/event handlers are stripped server-side.",
39
+ "NO EMOJI in `text` (read aloud — an emoji becomes spoken noise). Write identifiers LITERALLY (288, auth.json, viber-dev.dgypx.dev) — never spell out dots/dashes; a lone long token or path is better placed in the artifact.",
40
+ ]
41
+ .filter((line) => line.length > 0)
42
+ .join(" ");
43
+ }
@@ -297,6 +297,12 @@ export function runPersistentControlStream(
297
297
  };
298
298
 
299
299
  const done = (async () => {
300
+ // #389 step-02: announce ONCE that the instance control stream actually
301
+ // connected server-side (the SSE `connected` open-marker arrived) — proof
302
+ // the agent is ONLINE/reachable, not merely registered. vibe-master's spawn
303
+ // readiness gate polls the bridge log for this line, so a launch that
304
+ // registers then fails to connect (CF/401/network) is NOT reported as ready.
305
+ let announcedOnline = false;
300
306
  try {
301
307
  while (true) {
302
308
  if (signal?.aborted) {
@@ -305,6 +311,13 @@ export function runPersistentControlStream(
305
311
  }
306
312
  try {
307
313
  for await (const ev of subscribeControlStream(opts)) {
314
+ if (ev.event === "connected") {
315
+ if (!announcedOnline) {
316
+ announcedOnline = true;
317
+ log("[control-stream] online — instance control stream connected\n");
318
+ }
319
+ continue;
320
+ }
308
321
  if (ev.event === "stop") {
309
322
  settleFirstReject(new ControlStreamStopped());
310
323
  handlers.onStop?.("stopped");
@@ -314,7 +327,7 @@ export function runPersistentControlStream(
314
327
  handlers.onBeatNow?.();
315
328
  continue;
316
329
  }
317
- if (ev.event !== "join") continue; // connected / ping / unknown
330
+ if (ev.event !== "join") continue; // ping / unknown (connected handled above)
318
331
  const minted = parseJoinPayload(ev.data, log);
319
332
  if (!minted) continue;
320
333
  handlers.onJoin?.(minted);
package/package.json CHANGED
@@ -1,46 +1,46 @@
1
- {
2
- "name": "viber-channel",
3
- "version": "0.8.1",
4
- "description": "Voice + text MCP channel between a Claude Code session and the Viber UI (https://viber.dgypx.dev). Push transcripts to Claude; send_message tool delivers text back to the UI.",
5
- "type": "module",
6
- "bin": {
7
- "viber-channel": "./viber-channel.ts",
8
- "viber-codex-bridge": "./viber-codex-bridge.ts",
9
- "viber-gemma-bridge": "./viber-gemma-bridge.ts",
10
- "viber-codex-supervisor": "./viber-codex-supervisor.ts"
11
- },
12
- "files": [
13
- "viber-channel.ts",
14
- "viber-codex-bridge.ts",
15
- "viber-gemma-bridge.ts",
16
- "viber-codex-supervisor.ts",
17
- "lib/",
18
- "README.md"
19
- ],
20
- "repository": {
21
- "type": "git",
22
- "url": "git+https://github.com/dgx80/viber.git",
23
- "directory": "viber-channel"
24
- },
25
- "homepage": "https://viber.dgypx.dev",
26
- "bugs": {
27
- "url": "https://github.com/dgx80/viber/issues"
28
- },
29
- "keywords": [
30
- "viber",
31
- "claude-code",
32
- "mcp",
33
- "channel",
34
- "voice",
35
- "transcription"
36
- ],
37
- "license": "MIT",
38
- "scripts": {
39
- "start": "bun run viber-channel.ts",
40
- "start:codex-bridge": "bun run viber-codex-bridge.ts",
41
- "test": "bun test"
42
- },
43
- "dependencies": {
44
- "@modelcontextprotocol/sdk": "^1.0.0"
45
- }
46
- }
1
+ {
2
+ "name": "viber-channel",
3
+ "version": "0.8.3",
4
+ "description": "Voice + text MCP channel between a Claude Code session and the Viber UI (https://viber.dgypx.dev). Push transcripts to Claude; send_message tool delivers text back to the UI.",
5
+ "type": "module",
6
+ "bin": {
7
+ "viber-channel": "./viber-channel.ts",
8
+ "viber-codex-bridge": "./viber-codex-bridge.ts",
9
+ "viber-gemma-bridge": "./viber-gemma-bridge.ts",
10
+ "viber-codex-supervisor": "./viber-codex-supervisor.ts"
11
+ },
12
+ "files": [
13
+ "viber-channel.ts",
14
+ "viber-codex-bridge.ts",
15
+ "viber-gemma-bridge.ts",
16
+ "viber-codex-supervisor.ts",
17
+ "lib/",
18
+ "README.md"
19
+ ],
20
+ "repository": {
21
+ "type": "git",
22
+ "url": "git+https://github.com/dgx80/viber.git",
23
+ "directory": "viber-channel"
24
+ },
25
+ "homepage": "https://viber.dgypx.dev",
26
+ "bugs": {
27
+ "url": "https://github.com/dgx80/viber/issues"
28
+ },
29
+ "keywords": [
30
+ "viber",
31
+ "claude-code",
32
+ "mcp",
33
+ "channel",
34
+ "voice",
35
+ "transcription"
36
+ ],
37
+ "license": "MIT",
38
+ "scripts": {
39
+ "start": "bun run viber-channel.ts",
40
+ "start:codex-bridge": "bun run viber-codex-bridge.ts",
41
+ "test": "bun test"
42
+ },
43
+ "dependencies": {
44
+ "@modelcontextprotocol/sdk": "^1.0.0"
45
+ }
46
+ }
package/viber-channel.ts CHANGED
@@ -55,6 +55,7 @@ import {
55
55
  } from "./lib/token_refresh.ts";
56
56
  import { awaitStableStartup } from "./lib/startup_gate.ts";
57
57
  import { agentIdentityFromEnv } from "./lib/bridge_core.ts";
58
+ import { buildChannelInstructions } from "./lib/channel_instructions.ts";
58
59
  import { warnIfStale } from "./lib/version_check.ts";
59
60
  import { capabilitiesText } from "./lib/capabilities.ts";
60
61
 
@@ -353,32 +354,10 @@ const mcp = new Server(
353
354
  experimental: { "claude/channel": {} },
354
355
  tools: {},
355
356
  },
356
- // #369 truncation fix: Claude Code truncates long MCP `instructions`
357
- // ("…[truncated]") and this blob was ~3 KB → the TAIL was dropped before the
358
- // agent saw it (that hid the agent's name #328, and threatened the #332
359
- // pointer + exit-intent). So: CRITICAL lines are FRONT-LOADED (identity →
360
- // core reply → actions/questions → anti-deadlock → capabilities → exit) and
361
- // the nice-to-have style notes live at the tail where a cut is harmless.
362
- // Condensed vs the old prose (same substance — reviewed). Identity is "" for
363
- // JP's own session (no explicit-name flag) → the identity LINE is filtered
364
- // out for him; the rest of the (now shorter, reordered) blob applies to every
365
- // session including his.
366
- instructions: [
367
- agentIdentityFromEnv(),
368
- // --- critical: front-loaded so truncation can never drop them ---
369
- 'Voice transcripts arrive as <channel source="viber-channel"> events (the user\'s microphone speech). Reply with send_message. The `text` is BOTH read aloud (TTS) AND shown as Markdown — make it natural aloud AND easy to read; lead with the answer, no preamble.',
370
- "Put ACTIONS and QUESTIONS in `text`, visibly — never bury them in the artifact. Use light Markdown (short bullet/numbered lists, **bold**) to stay readable; a short spoken list is fine.",
371
- // #343 anti-deadlock (hands-free: the user is NOT watching the terminal).
372
- "CRITICAL — while this channel is active, NEVER block on a terminal prompt or AskUserQuestion: the user is hands-free and cannot see the terminal, so it deadlocks. Put EVERY question or choice in send_message `text` and take the answer from the next voice transcript.",
373
- // #332 discoverability pointer (the how-to lives in the capabilities tool).
374
- "To orchestrate or spawn OTHER agents (Codex, Claude, Gemma) on this machine, call the `capabilities` tool for how — only when relevant.",
375
- "On exit intent (bye, au revoir, stop) call stop_conversation(), speak a brief farewell, and stop.",
376
- // --- nice-to-have style notes (safe near the tail) ---
377
- "The `artifact` (format: markdown default, or code/json/html) is for HEAVY/LONG content (big code, large tables, long analyses, JSON, file lists); keep `text` a brief spoken summary that points to it ('details on the side'). Scripts/event handlers are stripped server-side.",
378
- "NO EMOJI in `text` (read aloud — an emoji becomes spoken noise). Write identifiers LITERALLY (288, auth.json, viber-dev.dgypx.dev) — never spell out dots/dashes; a lone long token or path is better placed in the artifact.",
379
- ]
380
- .filter((line) => line.length > 0)
381
- .join(" "),
357
+ // The blob lives in a PURE builder (lib/channel_instructions.ts) so it can be
358
+ // asserted in tests without booting the server (#390). Identity is "" for JP's
359
+ // own session (no explicit-name flag) → filtered out; the rest applies to all.
360
+ instructions: buildChannelInstructions(agentIdentityFromEnv()),
382
361
  }
383
362
  );
384
363
 
@@ -389,6 +368,7 @@ mcp.setRequestHandler(ListToolsRequestSchema, async () => ({
389
368
  {
390
369
  name: "send_message",
391
370
  description:
371
+ "Reply to the USER. This is how you talk to the user — NOT the terminal / your stdout (they do not watch it). Calling this IS your reply for the turn. " +
392
372
  "Send a message into the Viber conversation this channel is bound to. " +
393
373
  "`text` is the conversational reply — it is BOTH read aloud (TTS) AND rendered as Markdown on screen, so keep it short and natural but use light Markdown (short bullet/numbered lists, **bold**) when it aids readability. Put any ACTIONS or QUESTIONS for the user here, visibly. " +
394
374
  "`artifact` is optional and carries HEAVY/LONG content (big code, large tables, long analyses, JSON, sanitized HTML) rendered in a side viewer; when used, keep `text` a brief summary that points to it. " +
@@ -444,7 +424,7 @@ mcp.setRequestHandler(ListToolsRequestSchema, async () => ({
444
424
  {
445
425
  name: "message_agent",
446
426
  description:
447
- "Send a direct message to another agent (#288). " +
427
+ "Send a direct message to ANOTHER agent (#288) — NOT for replying to the user (use send_message for that). " +
448
428
  "Pass the target agent's instance id (from list_agents, or the from_instance_id of a DM you received) " +
449
429
  "and the message text. Opens or reuses a private 1:1 DM with that agent and posts your message; " +
450
430
  "the agent's reply arrives back on this channel tagged source=agent-dm. " +
@@ -356,7 +356,7 @@ function instructionsForTier(tier: AgentTier): string {
356
356
  return [
357
357
  "You are a bridge-owned Codex agent connected to Viber, in READ-WRITE mode.",
358
358
  "You MAY read files, create/modify files within the workspace, and run commands (including state-changing ones) to carry out the user's requests.",
359
- "The sandbox is workspace-write: writes are confined to the workspace and network access is blocked at the OS level.",
359
+ "The sandbox is workspace-write: writes are confined to the workspace, and network access is ENABLED — so you can run git (including `git push`) and open pull requests (e.g. with `gh`).",
360
360
  "There is NO human approval step, so be deliberate: make only the changes the user asked for, and avoid destructive commands unless explicitly requested.",
361
361
  VOICE_CONCISE_LINE,
362
362
  CHANNEL_TOOLS_LINE,
@@ -385,17 +385,43 @@ function instructionsForTier(tier: AgentTier): string {
385
385
  // Approval policy per tier. `read`/`write` use "never" so Codex runs its commands
386
386
  // WITHOUT a human approver (there is none on the bridge) — the SANDBOX is the real
387
387
  // guardrail (read-only blocks writes/network; workspace-write confines writes to
388
- // the workspace and still blocks network). `chat` keeps "on-request" (no exec anyway).
389
- function approvalForTier(tier: AgentTier): "never" | "on-request" {
388
+ // the workspace). `chat` keeps "on-request" (no exec anyway).
389
+ export function approvalForTier(tier: AgentTier): "never" | "on-request" {
390
390
  return tier === "chat" ? "on-request" : "never";
391
391
  }
392
392
 
393
393
  // Codex OS sandbox per tier. `write` widens to workspace-write (edits confined to
394
- // the workspace, network still blocked); chat/read stay read-only.
395
- function sandboxForTier(tier: AgentTier): "read-only" | "workspace-write" {
394
+ // the workspace); chat/read stay read-only.
395
+ export function sandboxForTier(tier: AgentTier): "read-only" | "workspace-write" {
396
396
  return tier === "write" ? "workspace-write" : "read-only";
397
397
  }
398
398
 
399
+ // Network access per tier (#389 step-03). ONLY the trusted `write` tier gets
400
+ // network — so a codex coder can `git push` and open its PR (git-over-https +
401
+ // `gh`). read/chat stay offline. SECURITY: codex's workspace-write sandbox opens
402
+ // ALL network, not git-only (the app-server sandbox has no per-domain allowlist),
403
+ // so this is a TRUST decision scoped to the write tier, which is already
404
+ // operator-authorized-at-launch for unattended command execution. If codex ever
405
+ // exposes an outbound-domain allowlist, tighten here (follow-up issue).
406
+ export function networkForTier(tier: AgentTier): boolean {
407
+ return tier === "write";
408
+ }
409
+
410
+ // The EXACT `sandboxPolicy` object sent on every `turn/start` (#389 step-03).
411
+ // Per the codex app-server protocol (generate-ts), `networkAccess` lives ONLY on
412
+ // this per-turn policy object (workspaceWrite/readOnly variants) — NOT on
413
+ // thread/start|resume, which take a bare `sandbox` MODE string. So the turn-level
414
+ // policy is the single lever for network. Exported to unit-test the payload/tier.
415
+ export function buildTurnSandboxPolicy(
416
+ tier: AgentTier,
417
+ ):
418
+ | { type: "workspaceWrite"; networkAccess: boolean }
419
+ | { type: "readOnly"; networkAccess: boolean } {
420
+ return sandboxForTier(tier) === "workspace-write"
421
+ ? { type: "workspaceWrite", networkAccess: networkForTier(tier) }
422
+ : { type: "readOnly", networkAccess: false };
423
+ }
424
+
399
425
  const AGENT_TIER: AgentTier = resolveAgentTier();
400
426
  // Identity line (#309 step-17) is appended AFTER the tier instructions so the
401
427
  // permission rules stay first + strongest; the role is framed as descriptive and
@@ -635,10 +661,7 @@ class CodexAppServer {
635
661
  cwd: process.cwd(),
636
662
  runtimeWorkspaceRoots: [process.cwd()],
637
663
  approvalPolicy: AGENT_APPROVAL_POLICY,
638
- sandboxPolicy:
639
- AGENT_SANDBOX === "workspace-write"
640
- ? { type: "workspaceWrite", networkAccess: false }
641
- : { type: "readOnly", networkAccess: false },
664
+ sandboxPolicy: buildTurnSandboxPolicy(AGENT_TIER),
642
665
  developerInstructions: AGENT_INSTRUCTIONS,
643
666
  responsesapiClientMetadata: { source: "viber-codex-bridge" },
644
667
  });
@@ -1115,7 +1138,7 @@ function setupCodexRuntime(deps: {
1115
1138
 
1116
1139
  function logCodexSafety(): void {
1117
1140
  process.stderr.write(
1118
- `${LOG_PREFIX} codex safety: tier=${AGENT_TIER}, sandbox=${AGENT_SANDBOX}, approvalPolicy=${AGENT_APPROVAL_POLICY}, network=false, ` +
1141
+ `${LOG_PREFIX} codex safety: tier=${AGENT_TIER}, sandbox=${AGENT_SANDBOX}, approvalPolicy=${AGENT_APPROVAL_POLICY}, network=${networkForTier(AGENT_TIER)}, ` +
1119
1142
  `instructions=${
1120
1143
  AGENT_TIER === "write"
1121
1144
  ? "read+write/run commands (workspace)"