agello 0.4.2 → 0.4.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -8,6 +8,7 @@
8
8
  - Interactive terminal view of the existing herdr pane: live cursor, keyboard input, and automatic resize
9
9
  - Optional live view of a `terminal-browser` screen, with an on/off switch to control it yourself
10
10
  - Annotate: point at an element on that screen and send a request about it, with its CSS selector
11
+ - Tab sound: hear that screen's audio on the page instead of the agent's machine
11
12
  - Embeddable `<agent-bridge>` web component
12
13
 
13
14
  Chat currently supports **Claude Code** sessions. The terminal view connects directly to the pane, including ordinary shells and other terminal applications.
@@ -73,6 +74,10 @@ The `주석` button on the screen panel turns on annotate mode: hovering tints t
73
74
 
74
75
  Selector: nearest unique `id` / `data-testid` ancestor, then `tag:nth-of-type` steps (light DOM only). Coordinates are CSS px of the viewport. Off while presenting and while "내 조작" is on.
75
76
 
77
+ ### Tab sound
78
+
79
+ The `소리` button on the screen panel plays the relayed tab's audio on that page. While at least one page has it on, the tab is muted in the agent's terminal-browser; when the last one turns it off or closes, local playback returns within 4 seconds. agello reroutes the page's Web Audio output and `<audio>`/`<video>` elements into PCM chunks (`GET /screen/audio`, SSE), since CDP has no audio stream. Not captured: Web Audio connected before it was turned on (reload the tab), cross-origin media without CORS (keeps playing locally), cross-site iframes. Audio trails the screen by about 0.2s.
80
+
76
81
  ### Commands
77
82
 
78
83
  | Command | Description |
@@ -113,7 +118,8 @@ The xterm.js renderer and its styles are served locally; no terminal CDN is requ
113
118
  - agent replies appear over it as bubbles that fade after 10s
114
119
  - `agello present stop` returns to the normal layout; the replies are in the chat as usual
115
120
  - the viewer can ask a question with the raise-hand button (its bubble fades after 10s too) or end the presentation with the end button; the agent then receives `[browser] action=present-stop`
116
- - raising the hand tells the agent at once (`action=hand-raise`) so it can pause; closing the input without asking sends `action=hand-lower`
121
+ - raising the hand tells the agent at once (`action=hand-raise`) so it can pause; closing it with X (or Esc) before asking sends `action=hand-lower`
122
+ - the input stays open for the whole question: after the agent answers, follow-up questions go in the same input (no need to raise the hand again); X (or Esc) then sends `action=hand-done`, and the agent resumes
117
123
 
118
124
  Run `present` again to move the crop. User control is off while presenting.
119
125
 
@@ -130,10 +136,35 @@ Write the talk ahead so each line appears the moment its screen does:
130
136
 
131
137
  - step: `go` brings the screen there (`#n` sets `location.hash`, anything else is JS run in the page; agello knows nothing about the deck), `say` lines are one bubble each, `hold` seconds a line stays on screen (default from length, 10 to 20s; never below 10s). Pacing: screen change, 1s, line, fade out, 0.3s, next line; after a step's last line 0.7s before the next screen
132
138
  - `present resume` plays from where it stopped (starts presentation mode with the script's `rect`); it first re-runs the current step's `go`, so the agent may move the screen freely while answering
133
- - raising a hand pauses at once; the agent gets `action=hand-raise` with the position (`마지막 표시 2.1 · 다음 2.2`)
139
+ - raising a hand pauses at once and rewinds one line, so `present resume` replays the interrupted line; the agent gets `action=hand-raise` with the position (`마지막 표시 2.1 · 다음 2.1`)
140
+ - raising, cancelling (X before asking) and finishing a question (X after an answer) show a small light event bubble for 5s right away, apart from the conversation bubbles; sending a question shows none
134
141
  - `script load` again after editing keeps the position; `present goto 2.2` then `present resume` replays from a fixed line
135
142
  - at the end the agent gets `action=present-done`; while the script is paused and the agent is working, the page shows a small loader
136
143
 
144
+ #### Optional TTS and recording
145
+
146
+ The screen panel's **발표 설정** remains available during presentation. TTS and recording both start OFF and can be used independently:
147
+
148
+ | TTS | Recording | Result |
149
+ |---|---|---|
150
+ | OFF | OFF | Existing timed script captions |
151
+ | OFF | ON | Presentation images and captions in a silent video |
152
+ | ON | OFF | Spoken script, with captions synchronized to audio playback |
153
+ | ON | ON | Presentation images, captions, and TTS audio in the controlling viewer's video |
154
+
155
+ 1. Open **발표 설정**. Enter a Typecast API key and optionally change the voice ID, then click **키·목소리 적용**. The default voice is `tc_6a0e85a97f7750959b970d5d`; the provider uses Korean (`kor`) and model `ssfm-v30`.
156
+ 2. Alternatively, **로컬 키 불러오기** reads only `TYPECAST_API_KEY` from `../yt-outlier/.env`, relative to the server's working directory. It does not execute the file or import other settings. A server started with `TYPECAST_API_KEY` in its environment also has a key ready, with TTS still OFF.
157
+ 3. Click **TTS 켜기** in the viewer that should play sound. This click grants browser audio playback permission. Click **대본 재생** after loading a script; **일시정지** stops playback, and **다시 재생** starts a completed script again from its first step.
158
+ 4. Click **녹화 시작**, then **녹화 종료** and **영상 저장** to download the recording. Recording can start before presentation if a screen image is already available. Stopping recording leaves script and audio playback running.
159
+
160
+ With TTS enabled, each script caption appears after that viewer starts the actual audio, and closes when the audio finishes. The audio's duration replaces the script's `hold`; ordinary agent replies still use their existing caption timer. Raising a hand immediately stops audio locally, and resumed playback repeats the interrupted line. Pausing, changing steps, updating a script, or disconnecting cancels outdated speech. A TTS failure pauses the script; fix the key or connection, or turn TTS OFF, then resume.
161
+
162
+ Only the viewer that enabled TTS plays audio and acknowledges speech progress. Other viewers receive the same captions and can record a silent video. To include TTS audio, record in the viewer that enabled TTS. Audio settings are fixed when recording starts: stop recording before changing TTS. If another viewer changes the controlling audio role, a recording whose audio setting changes is finalized automatically.
163
+
164
+ Recording uses a 1280×720 canvas that fits the relayed browser image with letterboxing and draws the current captions below it. It records these images and plain caption text, rather than the page's entire DOM: chat, settings, question input, and control buttons are excluded; caption Markdown styling is flattened. Long captions are wrapped and truncated to the safe area. The canvas targets 30fps, while source screen updates can be slower (presentation screenshots are approximately 10fps). A background or minimized tab may render more slowly. Use a current Chromium browser for the verified recording path; other browsers depend on their `canvas.captureStream`, Web Audio, and `MediaRecorder` support. MP4 is preferred when supported, with WebM as a fallback. Streaming WebM duration metadata is finalized when its header supports the correction; otherwise the original browser recording is retained.
165
+
166
+ Recording stops at 15 minutes or 256MB, at presentation end, on server changes, or on connection loss. Keep the page open until **영상 저장** appears and download it before closing or reloading; recording data is held in browser memory. A later recording replaces the earlier download link.
167
+
137
168
  ## Embed
138
169
 
139
170
  ```html
@@ -165,9 +196,11 @@ See `web/embed-example.html`.
165
196
 
166
197
  The server listens on `127.0.0.1` only. Data and input endpoints accept requests from the same origin and `localhost` pages only; other origins get `403` unless added with `--allow-origin`. Allowed origins can send chat prompts and connect to `/terminal` to type directly into the pane, so keep the allow list short.
167
198
 
199
+ TTS keys remain in server memory until replaced, cleared with **키 삭제**, or the server stops. Password inputs are cleared after applying; keys and speech-driver tokens are never put in browser storage, chat history, scripts, or SSE broadcasts. Keys are sent only to the local server and Typecast. Script text is sent to Typecast when TTS is ON, and generated audio is cached in server memory for reuse. Allowed origins can also configure TTS and read generated audio, so allow only trusted pages.
200
+
168
201
  ## Development
169
202
 
170
- Run `bun install` and `bun test`. Tests cover the pane tree and creation (against a mock herdr), local terminal assets, origin rejection, ANSI frames, keyboard bytes, resize, exclusive control, and reconnect.
203
+ Run `bun install` and `bun test`. Tests cover the pane tree and creation (against a mock herdr), local terminal assets, origin rejection, ANSI frames, keyboard bytes, resize, exclusive control, reconnect, TTS coordination, stale speech cancellation, and recording lifecycle and metadata.
171
204
 
172
205
  ## License
173
206
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "agello",
3
- "version": "0.4.2",
3
+ "version": "0.4.4",
4
4
  "description": "Talk to a coding agent in a herdr pane from the browser: chat, live status, and a shared browser screen.",
5
5
  "keywords": [
6
6
  "herdr",
@@ -0,0 +1,159 @@
1
+ // Page audio relay, page side. Injected into the relayed terminal-browser tab
2
+ // through CDP while at least one viewer listens (GET /screen/audio).
3
+ //
4
+ // CDP has no audio stream and terminal-browser (Electron) denies tab capture,
5
+ // so the page's own audio graph is rerouted:
6
+ // - AudioNode.connect(ctx.destination) goes to a per-context tap instead
7
+ // - <audio>/<video> playing while remote are moved into a WebAudio graph
8
+ // (createMediaElementSource), which also takes them off the speakers
9
+ // tap -> out (gain 1 local, 0 remote) -> real destination
10
+ // tap -> ScriptProcessor -> 16-bit stereo PCM -> binding -> agello server
11
+ //
12
+ // Remote mode lasts while the server keeps calling ping() (every ~1.5s); after
13
+ // 4s without a ping (server gone, tab no longer relayed) local playback returns.
14
+ //
15
+ // Not captured: connections made before injection (reload the page),
16
+ // cross-origin media without CORS (routing them would play silence, so they
17
+ // keep playing locally), cross-site iframes (separate CDP targets).
18
+
19
+ export const AUDIO_BINDING = "__agelloAudioPcm";
20
+
21
+ export const PAGE_AUDIO_JS = `(() => {
22
+ if (window.__agelloAudio) return;
23
+ const AC = window.AudioContext || window.webkitAudioContext;
24
+ if (!AC || !window.AudioNode) return;
25
+ const BINDING = ${JSON.stringify(AUDIO_BINDING)};
26
+ const QUIET_CHUNKS = 24; // ~2s of silence before chunks stop being sent
27
+ const origConnect = AudioNode.prototype.connect;
28
+ const origDisconnect = AudioNode.prototype.disconnect;
29
+ const origCreateSource = AC.prototype.createMediaElementSource;
30
+ const origPlay = HTMLMediaElement.prototype.play;
31
+ const taps = new WeakMap(); // context -> tap record
32
+ const records = new Set();
33
+ const owned = new WeakSet(); // media elements moved into our graph
34
+ let remote = false, lastPing = 0, nextId = 1, mediaCtx = null;
35
+
36
+ function send(t, buf) {
37
+ const n = buf.length;
38
+ const L = buf.getChannelData(0), R = buf.numberOfChannels > 1 ? buf.getChannelData(1) : L;
39
+ const pcm = new Int16Array(n * 2);
40
+ let loud = false;
41
+ for (let i = 0; i < n; i++) {
42
+ const l = L[i], r = R[i];
43
+ if (l || r) loud = true;
44
+ pcm[2 * i] = (l < -1 ? -1 : l > 1 ? 1 : l) * 32767;
45
+ pcm[2 * i + 1] = (r < -1 ? -1 : r > 1 ? 1 : r) * 32767;
46
+ }
47
+ t.quiet = loud ? 0 : t.quiet + 1;
48
+ if (t.quiet > QUIET_CHUNKS) return;
49
+ const bytes = new Uint8Array(pcm.buffer);
50
+ let s = "";
51
+ for (let i = 0; i < bytes.length; i += 0x8000) s += String.fromCharCode.apply(null, bytes.subarray(i, i + 0x8000));
52
+ try { window[BINDING](JSON.stringify({ c: t.id, r: t.ctx.sampleRate, d: btoa(s) })); } catch {}
53
+ }
54
+
55
+ function tapOf(ctx) {
56
+ let t = taps.get(ctx);
57
+ if (t) return t;
58
+ const tap = ctx.createGain(), out = ctx.createGain();
59
+ out.gain.value = remote ? 0 : 1;
60
+ origConnect.call(tap, out);
61
+ origConnect.call(out, ctx.destination);
62
+ t = { id: nextId++, ctx, tap, out, quiet: QUIET_CHUNKS + 1 };
63
+ if (ctx.createScriptProcessor) {
64
+ const proc = ctx.createScriptProcessor(4096, 2, 2), sink = ctx.createGain();
65
+ sink.gain.value = 0;
66
+ origConnect.call(tap, proc);
67
+ origConnect.call(proc, sink);
68
+ origConnect.call(sink, ctx.destination);
69
+ proc.onaudioprocess = (e) => { if (remote) send(t, e.inputBuffer); };
70
+ t.proc = proc;
71
+ }
72
+ taps.set(ctx, t);
73
+ records.add(t);
74
+ return t;
75
+ }
76
+
77
+ AudioNode.prototype.connect = function (dest, output) {
78
+ const ctx = this.context;
79
+ if (dest && ctx instanceof AC && dest === ctx.destination) {
80
+ origConnect.call(this, tapOf(ctx).tap, output || 0);
81
+ return dest;
82
+ }
83
+ return origConnect.apply(this, arguments);
84
+ };
85
+ AudioNode.prototype.disconnect = function (dest) {
86
+ const ctx = this.context, t = ctx && taps.get(ctx);
87
+ if (t && dest && dest === ctx.destination) {
88
+ const args = [...arguments];
89
+ args[0] = t.tap;
90
+ if (args.length > 2) args.length = 2; // the tap has one input
91
+ return origDisconnect.apply(this, args);
92
+ }
93
+ return origDisconnect.apply(this, arguments);
94
+ };
95
+
96
+ function capturable(el) {
97
+ if (el.srcObject) return true;
98
+ const src = el.currentSrc || el.src;
99
+ if (!src) return false;
100
+ if (el.crossOrigin != null) return true;
101
+ try {
102
+ const u = new URL(src, location.href);
103
+ return u.protocol === "data:" || u.origin === location.origin;
104
+ } catch { return false; }
105
+ }
106
+ function own(el) {
107
+ if (!remote || !(el instanceof HTMLMediaElement) || owned.has(el) || !capturable(el)) return;
108
+ try {
109
+ mediaCtx = mediaCtx || new AC();
110
+ const src = origCreateSource.call(mediaCtx, el);
111
+ owned.add(el);
112
+ origConnect.call(src, tapOf(mediaCtx).tap);
113
+ if (mediaCtx.state !== "running") mediaCtx.resume().catch(() => {});
114
+ } catch {}
115
+ }
116
+ HTMLMediaElement.prototype.play = function () {
117
+ own(this);
118
+ return origPlay.apply(this, arguments);
119
+ };
120
+ for (const type of ["play", "playing"]) document.addEventListener(type, (e) => own(e.target), true);
121
+
122
+ function setRemote(on) {
123
+ if (on) lastPing = Date.now();
124
+ if (remote === on) return;
125
+ remote = on;
126
+ for (const t of records) {
127
+ t.out.gain.value = on ? 0 : 1;
128
+ t.quiet = QUIET_CHUNKS + 1;
129
+ }
130
+ if (on) {
131
+ document.querySelectorAll("audio,video").forEach((el) => { if (!el.paused) own(el); });
132
+ if (mediaCtx && mediaCtx.state !== "running") mediaCtx.resume().catch(() => {});
133
+ }
134
+ }
135
+ setInterval(() => { if (remote && Date.now() - lastPing > 4000) setRemote(false); }, 1000);
136
+
137
+ window.__agelloAudio = {
138
+ ping: () => setRemote(true),
139
+ off: () => setRemote(false),
140
+ state: () => ({ remote, contexts: records.size, media: !!mediaCtx }),
141
+ };
142
+ })()`;
143
+
144
+ // Viewer-bound chunk, validated (the binding is callable by the page itself).
145
+ export type PcmChunk = { c: number; r: number; d: string };
146
+ export function parsePcm(payload: string): PcmChunk | null {
147
+ if (payload.length > 512 * 1024) return null;
148
+ let m: any;
149
+ try {
150
+ m = JSON.parse(payload);
151
+ } catch {
152
+ return null;
153
+ }
154
+ const { c, r, d } = m ?? {};
155
+ if (!Number.isInteger(c) || c < 1 || c > 1e6) return null;
156
+ if (!Number.isFinite(r) || r < 8000 || r > 192000) return null;
157
+ if (typeof d !== "string" || !d || d.length % 4 || !/^[A-Za-z0-9+/]+={0,2}$/.test(d)) return null;
158
+ return { c, r, d };
159
+ }
package/src/player.ts CHANGED
@@ -9,6 +9,7 @@
9
9
  // line is sent and held for its reading time (see PACE). pause() stops at once and keeps
10
10
  // the position, resume() re-runs the current step's `go` (the agent may have
11
11
  // moved the screen meanwhile) and continues from the next unshown line.
12
+ // rewind() (after a raised hand) moves back so the last shown line plays again.
12
13
  // goto() moves the position and shows that step without playing.
13
14
  //
14
15
  // Positions are "step.line", 1-based, in everything that leaves this module.
@@ -22,6 +23,7 @@ export type PlayerInfo = {
22
23
  next?: string; // next line to show ("3.2")
23
24
  last?: string; // last line shown
24
25
  steps: number;
26
+ error?: string;
25
27
  };
26
28
 
27
29
  // Pacing: the audience looks at a new screen first, then reads and thinks.
@@ -79,7 +81,10 @@ export function parsePos(v: string): { step: number; line: number } | null {
79
81
 
80
82
  export function createPlayer(deps: {
81
83
  go: (expr: string) => Promise<void>;
82
- say: (text: string, at: string, hold: number) => void;
84
+ say: (text: string, at: string, hold: number, narration?: string) => void;
85
+ // null selects text pacing; a promise resolves only after actual playback ends.
86
+ narrate?: (text: string, at: string, signal: AbortSignal,
87
+ started: (duration: number, id: string) => void) => Promise<void> | null;
83
88
  done: () => void;
84
89
  changed: () => void;
85
90
  }) {
@@ -90,6 +95,8 @@ export function createPlayer(deps: {
90
95
  let entered = false; // `go` of cur.step already run in this play run
91
96
  let run = 0; // bumps on every pause/goto/load, stale timers check it
92
97
  let timer: Timer | null = null;
98
+ let speech: AbortController | null = null;
99
+ let error: string | undefined;
93
100
 
94
101
  const fmt = (p: { step: number; line: number }) => `${p.step + 1}.${p.line + 1}`;
95
102
  const set = (s: PlayerInfo["state"]) => {
@@ -100,6 +107,8 @@ export function createPlayer(deps: {
100
107
  run++;
101
108
  if (timer) clearTimeout(timer);
102
109
  timer = null;
110
+ speech?.abort();
111
+ speech = null;
103
112
  };
104
113
  const wait = (ms: number, id: number, fn: () => void) => {
105
114
  timer = setTimeout(() => id === run && fn(), ms);
@@ -116,6 +125,7 @@ export function createPlayer(deps: {
116
125
  steps,
117
126
  next: script && p.step < steps ? fmt(p) : undefined,
118
127
  last: last ? fmt(last) : undefined,
128
+ ...(error ? { error } : {}),
119
129
  };
120
130
  }
121
131
 
@@ -154,13 +164,38 @@ export function createPlayer(deps: {
154
164
  return;
155
165
  }
156
166
  const text = step.say[cur.line];
157
- const hold = lineHold(text, step.hold);
167
+ const position = { ...cur };
158
168
  const lastLine = cur.line === step.say.length - 1;
169
+ const gap = lastLine ? PACE.afterStep : PACE.betweenLines;
170
+ speech = new AbortController();
171
+ const narration = deps.narrate?.(text, fmt(cur), speech.signal, (duration, narrationId) => {
172
+ if (id !== run || state !== "playing") return;
173
+ last = position;
174
+ deps.say(text, fmt(position), duration, narrationId);
175
+ deps.changed();
176
+ });
177
+ if (narration) {
178
+ try {
179
+ await narration;
180
+ if (id !== run || state !== "playing") return;
181
+ speech = null;
182
+ cur = { step: position.step, line: position.line + 1 };
183
+ deps.changed();
184
+ wait(PACE.fade + gap, id, () => tick(id));
185
+ } catch {
186
+ if (id !== run) return;
187
+ cancel();
188
+ error = "narration_failed";
189
+ set("paused");
190
+ }
191
+ return;
192
+ }
193
+ speech = null;
194
+ const hold = lineHold(text, step.hold);
159
195
  last = { ...cur };
160
196
  deps.say(text, fmt(cur), hold);
161
197
  cur = { step: cur.step, line: cur.line + 1 };
162
198
  deps.changed();
163
- const gap = lastLine ? PACE.afterStep : PACE.betweenLines;
164
199
  wait(hold * 1000 + PACE.fade + gap, id, () => tick(id));
165
200
  }
166
201
 
@@ -172,6 +207,7 @@ export function createPlayer(deps: {
172
207
  load(next: Script) {
173
208
  const playing = state === "playing";
174
209
  cancel();
210
+ error = undefined;
175
211
  script = next;
176
212
  if (cur.step > next.steps.length) cur = { step: next.steps.length, line: 0 };
177
213
  normalize();
@@ -189,6 +225,7 @@ export function createPlayer(deps: {
189
225
  if (state === "playing") return null;
190
226
  if (state === "done") return "done";
191
227
  cancel();
228
+ error = undefined;
192
229
  entered = false; // bring the screen back to the current step first
193
230
  set("playing");
194
231
  tick(run);
@@ -201,6 +238,15 @@ export function createPlayer(deps: {
201
238
  set("paused");
202
239
  },
203
240
 
241
+ // Paused by an interruption (raised hand): the next resume replays the
242
+ // last shown line, since the viewer may have missed it.
243
+ rewind() {
244
+ if (state !== "paused" || !last) return;
245
+ cur = { ...last };
246
+ entered = false;
247
+ deps.changed();
248
+ },
249
+
204
250
  // Move to a step (and line) and show its screen. Does not play; if it was
205
251
  // playing, it keeps playing from there.
206
252
  async goto(pos: { step: number; line: number }): Promise<string | null> {
@@ -210,6 +256,7 @@ export function createPlayer(deps: {
210
256
  if (pos.line > 0 && pos.line >= step.say.length) return "no_such_line";
211
257
  const playing = state === "playing";
212
258
  cancel();
259
+ error = undefined;
213
260
  cur = { ...pos };
214
261
  entered = false;
215
262
  if (playing) {
@@ -225,8 +272,8 @@ export function createPlayer(deps: {
225
272
 
226
273
  // Presentation ended: stop where it is.
227
274
  stop() {
275
+ cancel();
228
276
  if (state === "playing") {
229
- cancel();
230
277
  set("paused");
231
278
  }
232
279
  },
package/src/screencast.ts CHANGED
@@ -1,4 +1,5 @@
1
1
  import { runJson } from "./run.ts";
2
+ import { AUDIO_BINDING, PAGE_AUDIO_JS, parsePcm } from "./page-audio.ts";
2
3
 
3
4
  // terminal-browser screen relay: CDP Page.startScreencast -> SSE (/screen).
4
5
  //
@@ -6,6 +7,10 @@ import { runJson } from "./run.ts";
6
7
  // follows its active tab, and streams JPEG frames only while at least one
7
8
  // viewer is connected. input() forwards whitelisted CDP Input.* commands
8
9
  // from the viewer to the page (user control).
10
+ //
11
+ // Page audio (GET /screen/audio, SSE): while at least one viewer listens, the
12
+ // tab's audio is rerouted into PCM chunks for those viewers and muted locally
13
+ // (see page-audio.ts). The last listener leaving restores local playback.
9
14
 
10
15
  type Browser = {
11
16
  key: string;
@@ -24,10 +29,12 @@ export type ScreenMeta = {
24
29
  url?: string;
25
30
  title?: string;
26
31
  agentControlled?: boolean;
32
+ audioListeners?: number; // status only
27
33
  };
28
34
 
29
35
  export function createScreencast(opts: { browserKey?: string; herdrTab?: () => Promise<string | undefined> }) {
30
36
  const viewers = new Set<ReadableStreamDefaultController>();
37
+ const listeners = new Set<ReadableStreamDefaultController>(); // page audio
31
38
  const enc = new TextEncoder();
32
39
 
33
40
  let meta: ScreenMeta = { connected: false };
@@ -39,6 +46,7 @@ export function createScreencast(opts: { browserKey?: string; herdrTab?: () => P
39
46
  let wsTarget = "";
40
47
  let msgId = 0;
41
48
  let timer: Timer | null = null;
49
+ let audioOn = false; // audio hook installed on the current CDP session
42
50
  const pending = new Map<number, (result: any) => void>();
43
51
 
44
52
  const emit = (c: ReadableStreamDefaultController, event: string, data: unknown) => {
@@ -46,9 +54,12 @@ export function createScreencast(opts: { browserKey?: string; herdrTab?: () => P
46
54
  c.enqueue(enc.encode(`event: ${event}\ndata: ${JSON.stringify(data)}\n\n`));
47
55
  } catch {
48
56
  viewers.delete(c);
57
+ listeners.delete(c);
49
58
  }
50
59
  };
51
60
  const broadcast = (event: string, data: unknown) => viewers.forEach((c) => emit(c, event, data));
61
+ const AUDIO_OFF = "window.__agelloAudio?.off()";
62
+ const AUDIO_PING = "window.__agelloAudio?.ping()";
52
63
 
53
64
  function setMeta(next: ScreenMeta) {
54
65
  if (JSON.stringify(next) === JSON.stringify(meta)) return;
@@ -78,6 +89,10 @@ export function createScreencast(opts: { browserKey?: string; herdrTab?: () => P
78
89
  }
79
90
 
80
91
  function disconnect() {
92
+ // the old tab plays locally again (its own 4s timeout covers a lost message)
93
+ if (audioOn && ws?.readyState === WebSocket.OPEN)
94
+ ws.send(JSON.stringify({ id: ++msgId, method: "Runtime.evaluate", params: { expression: AUDIO_OFF } }));
95
+ audioOn = false;
81
96
  ws?.close();
82
97
  ws = null;
83
98
  wsTarget = "";
@@ -153,6 +168,7 @@ export function createScreencast(opts: { browserKey?: string; herdrTab?: () => P
153
168
  call("Page.enable");
154
169
  call("Page.startScreencast", { format: "jpeg", quality: 70, maxWidth: 1600, maxHeight: 1600 });
155
170
  if (clip) capture();
171
+ if (listeners.size) audioStart();
156
172
  };
157
173
  sock.onmessage = (ev) => {
158
174
  const msg = JSON.parse(String(ev.data));
@@ -162,6 +178,11 @@ export function createScreencast(opts: { browserKey?: string; herdrTab?: () => P
162
178
  resolve(msg.result ?? null);
163
179
  return;
164
180
  }
181
+ if (msg.method === "Runtime.bindingCalled" && msg.params?.name === AUDIO_BINDING) {
182
+ const chunk = listeners.size ? parsePcm(String(msg.params.payload ?? "")) : null;
183
+ if (chunk) listeners.forEach((c) => emit(c, "pcm", chunk));
184
+ return;
185
+ }
165
186
  if (msg.method !== "Page.screencastFrame") return;
166
187
  const { data, metadata, sessionId } = msg.params;
167
188
  call("Page.screencastFrameAck", { sessionId });
@@ -177,6 +198,25 @@ export function createScreencast(opts: { browserKey?: string; herdrTab?: () => P
177
198
  };
178
199
  }
179
200
 
201
+ // Install the page hook (new documents too) and switch the tab to remote.
202
+ function audioStart() {
203
+ const sock = ws;
204
+ if (audioOn || !sock || sock.readyState !== WebSocket.OPEN) return;
205
+ audioOn = true;
206
+ const call = (method: string, params: object = {}) =>
207
+ sock.send(JSON.stringify({ id: ++msgId, method, params }));
208
+ call("Runtime.enable");
209
+ call("Runtime.addBinding", { name: AUDIO_BINDING });
210
+ call("Page.addScriptToEvaluateOnNewDocument", { source: PAGE_AUDIO_JS });
211
+ call("Runtime.evaluate", { expression: PAGE_AUDIO_JS });
212
+ call("Runtime.evaluate", { expression: AUDIO_PING });
213
+ }
214
+ function audioPing() {
215
+ if (!listeners.size || !ws || ws.readyState !== WebSocket.OPEN) return;
216
+ if (!audioOn) return audioStart();
217
+ ws.send(JSON.stringify({ id: ++msgId, method: "Runtime.evaluate", params: { expression: AUDIO_PING } }));
218
+ }
219
+
180
220
  let syncing = false; // skip a tick while the previous sync is still running
181
221
  async function sync() {
182
222
  if (syncing) return;
@@ -200,6 +240,7 @@ export function createScreencast(opts: { browserKey?: string; herdrTab?: () => P
200
240
  }
201
241
  const url = `ws://127.0.0.1:${b.cdpPort}/devtools/page/${tab.targetId}`;
202
242
  if (url !== wsTarget || !ws) connect(b.cdpPort, tab.targetId);
243
+ else audioPing();
203
244
  setMeta({
204
245
  connected: true,
205
246
  browser: b.key,
@@ -248,6 +289,31 @@ export function createScreencast(opts: { browserKey?: string; herdrTab?: () => P
248
289
  });
249
290
  }
250
291
 
292
+ // SSE of page audio for one listener. Only meaningful alongside /screen
293
+ // (the relay runs while screen viewers are connected).
294
+ function handleAudio(): Response {
295
+ let ctrl!: ReadableStreamDefaultController;
296
+ let ping: Timer;
297
+ const stream = new ReadableStream({
298
+ start(c) {
299
+ ctrl = c;
300
+ listeners.add(c);
301
+ emit(c, "ready", {});
302
+ ping = setInterval(() => emit(c, "ping", {}), 15000);
303
+ audioPing();
304
+ },
305
+ cancel() {
306
+ clearInterval(ping);
307
+ listeners.delete(ctrl);
308
+ if (listeners.size === 0 && audioOn && ws?.readyState === WebSocket.OPEN)
309
+ ws.send(JSON.stringify({ id: ++msgId, method: "Runtime.evaluate", params: { expression: AUDIO_OFF } }));
310
+ },
311
+ });
312
+ return new Response(stream, {
313
+ headers: { "Content-Type": "text/event-stream", "Cache-Control": "no-cache", Connection: "keep-alive" },
314
+ });
315
+ }
316
+
251
317
  const INPUT_METHODS = new Set(["Input.dispatchMouseEvent", "Input.dispatchKeyEvent", "Input.insertText"]);
252
318
  function input(raw: string) {
253
319
  let m: any;
@@ -263,7 +329,7 @@ export function createScreencast(opts: { browserKey?: string; herdrTab?: () => P
263
329
 
264
330
  // Which browser the relay uses (or would use), for status reporting.
265
331
  async function describe(): Promise<ScreenMeta> {
266
- if (meta.connected) return meta;
332
+ if (meta.connected) return { ...meta, audioListeners: listeners.size };
267
333
  const { browser, reason } = await pickBrowser();
268
334
  return browser ? { connected: false, browser: browser.key } : { connected: false, reason };
269
335
  }
@@ -287,7 +353,7 @@ export function createScreencast(opts: { browserKey?: string; herdrTab?: () => P
287
353
  return r && !r.exceptionDetails ? (r.result?.value ?? null) : null;
288
354
  }
289
355
 
290
- return { handle, input, describe, setClip, evaluate, inspect };
356
+ return { handle, handleAudio, input, describe, setClip, evaluate, inspect };
291
357
  }
292
358
 
293
359
  export type InspectQuery = { x?: number; y?: number; selector?: string; full?: boolean };