@yieldfabric/terminal 1.3.1 → 1.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +30 -1
- package/dist/index.d.cts +21 -1
- package/dist/index.d.ts +21 -1
- package/dist/index.js +30 -1
- package/package.json +1 -1
package/dist/index.cjs
CHANGED
|
@@ -8068,6 +8068,10 @@ var VoiceSession = class {
|
|
|
8068
8068
|
this.finishTimer = null;
|
|
8069
8069
|
this.muted = false;
|
|
8070
8070
|
this.blobUrl = null;
|
|
8071
|
+
/** What the person's page shows, as the host describes it, and what the
|
|
8072
|
+
* relay was last told (it keeps one note of it, replaced on change). */
|
|
8073
|
+
this.screen = "";
|
|
8074
|
+
this.screenSent = null;
|
|
8071
8075
|
}
|
|
8072
8076
|
/** Call from the tap itself: iOS only lets audio start inside a gesture,
|
|
8073
8077
|
* so the AudioContext is created before the first `await`. */
|
|
@@ -8168,6 +8172,23 @@ var VoiceSession = class {
|
|
|
8168
8172
|
stop() {
|
|
8169
8173
|
this.end("stopped");
|
|
8170
8174
|
}
|
|
8175
|
+
/**
|
|
8176
|
+
* What the person's page shows, in a few plain lines: the voice keeps it
|
|
8177
|
+
* as one note, replaced as it changes, so "this photo", "this document"
|
|
8178
|
+
* or "this claim" means what they are looking at and it never asks which.
|
|
8179
|
+
* Call it whenever the page changes; it is sent once the session is
|
|
8180
|
+
* ready, and only when it differs. Empty clears it. An agents deployment
|
|
8181
|
+
* older than the `screen` message answers with a warning and nothing else.
|
|
8182
|
+
*/
|
|
8183
|
+
setScreen(text) {
|
|
8184
|
+
this.screen = text.trim();
|
|
8185
|
+
this.sendScreen();
|
|
8186
|
+
}
|
|
8187
|
+
sendScreen() {
|
|
8188
|
+
if (!this.ready || this.ended || this.screen === (this.screenSent ?? "")) return;
|
|
8189
|
+
this.send({ type: "screen", text: this.screen });
|
|
8190
|
+
this.screenSent = this.screen;
|
|
8191
|
+
}
|
|
8171
8192
|
/**
|
|
8172
8193
|
* Mute the microphone, or not. The track goes silent rather than the
|
|
8173
8194
|
* stream stopping: the service keeps hearing (silence), so whatever the
|
|
@@ -8224,6 +8245,7 @@ var VoiceSession = class {
|
|
|
8224
8245
|
case "session.ready":
|
|
8225
8246
|
this.ready = true;
|
|
8226
8247
|
this.setPhase("listening");
|
|
8248
|
+
this.sendScreen();
|
|
8227
8249
|
break;
|
|
8228
8250
|
case "input_audio_buffer.speech_started":
|
|
8229
8251
|
this.cancelFinish();
|
|
@@ -8410,6 +8432,7 @@ function useVoiceSession(host, connection) {
|
|
|
8410
8432
|
hostRef.current = host;
|
|
8411
8433
|
const connectionRef = React15.useRef(connection);
|
|
8412
8434
|
connectionRef.current = connection;
|
|
8435
|
+
const screen = React15.useRef("");
|
|
8413
8436
|
const { baseUrl } = connection;
|
|
8414
8437
|
React15.useEffect(() => {
|
|
8415
8438
|
const controller = new AbortController();
|
|
@@ -8450,9 +8473,14 @@ function useVoiceSession(host, connection) {
|
|
|
8450
8473
|
{ ...connectionRef.current }
|
|
8451
8474
|
);
|
|
8452
8475
|
session.current = s;
|
|
8476
|
+
s.setScreen(screen.current);
|
|
8453
8477
|
void s.start();
|
|
8454
8478
|
}, []);
|
|
8455
8479
|
const stop = React15.useCallback(() => session.current?.stop(), []);
|
|
8480
|
+
const setScreen = React15.useCallback((text) => {
|
|
8481
|
+
screen.current = text;
|
|
8482
|
+
session.current?.setScreen(text);
|
|
8483
|
+
}, []);
|
|
8456
8484
|
const toggle = React15.useCallback(() => session.current ? stop() : start(), [start, stop]);
|
|
8457
8485
|
const dismiss = React15.useCallback(() => setEnded(null), []);
|
|
8458
8486
|
const toggleMute = React15.useCallback(() => {
|
|
@@ -8484,7 +8512,8 @@ function useVoiceSession(host, connection) {
|
|
|
8484
8512
|
start,
|
|
8485
8513
|
stop,
|
|
8486
8514
|
toggle,
|
|
8487
|
-
dismiss
|
|
8515
|
+
dismiss,
|
|
8516
|
+
setScreen
|
|
8488
8517
|
};
|
|
8489
8518
|
}
|
|
8490
8519
|
function voiceEndedMessage(ended, name = "the assistant") {
|
package/dist/index.d.cts
CHANGED
|
@@ -1365,7 +1365,9 @@ declare function encodePersona(persona: VoicePersona | undefined): string | null
|
|
|
1365
1365
|
* Wire (agents `src/voice/protocol.rs`): binary frames are PCM16 mono
|
|
1366
1366
|
* 24 kHz both ways; text frames are JSON events. The browser may send only
|
|
1367
1367
|
* `tool_result`, `progress` (what the host has done and found while it
|
|
1368
|
-
* works, told as paced updates), `
|
|
1368
|
+
* works, told as paced updates), `screen` (what the person's page shows,
|
|
1369
|
+
* so "this photo" or "this document" means what they are looking at),
|
|
1370
|
+
* `cancel_response` and `clear_input`.
|
|
1369
1371
|
*/
|
|
1370
1372
|
|
|
1371
1373
|
/** How to reach agents, as whom, and who is talking. */
|
|
@@ -1488,11 +1490,25 @@ declare class VoiceSession {
|
|
|
1488
1490
|
private finishTimer;
|
|
1489
1491
|
private muted;
|
|
1490
1492
|
private blobUrl;
|
|
1493
|
+
/** What the person's page shows, as the host describes it, and what the
|
|
1494
|
+
* relay was last told (it keeps one note of it, replaced on change). */
|
|
1495
|
+
private screen;
|
|
1496
|
+
private screenSent;
|
|
1491
1497
|
constructor(h: VoiceHandlers, connection: VoiceConnection);
|
|
1492
1498
|
/** Call from the tap itself: iOS only lets audio start inside a gesture,
|
|
1493
1499
|
* so the AudioContext is created before the first `await`. */
|
|
1494
1500
|
start(): Promise<void>;
|
|
1495
1501
|
stop(): void;
|
|
1502
|
+
/**
|
|
1503
|
+
* What the person's page shows, in a few plain lines: the voice keeps it
|
|
1504
|
+
* as one note, replaced as it changes, so "this photo", "this document"
|
|
1505
|
+
* or "this claim" means what they are looking at and it never asks which.
|
|
1506
|
+
* Call it whenever the page changes; it is sent once the session is
|
|
1507
|
+
* ready, and only when it differs. Empty clears it. An agents deployment
|
|
1508
|
+
* older than the `screen` message answers with a warning and nothing else.
|
|
1509
|
+
*/
|
|
1510
|
+
setScreen(text: string): void;
|
|
1511
|
+
private sendScreen;
|
|
1496
1512
|
/**
|
|
1497
1513
|
* Mute the microphone, or not. The track goes silent rather than the
|
|
1498
1514
|
* stream stopping: the service keeps hearing (silence), so whatever the
|
|
@@ -1552,6 +1568,10 @@ interface VoiceSessionState {
|
|
|
1552
1568
|
toggle: () => void;
|
|
1553
1569
|
/** Clear `ended`. */
|
|
1554
1570
|
dismiss: () => void;
|
|
1571
|
+
/** What the person's page shows, in a few plain lines, so "this photo"
|
|
1572
|
+
* or "this document" means what they are looking at. Kept between
|
|
1573
|
+
* sessions: a session that starts later is told it once it is ready. */
|
|
1574
|
+
setScreen: (text: string) => void;
|
|
1555
1575
|
}
|
|
1556
1576
|
/**
|
|
1557
1577
|
* One live voice session for a conversation surface.
|
package/dist/index.d.ts
CHANGED
|
@@ -1365,7 +1365,9 @@ declare function encodePersona(persona: VoicePersona | undefined): string | null
|
|
|
1365
1365
|
* Wire (agents `src/voice/protocol.rs`): binary frames are PCM16 mono
|
|
1366
1366
|
* 24 kHz both ways; text frames are JSON events. The browser may send only
|
|
1367
1367
|
* `tool_result`, `progress` (what the host has done and found while it
|
|
1368
|
-
* works, told as paced updates), `
|
|
1368
|
+
* works, told as paced updates), `screen` (what the person's page shows,
|
|
1369
|
+
* so "this photo" or "this document" means what they are looking at),
|
|
1370
|
+
* `cancel_response` and `clear_input`.
|
|
1369
1371
|
*/
|
|
1370
1372
|
|
|
1371
1373
|
/** How to reach agents, as whom, and who is talking. */
|
|
@@ -1488,11 +1490,25 @@ declare class VoiceSession {
|
|
|
1488
1490
|
private finishTimer;
|
|
1489
1491
|
private muted;
|
|
1490
1492
|
private blobUrl;
|
|
1493
|
+
/** What the person's page shows, as the host describes it, and what the
|
|
1494
|
+
* relay was last told (it keeps one note of it, replaced on change). */
|
|
1495
|
+
private screen;
|
|
1496
|
+
private screenSent;
|
|
1491
1497
|
constructor(h: VoiceHandlers, connection: VoiceConnection);
|
|
1492
1498
|
/** Call from the tap itself: iOS only lets audio start inside a gesture,
|
|
1493
1499
|
* so the AudioContext is created before the first `await`. */
|
|
1494
1500
|
start(): Promise<void>;
|
|
1495
1501
|
stop(): void;
|
|
1502
|
+
/**
|
|
1503
|
+
* What the person's page shows, in a few plain lines: the voice keeps it
|
|
1504
|
+
* as one note, replaced as it changes, so "this photo", "this document"
|
|
1505
|
+
* or "this claim" means what they are looking at and it never asks which.
|
|
1506
|
+
* Call it whenever the page changes; it is sent once the session is
|
|
1507
|
+
* ready, and only when it differs. Empty clears it. An agents deployment
|
|
1508
|
+
* older than the `screen` message answers with a warning and nothing else.
|
|
1509
|
+
*/
|
|
1510
|
+
setScreen(text: string): void;
|
|
1511
|
+
private sendScreen;
|
|
1496
1512
|
/**
|
|
1497
1513
|
* Mute the microphone, or not. The track goes silent rather than the
|
|
1498
1514
|
* stream stopping: the service keeps hearing (silence), so whatever the
|
|
@@ -1552,6 +1568,10 @@ interface VoiceSessionState {
|
|
|
1552
1568
|
toggle: () => void;
|
|
1553
1569
|
/** Clear `ended`. */
|
|
1554
1570
|
dismiss: () => void;
|
|
1571
|
+
/** What the person's page shows, in a few plain lines, so "this photo"
|
|
1572
|
+
* or "this document" means what they are looking at. Kept between
|
|
1573
|
+
* sessions: a session that starts later is told it once it is ready. */
|
|
1574
|
+
setScreen: (text: string) => void;
|
|
1555
1575
|
}
|
|
1556
1576
|
/**
|
|
1557
1577
|
* One live voice session for a conversation surface.
|
package/dist/index.js
CHANGED
|
@@ -8059,6 +8059,10 @@ var VoiceSession = class {
|
|
|
8059
8059
|
this.finishTimer = null;
|
|
8060
8060
|
this.muted = false;
|
|
8061
8061
|
this.blobUrl = null;
|
|
8062
|
+
/** What the person's page shows, as the host describes it, and what the
|
|
8063
|
+
* relay was last told (it keeps one note of it, replaced on change). */
|
|
8064
|
+
this.screen = "";
|
|
8065
|
+
this.screenSent = null;
|
|
8062
8066
|
}
|
|
8063
8067
|
/** Call from the tap itself: iOS only lets audio start inside a gesture,
|
|
8064
8068
|
* so the AudioContext is created before the first `await`. */
|
|
@@ -8159,6 +8163,23 @@ var VoiceSession = class {
|
|
|
8159
8163
|
stop() {
|
|
8160
8164
|
this.end("stopped");
|
|
8161
8165
|
}
|
|
8166
|
+
/**
|
|
8167
|
+
* What the person's page shows, in a few plain lines: the voice keeps it
|
|
8168
|
+
* as one note, replaced as it changes, so "this photo", "this document"
|
|
8169
|
+
* or "this claim" means what they are looking at and it never asks which.
|
|
8170
|
+
* Call it whenever the page changes; it is sent once the session is
|
|
8171
|
+
* ready, and only when it differs. Empty clears it. An agents deployment
|
|
8172
|
+
* older than the `screen` message answers with a warning and nothing else.
|
|
8173
|
+
*/
|
|
8174
|
+
setScreen(text) {
|
|
8175
|
+
this.screen = text.trim();
|
|
8176
|
+
this.sendScreen();
|
|
8177
|
+
}
|
|
8178
|
+
sendScreen() {
|
|
8179
|
+
if (!this.ready || this.ended || this.screen === (this.screenSent ?? "")) return;
|
|
8180
|
+
this.send({ type: "screen", text: this.screen });
|
|
8181
|
+
this.screenSent = this.screen;
|
|
8182
|
+
}
|
|
8162
8183
|
/**
|
|
8163
8184
|
* Mute the microphone, or not. The track goes silent rather than the
|
|
8164
8185
|
* stream stopping: the service keeps hearing (silence), so whatever the
|
|
@@ -8215,6 +8236,7 @@ var VoiceSession = class {
|
|
|
8215
8236
|
case "session.ready":
|
|
8216
8237
|
this.ready = true;
|
|
8217
8238
|
this.setPhase("listening");
|
|
8239
|
+
this.sendScreen();
|
|
8218
8240
|
break;
|
|
8219
8241
|
case "input_audio_buffer.speech_started":
|
|
8220
8242
|
this.cancelFinish();
|
|
@@ -8401,6 +8423,7 @@ function useVoiceSession(host, connection) {
|
|
|
8401
8423
|
hostRef.current = host;
|
|
8402
8424
|
const connectionRef = useRef(connection);
|
|
8403
8425
|
connectionRef.current = connection;
|
|
8426
|
+
const screen = useRef("");
|
|
8404
8427
|
const { baseUrl } = connection;
|
|
8405
8428
|
useEffect(() => {
|
|
8406
8429
|
const controller = new AbortController();
|
|
@@ -8441,9 +8464,14 @@ function useVoiceSession(host, connection) {
|
|
|
8441
8464
|
{ ...connectionRef.current }
|
|
8442
8465
|
);
|
|
8443
8466
|
session.current = s;
|
|
8467
|
+
s.setScreen(screen.current);
|
|
8444
8468
|
void s.start();
|
|
8445
8469
|
}, []);
|
|
8446
8470
|
const stop = useCallback(() => session.current?.stop(), []);
|
|
8471
|
+
const setScreen = useCallback((text) => {
|
|
8472
|
+
screen.current = text;
|
|
8473
|
+
session.current?.setScreen(text);
|
|
8474
|
+
}, []);
|
|
8447
8475
|
const toggle = useCallback(() => session.current ? stop() : start(), [start, stop]);
|
|
8448
8476
|
const dismiss = useCallback(() => setEnded(null), []);
|
|
8449
8477
|
const toggleMute = useCallback(() => {
|
|
@@ -8475,7 +8503,8 @@ function useVoiceSession(host, connection) {
|
|
|
8475
8503
|
start,
|
|
8476
8504
|
stop,
|
|
8477
8505
|
toggle,
|
|
8478
|
-
dismiss
|
|
8506
|
+
dismiss,
|
|
8507
|
+
setScreen
|
|
8479
8508
|
};
|
|
8480
8509
|
}
|
|
8481
8510
|
function voiceEndedMessage(ended, name = "the assistant") {
|