@memberjunction/ai-agents 5.40.2 → 5.42.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +53 -0
- package/dist/AgentRunner.d.ts +5 -2
- package/dist/AgentRunner.d.ts.map +1 -1
- package/dist/AgentRunner.js +14 -4
- package/dist/AgentRunner.js.map +1 -1
- package/dist/MemoryWriteManager.d.ts +188 -0
- package/dist/MemoryWriteManager.d.ts.map +1 -0
- package/dist/MemoryWriteManager.js +299 -0
- package/dist/MemoryWriteManager.js.map +1 -0
- package/dist/agent-context-injector.d.ts +29 -0
- package/dist/agent-context-injector.d.ts.map +1 -1
- package/dist/agent-context-injector.js +90 -32
- package/dist/agent-context-injector.js.map +1 -1
- package/dist/agent-memory-context-builder.d.ts +100 -0
- package/dist/agent-memory-context-builder.d.ts.map +1 -0
- package/dist/agent-memory-context-builder.js +172 -0
- package/dist/agent-memory-context-builder.js.map +1 -0
- package/dist/agent-types/index.d.ts +1 -0
- package/dist/agent-types/index.d.ts.map +1 -1
- package/dist/agent-types/index.js +1 -0
- package/dist/agent-types/index.js.map +1 -1
- package/dist/agent-types/loop-agent-response-type.d.ts +12 -1
- package/dist/agent-types/loop-agent-response-type.d.ts.map +1 -1
- package/dist/agent-types/loop-agent-response-type.js.map +1 -1
- package/dist/agent-types/loop-agent-type.d.ts.map +1 -1
- package/dist/agent-types/loop-agent-type.js +4 -0
- package/dist/agent-types/loop-agent-type.js.map +1 -1
- package/dist/agent-types/realtime-agent-type.d.ts +146 -0
- package/dist/agent-types/realtime-agent-type.d.ts.map +1 -0
- package/dist/agent-types/realtime-agent-type.js +176 -0
- package/dist/agent-types/realtime-agent-type.js.map +1 -0
- package/dist/base-agent.d.ts +386 -39
- package/dist/base-agent.d.ts.map +1 -1
- package/dist/base-agent.js +1121 -261
- package/dist/base-agent.js.map +1 -1
- package/dist/index.d.ts +13 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +17 -0
- package/dist/index.js.map +1 -1
- package/dist/memory-manager-agent.d.ts +99 -4
- package/dist/memory-manager-agent.d.ts.map +1 -1
- package/dist/memory-manager-agent.js +349 -117
- package/dist/memory-manager-agent.js.map +1 -1
- package/dist/realtime/bridge-realtime-session-factory.d.ts +111 -0
- package/dist/realtime/bridge-realtime-session-factory.d.ts.map +1 -0
- package/dist/realtime/bridge-realtime-session-factory.js +163 -0
- package/dist/realtime/bridge-realtime-session-factory.js.map +1 -0
- package/dist/realtime/bridge-room-transcript-sink.d.ts +58 -0
- package/dist/realtime/bridge-room-transcript-sink.d.ts.map +1 -0
- package/dist/realtime/bridge-room-transcript-sink.js +127 -0
- package/dist/realtime/bridge-room-transcript-sink.js.map +1 -0
- package/dist/realtime/meeting-controls-channel-server.d.ts +198 -0
- package/dist/realtime/meeting-controls-channel-server.d.ts.map +1 -0
- package/dist/realtime/meeting-controls-channel-server.js +319 -0
- package/dist/realtime/meeting-controls-channel-server.js.map +1 -0
- package/dist/realtime/meeting-controls-state.d.ts +191 -0
- package/dist/realtime/meeting-controls-state.d.ts.map +1 -0
- package/dist/realtime/meeting-controls-state.js +219 -0
- package/dist/realtime/meeting-controls-state.js.map +1 -0
- package/dist/realtime/realtime-channel-server-host.d.ts +166 -0
- package/dist/realtime/realtime-channel-server-host.d.ts.map +1 -0
- package/dist/realtime/realtime-channel-server-host.js +378 -0
- package/dist/realtime/realtime-channel-server-host.js.map +1 -0
- package/dist/realtime/realtime-client-session-service.d.ts +1026 -0
- package/dist/realtime/realtime-client-session-service.d.ts.map +1 -0
- package/dist/realtime/realtime-client-session-service.js +1607 -0
- package/dist/realtime/realtime-client-session-service.js.map +1 -0
- package/dist/realtime/realtime-coagent-config.d.ts +258 -0
- package/dist/realtime/realtime-coagent-config.d.ts.map +1 -0
- package/dist/realtime/realtime-coagent-config.js +408 -0
- package/dist/realtime/realtime-coagent-config.js.map +1 -0
- package/dist/realtime/realtime-narration.d.ts +67 -0
- package/dist/realtime/realtime-narration.d.ts.map +1 -0
- package/dist/realtime/realtime-narration.js +127 -0
- package/dist/realtime/realtime-narration.js.map +1 -0
- package/dist/realtime/realtime-session-runner.d.ts +383 -0
- package/dist/realtime/realtime-session-runner.d.ts.map +1 -0
- package/dist/realtime/realtime-session-runner.js +532 -0
- package/dist/realtime/realtime-session-runner.js.map +1 -0
- package/dist/realtime/realtime-tool-broker.d.ts +294 -0
- package/dist/realtime/realtime-tool-broker.d.ts.map +1 -0
- package/dist/realtime/realtime-tool-broker.js +206 -0
- package/dist/realtime/realtime-tool-broker.js.map +1 -0
- package/dist/realtime/whiteboard-channel-server.d.ts +50 -0
- package/dist/realtime/whiteboard-channel-server.d.ts.map +1 -0
- package/dist/realtime/whiteboard-channel-server.js +85 -0
- package/dist/realtime/whiteboard-channel-server.js.map +1 -0
- package/package.json +17 -17
|
@@ -0,0 +1,1607 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Server-agnostic preparer + tool relay for a CLIENT-DIRECT realtime session
|
|
3
|
+
* (the Realtime Co-Agent dual-topology design).
|
|
4
|
+
*
|
|
5
|
+
* In the client-direct topology the browser opens its OWN provider socket (e.g. WebRTC) using a
|
|
6
|
+
* server-minted ephemeral token, but the **server** still owns the system prompt and tool set and
|
|
7
|
+
* **executes** every tool call the browser relays back. This service is the server-side half of
|
|
8
|
+
* that contract. It does two things:
|
|
9
|
+
*
|
|
10
|
+
* 1. {@link RealtimeClientSessionService.PrepareClientSession} — resolves the Realtime model,
|
|
11
|
+
* assembles the companion system prompt (co-agent prompt + target identity + history + memory),
|
|
12
|
+
* builds the stable, target-independent tool set (always including `invoke-target-agent`), and
|
|
13
|
+
* asks the model to mint a {@link ClientRealtimeSessionConfig} (ephemeral token + provider
|
|
14
|
+
* session config) the browser applies verbatim.
|
|
15
|
+
* 2. {@link RealtimeClientSessionService.ExecuteRelayedTool} — executes a single tool call the
|
|
16
|
+
* browser relayed, routing it through the shared {@link RealtimeToolBroker} so the result is
|
|
17
|
+
* byte-for-byte identical to the server-bridged path. `invoke-target-agent` delegates to the
|
|
18
|
+
* target agent via {@link AgentRunner.RunAgent}; every other tool returns a structured
|
|
19
|
+
* "not available" result for now (action wiring is a later phase).
|
|
20
|
+
*
|
|
21
|
+
* **Why this duplicates BaseAgent.** The private helpers in `BaseAgent.executeRealtimeSession`
|
|
22
|
+
* (model resolution, companion-prompt assembly, target-agent resolution, delegation) are the
|
|
23
|
+
* server-bridged equivalents of the logic here, but they are `private` to `BaseAgent` and bound to
|
|
24
|
+
* an in-flight `AIAgentRun`/`StartSession` lifecycle. This service mirrors that logic for the
|
|
25
|
+
* client-direct topology, which has no server-side session loop. **A future refactor should extract
|
|
26
|
+
* a shared `RealtimeSessionPreparer`** that both `BaseAgent` and this service consume, eliminating
|
|
27
|
+
* the duplication. Until then, keep the two in sync intentionally.
|
|
28
|
+
*
|
|
29
|
+
* @module @memberjunction/ai-agents
|
|
30
|
+
* @author MemberJunction.com
|
|
31
|
+
*/
|
|
32
|
+
import { LogError, LogStatus, RunView } from '@memberjunction/core';
|
|
33
|
+
import { MJGlobal, UUIDsEqual } from '@memberjunction/global';
|
|
34
|
+
import { BaseRealtimeModel, GetAIAPIKey } from '@memberjunction/ai';
|
|
35
|
+
import { AIEngine } from '@memberjunction/aiengine';
|
|
36
|
+
import { AgentMemoryContextBuilder } from '../agent-memory-context-builder.js';
|
|
37
|
+
import { AgentRunner } from '../AgentRunner.js';
|
|
38
|
+
import { RealtimeToolBroker, INVOKE_TARGET_AGENT_TOOL_NAME, BuildRealtimeAgentFraming } from './realtime-tool-broker.js';
|
|
39
|
+
import { NARRATION_PROMPT_NAME, LEGACY_NARRATION_PROMPT_NAME, ResolveNarrationInstructionsTemplate } from './realtime-narration.js';
|
|
40
|
+
import { BuildVoiceMannerSection, DeepMergeConfigs, GetNarrationPaceMs, GetProviderVoiceSettings, ResolveEffectiveRealtimeConfig } from './realtime-coagent-config.js';
|
|
41
|
+
/**
|
|
42
|
+
* Server-agnostic service that prepares a client-direct realtime session and executes the tool
|
|
43
|
+
* calls the browser relays back. Constructed per-request (a normal injectable service — NOT a
|
|
44
|
+
* singleton) so the {@link UserInfo} and {@link IMetadataProvider} are always request-scoped.
|
|
45
|
+
*
|
|
46
|
+
* Every public method takes the `contextUser` and `provider` explicitly — this service never
|
|
47
|
+
* reaches for the global default provider, so it is safe in multi-provider/multi-tenant servers.
|
|
48
|
+
*/
|
|
49
|
+
export class RealtimeClientSessionService {
|
|
50
|
+
constructor() {
|
|
51
|
+
/**
|
|
52
|
+
* IN-FLIGHT DELEGATION REGISTRY — the server half of the client-direct CANCEL channel.
|
|
53
|
+
*
|
|
54
|
+
* Every relayed tool call registers an {@link AbortController} under
|
|
55
|
+
* `(agentSessionID, callID)` for the duration of {@link ExecuteRelayedTool}; the
|
|
56
|
+
* `CancelRealtimeSessionTool` mutation aborts entries via
|
|
57
|
+
* {@link CancelInFlightDelegations} so an explicit user cancel (the overlay's per-card ✕)
|
|
58
|
+
* kills the delegated target-agent run mid-flight. Entries are removed on completion
|
|
59
|
+
* (success, failure, or abort), so the registry only ever holds truly in-flight calls.
|
|
60
|
+
*
|
|
61
|
+
* Keys are normalized (trimmed, lowercased) so SQL Server's uppercase UUIDs and
|
|
62
|
+
* PostgreSQL's lowercase UUIDs address the same entry.
|
|
63
|
+
*
|
|
64
|
+
* NOTE: this registry is per-service-instance state (the resolver holds ONE shared service
|
|
65
|
+
* per server process), not per-request state — it deliberately spans requests so the cancel
|
|
66
|
+
* mutation can reach the execute mutation's in-flight controller.
|
|
67
|
+
*/
|
|
68
|
+
this.inFlightDelegations = new Map();
|
|
69
|
+
/**
|
|
70
|
+
* Per-`AIPromptRun` write serialization. Both the high-frequency usage checkpoint
|
|
71
|
+
* ({@link AccumulatePromptRunUsage}) and the per-turn message append ({@link AppendPromptRunMessage})
|
|
72
|
+
* do load-modify-save on the SAME run row. Run concurrently, the frequent usage save would rewrite the
|
|
73
|
+
* whole row — including the STALE `Messages` it loaded — and perpetually clobber freshly-appended turns
|
|
74
|
+
* back to an empty snapshot (the "transcript never persists" bug). Funnelling every write for a given
|
|
75
|
+
* run through a single promise chain makes each load happen AFTER the prior save committed, so no writer
|
|
76
|
+
* overwrites another's field. Keyed by promptRunID; the entry is dropped on {@link finalizePromptRun}.
|
|
77
|
+
*/
|
|
78
|
+
this.promptRunWriteChains = new Map();
|
|
79
|
+
}
|
|
80
|
+
/**
|
|
81
|
+
* The seeded name of the `MJ: AI Prompts` row whose `TemplateText` carries the first-person
|
|
82
|
+
* progress-narration instructions (with a `{{ progressMessage }}` placeholder). Resolved at
|
|
83
|
+
* session prepare time so the browser narrates with DB-driven, product-tunable wording.
|
|
84
|
+
* Canonical value lives in `realtime-narration.ts` (shared with the server-bridged runner path).
|
|
85
|
+
*/
|
|
86
|
+
static { this.NarrationPromptName = NARRATION_PROMPT_NAME; }
|
|
87
|
+
/**
|
|
88
|
+
* DEPRECATED legacy name of the narration prompt, from before the co-agent's rename from
|
|
89
|
+
* "Voice Co-Agent" to "Realtime Co-Agent". Deployments that have not re-synced the prompt seed
|
|
90
|
+
* still carry this name, so {@link resolveNarrationInstructionsTemplate} falls back to it
|
|
91
|
+
* (with a deprecation log) when {@link RealtimeClientSessionService.NarrationPromptName} is absent.
|
|
92
|
+
*/
|
|
93
|
+
static { this.LegacyNarrationPromptName = LEGACY_NARRATION_PROMPT_NAME; }
|
|
94
|
+
/**
|
|
95
|
+
* Serializes `task` against all other writes to the same `AIPromptRun` (see {@link promptRunWriteChains}).
|
|
96
|
+
* Tasks run in call order; a failing task never breaks the chain for the next one. Returns the task's result.
|
|
97
|
+
*/
|
|
98
|
+
serializePromptRunWrite(promptRunID, task) {
|
|
99
|
+
const prior = this.promptRunWriteChains.get(promptRunID) ?? Promise.resolve();
|
|
100
|
+
const run = prior.then(task, task);
|
|
101
|
+
// Store an error-swallowing tail so one failed write doesn't reject every queued write behind it.
|
|
102
|
+
this.promptRunWriteChains.set(promptRunID, run.then(() => undefined, () => undefined));
|
|
103
|
+
return run;
|
|
104
|
+
}
|
|
105
|
+
/**
|
|
106
|
+
* Prepares a client-direct realtime session: resolves the model, assembles the companion
|
|
107
|
+
* system prompt + stable tool set, and mints the {@link ClientRealtimeSessionConfig}.
|
|
108
|
+
*
|
|
109
|
+
* Returns a failure result (never throws) when no Realtime model/key resolves or the provider
|
|
110
|
+
* cannot mint a client-direct session.
|
|
111
|
+
*
|
|
112
|
+
* @param input The co-agent/target/session inputs.
|
|
113
|
+
* @param contextUser The calling user (threaded to metadata + memory retrieval).
|
|
114
|
+
* @param provider The request-scoped metadata provider.
|
|
115
|
+
* @returns The prep result (Success + ClientConfig/SessionParams, or Success: false + ErrorMessage).
|
|
116
|
+
*/
|
|
117
|
+
async PrepareClientSession(input, contextUser, provider) {
|
|
118
|
+
// Build the canonical session params via the ONE shared producer (identity + cascade + tools +
|
|
119
|
+
// voice + memory) — then do the client-direct-specific bits: SupportsClientDirect gate, mint, obs.
|
|
120
|
+
const prep = await this.PrepareRealtimeSessionParams(input, contextUser, provider);
|
|
121
|
+
if (!prep.Success || !prep.CoAgent || !prep.Resolution || !prep.SessionParams || !prep.EffectiveConfig) {
|
|
122
|
+
return { Success: false, ErrorMessage: prep.ErrorMessage };
|
|
123
|
+
}
|
|
124
|
+
const { CoAgent: coAgent, Resolution: resolution, SessionParams: sessionParams, EffectiveConfig: effectiveConfig } = prep;
|
|
125
|
+
if (!resolution.Model.SupportsClientDirect) {
|
|
126
|
+
return {
|
|
127
|
+
Success: false,
|
|
128
|
+
ErrorMessage: `The resolved realtime model '${resolution.APIName}' does not support client-direct sessions.`
|
|
129
|
+
};
|
|
130
|
+
}
|
|
131
|
+
let clientConfig;
|
|
132
|
+
try {
|
|
133
|
+
clientConfig = await resolution.Model.CreateClientSession(sessionParams);
|
|
134
|
+
}
|
|
135
|
+
catch (error) {
|
|
136
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
137
|
+
return { Success: false, ErrorMessage: `Failed to mint client realtime session: ${message}` };
|
|
138
|
+
}
|
|
139
|
+
// Best-effort observability: create a server-side co-agent run (+ prompt run) so the voice
|
|
140
|
+
// session is visible in the agent-run timeline and delegated runs can nest under it. A
|
|
141
|
+
// failure here never fails the prepare — we just omit the ids.
|
|
142
|
+
const promptID = this.resolveCoAgentSystemPrompt(coAgent).PromptID;
|
|
143
|
+
const obs = await this.createCoAgentObservabilityRun(coAgent, promptID, resolution.ModelID, resolution.VendorID, input.UserID || contextUser?.ID, input.AgentSessionID, contextUser, provider, input.ConversationID);
|
|
144
|
+
return {
|
|
145
|
+
Success: true,
|
|
146
|
+
ClientConfig: clientConfig,
|
|
147
|
+
SessionParams: sessionParams,
|
|
148
|
+
CoAgentRunID: obs?.CoAgentRunID,
|
|
149
|
+
PromptRunID: obs?.PromptRunID,
|
|
150
|
+
CoAgentRunStepID: obs?.CoAgentRunStepID,
|
|
151
|
+
ModelID: resolution.ModelID,
|
|
152
|
+
ModelName: resolution.ModelName,
|
|
153
|
+
NarrationInstructionsTemplate: this.resolveNarrationInstructionsTemplate() ?? undefined,
|
|
154
|
+
EffectiveConfig: effectiveConfig,
|
|
155
|
+
NarrationPaceMs: GetNarrationPaceMs(effectiveConfig) ?? undefined,
|
|
156
|
+
};
|
|
157
|
+
}
|
|
158
|
+
/**
|
|
159
|
+
* Wires a **server long-lived (bridged)** realtime session onto the SAME core machinery the
|
|
160
|
+
* client-direct path uses — so a LiveKit (or future Zoom/Teams) agent does real work and is tracked
|
|
161
|
+
* identically, with **zero host-local re-implementation**. This is the Phase 2 counterpart to
|
|
162
|
+
* {@link PrepareClientSession}: the browser relays tool calls back over GraphQL to `ExecuteRelayedTool`,
|
|
163
|
+
* whereas here the server holds the live {@link IRealtimeSession} and we wire its `OnToolCall` directly to
|
|
164
|
+
* the SAME {@link ExecuteRelayedTool} (so `invoke-target-agent` runs the target via `AgentRunner`, nests
|
|
165
|
+
* under the co-agent run, supports barge-in cancel + paused-run resume — all of it, for free).
|
|
166
|
+
*
|
|
167
|
+
* Responsibilities, in order:
|
|
168
|
+
* 1. Create the co-agent observability run (+ prompt run + step) so the voice session shows up in the
|
|
169
|
+
* agent-run timeline and delegated runs nest under it (best-effort; a failure just omits the ids).
|
|
170
|
+
* 2. Wire `session.OnToolCall` → `ExecuteRelayedTool` → `session.SendToolResult`.
|
|
171
|
+
* 3. Guarantee finalize-once: wrap `session.Close()` and listen for an unexpected drop (`OnClose`), both
|
|
172
|
+
* routed through one idempotent finalizer. The bridge teardown calls `Close()`, so the run finalizes
|
|
173
|
+
* on graceful end; a dropped socket finalizes via `OnClose`.
|
|
174
|
+
*
|
|
175
|
+
* @param session The live realtime session the bridge owns (from `model.StartSession`).
|
|
176
|
+
* @param input The same prep input used to build the session (carries AgentSessionID, TargetAgentID, …).
|
|
177
|
+
* @param prep The successful {@link PrepareRealtimeSessionParams} result (CoAgent + Resolution).
|
|
178
|
+
* @param contextUser The calling user (threaded into observability + delegated runs).
|
|
179
|
+
* @param provider The request-scoped metadata provider.
|
|
180
|
+
* @returns A {@link BridgeRealtimeRuntime} the bridge holds for the session lifetime.
|
|
181
|
+
*/
|
|
182
|
+
async WireBridgeRealtimeSession(session, input, prep, contextUser, provider) {
|
|
183
|
+
const coAgent = prep.CoAgent;
|
|
184
|
+
const resolution = prep.Resolution;
|
|
185
|
+
if (!coAgent || !resolution) {
|
|
186
|
+
// Prep must have succeeded before wiring; degrade to a tool-error fallback rather than throw.
|
|
187
|
+
return this.wireBridgeFallbackRuntime(session);
|
|
188
|
+
}
|
|
189
|
+
const promptID = this.resolveCoAgentSystemPrompt(coAgent).PromptID;
|
|
190
|
+
const obs = await this.createCoAgentObservabilityRun(coAgent, promptID, resolution.ModelID, resolution.VendorID, input.UserID || contextUser?.ID, input.AgentSessionID, contextUser, provider, input.ConversationID);
|
|
191
|
+
let finalized = false;
|
|
192
|
+
const finalize = async (success) => {
|
|
193
|
+
if (finalized) {
|
|
194
|
+
return;
|
|
195
|
+
}
|
|
196
|
+
finalized = true;
|
|
197
|
+
await this.FinalizeCoAgentRun(obs?.CoAgentRunID ?? null, obs?.PromptRunID ?? null, contextUser, provider, success, obs?.CoAgentRunStepID ?? null);
|
|
198
|
+
};
|
|
199
|
+
// Tool calls → the shared delegation entry point, then hand the serialized result back to the model.
|
|
200
|
+
session.OnToolCall(async (call) => {
|
|
201
|
+
try {
|
|
202
|
+
const result = await this.ExecuteRelayedTool({ AgentSessionID: input.AgentSessionID, ParentRunID: obs?.CoAgentRunID, TargetAgentID: input.TargetAgentID, Call: call }, contextUser, provider);
|
|
203
|
+
await session.SendToolResult(call.CallID, result.ResultJson);
|
|
204
|
+
}
|
|
205
|
+
catch (error) {
|
|
206
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
207
|
+
LogError(`WireBridgeRealtimeSession: tool '${call.ToolName}' failed: ${message}`);
|
|
208
|
+
await session.SendToolResult(call.CallID, JSON.stringify({ success: false, error: message }));
|
|
209
|
+
}
|
|
210
|
+
});
|
|
211
|
+
// Finalize on graceful teardown (the bridge calls Close()) and on an unexpected drop. Both routed
|
|
212
|
+
// through the idempotent finalizer, so double-fire is harmless.
|
|
213
|
+
const originalClose = session.Close.bind(session);
|
|
214
|
+
session.Close = async () => {
|
|
215
|
+
await finalize(true);
|
|
216
|
+
await originalClose();
|
|
217
|
+
};
|
|
218
|
+
session.OnClose?.(() => { void finalize(true); });
|
|
219
|
+
return { CoAgentRunID: obs?.CoAgentRunID, PromptRunID: obs?.PromptRunID, Finalize: finalize };
|
|
220
|
+
}
|
|
221
|
+
/**
|
|
222
|
+
* Degenerate {@link BridgeRealtimeRuntime} for the rare case wiring is attempted without a resolved
|
|
223
|
+
* co-agent: answer every tool call with a clear "not available" error and a no-op finalize. Keeps the
|
|
224
|
+
* bridge from hanging on a tool call when prep was incomplete.
|
|
225
|
+
*/
|
|
226
|
+
wireBridgeFallbackRuntime(session) {
|
|
227
|
+
session.OnToolCall((call) => {
|
|
228
|
+
void session.SendToolResult(call.CallID, JSON.stringify({ success: false, error: 'Tool execution is unavailable — the co-agent did not resolve. Let the user know.' }));
|
|
229
|
+
});
|
|
230
|
+
return { Finalize: async () => { } };
|
|
231
|
+
}
|
|
232
|
+
/**
|
|
233
|
+
* **The single source of truth for realtime session prep.** Builds the {@link RealtimeSessionParams}
|
|
234
|
+
* for a co-agent voicing a target: resolves the co-agent, the effective config via the full precedence
|
|
235
|
+
* cascade (type-default < co-agent < **target** < runtime override), the realtime model, then assembles
|
|
236
|
+
* the companion system prompt (**first-person as the TARGET** — this is what gives every host the right
|
|
237
|
+
* identity), the stable tool set (**always including `invoke-target-agent`**), voice, and memory.
|
|
238
|
+
*
|
|
239
|
+
* EVERY realtime host consumes this — native chat via {@link PrepareClientSession} → `CreateClientSession`,
|
|
240
|
+
* and the server-bridged hosts (LiveKit, future Zoom/Teams) via `StartSession`. Hosts differ ONLY in how
|
|
241
|
+
* they OPEN the session and their media transport; identity/precedence/prompt/tools live here, once. Do
|
|
242
|
+
* NOT re-implement this in a host. See `plans/realtime/realtime-core-host-convergence.md`.
|
|
243
|
+
*
|
|
244
|
+
* Pure-ish and side-effect-free (no session opened, no observability run created) — those are the
|
|
245
|
+
* opener's concern. Never throws — returns `Success: false` on failure.
|
|
246
|
+
*
|
|
247
|
+
* @param input The co-agent/target/session inputs (the runtime override rides `ConfigOverridesJson`).
|
|
248
|
+
* @param contextUser The calling user (threaded to metadata + memory retrieval).
|
|
249
|
+
* @param provider The request-scoped metadata provider.
|
|
250
|
+
* @returns The prep result: `Success` + co-agent/resolution/effective-config/session-params, or `Success: false`.
|
|
251
|
+
*/
|
|
252
|
+
async PrepareRealtimeSessionParams(input, contextUser, provider) {
|
|
253
|
+
await this.configureEngine(contextUser, provider);
|
|
254
|
+
const coAgent = this.resolveCoAgent(input);
|
|
255
|
+
if (!coAgent) {
|
|
256
|
+
return { Success: false, ErrorMessage: 'The Realtime Co-Agent could not be resolved from the supplied id or entity.' };
|
|
257
|
+
}
|
|
258
|
+
// Effective config via the surface-agnostic cascade: type DefaultConfiguration < co-agent
|
|
259
|
+
// TypeConfiguration < TARGET agent TypeConfiguration < runtime overrides (authorization-gated
|
|
260
|
+
// upstream). This is the identical precedence on every host.
|
|
261
|
+
const targetAgent = this.resolveTargetAgent(input.TargetAgentID);
|
|
262
|
+
const effectiveConfig = this.resolveEffectiveConfig(coAgent, input.ConfigOverridesJson, targetAgent);
|
|
263
|
+
const outcome = await this.resolveModelForSession(input, coAgent, effectiveConfig);
|
|
264
|
+
if (!outcome.Resolution) {
|
|
265
|
+
return { Success: false, ErrorMessage: outcome.ErrorMessage ?? this.noModelMessage() };
|
|
266
|
+
}
|
|
267
|
+
const resolution = outcome.Resolution;
|
|
268
|
+
const sessionParams = await this.buildSessionParams(input, coAgent, resolution.APIName, contextUser, provider, effectiveConfig, resolution.DriverClass);
|
|
269
|
+
return { Success: true, CoAgent: coAgent, Resolution: resolution, EffectiveConfig: effectiveConfig, SessionParams: sessionParams };
|
|
270
|
+
}
|
|
271
|
+
/**
|
|
272
|
+
* Resolves the EFFECTIVE realtime configuration via the surface-agnostic precedence cascade:
|
|
273
|
+
* agent-TYPE `DefaultConfiguration` (base) < **co-agent** `TypeConfiguration` < **target agent**
|
|
274
|
+
* `TypeConfiguration` < (pre-authorized) runtime override — deep-merged per key and normalized.
|
|
275
|
+
* The target layer is what makes a voiced agent (Sage, Marketing Agent, …) carry its own voice/model
|
|
276
|
+
* regardless of host. Tolerant end-to-end: malformed layers contribute nothing and an unloaded metadata
|
|
277
|
+
* cache yields no type defaults. See `plans/realtime/realtime-core-host-convergence.md`.
|
|
278
|
+
*
|
|
279
|
+
* @param coAgent The resolved co-agent.
|
|
280
|
+
* @param overridesJson The pre-authorized runtime override layer, when present.
|
|
281
|
+
* @param targetAgent The TARGET agent being voiced, when distinct from the co-agent — contributes the
|
|
282
|
+
* per-voiced-agent layer (above the co-agent, below the runtime override). Omit when there is none.
|
|
283
|
+
* @returns The normalized effective configuration (possibly empty, never `null`).
|
|
284
|
+
*/
|
|
285
|
+
resolveEffectiveConfig(coAgent, overridesJson, targetAgent) {
|
|
286
|
+
return ResolveEffectiveRealtimeConfig(this.getAgentTypeDefaultConfiguration(coAgent), coAgent.TypeConfiguration ?? null, overridesJson ?? null, targetAgent?.TypeConfiguration ?? null);
|
|
287
|
+
}
|
|
288
|
+
/**
|
|
289
|
+
* Reads the co-agent's TYPE-level `DefaultConfiguration` from {@link AIEngine}'s cached agent
|
|
290
|
+
* types. **Overridable seam**; tolerant — an absent type or unloaded cache returns `null`.
|
|
291
|
+
*/
|
|
292
|
+
getAgentTypeDefaultConfiguration(coAgent) {
|
|
293
|
+
try {
|
|
294
|
+
if (!coAgent.TypeID) {
|
|
295
|
+
return null;
|
|
296
|
+
}
|
|
297
|
+
const type = (AIEngine.Instance.AgentTypes ?? []).find(t => UUIDsEqual(t.ID, coAgent.TypeID));
|
|
298
|
+
return type?.DefaultConfiguration ?? null;
|
|
299
|
+
}
|
|
300
|
+
catch {
|
|
301
|
+
return null;
|
|
302
|
+
}
|
|
303
|
+
}
|
|
304
|
+
/**
|
|
305
|
+
* Creates the server-side co-agent observability runs for a voice session: an `AIAgentRun`
|
|
306
|
+
* (Status `Running`), and — when a co-agent system prompt resolved — a linked `AIPromptRun`
|
|
307
|
+
* (Status `Running`, `AgentRunID` = the co-agent run, `AgentID` = the co-agent) plus a single
|
|
308
|
+
* `MJ: AI Agent Run Steps` row (StepType `Prompt`) so the co-agent run's Timeline is non-empty.
|
|
309
|
+
* Delegated target-agent runs nest under the returned `CoAgentRunID` via `ParentRunID`.
|
|
310
|
+
*
|
|
311
|
+
* Best-effort: returns `null` (and logs) when the co-agent run cannot be saved, so callers can
|
|
312
|
+
* continue without observability rather than failing the whole prepare. A failed prompt-run or
|
|
313
|
+
* run-step save just omits that id.
|
|
314
|
+
*
|
|
315
|
+
* @param coAgent The resolved co-agent (its id stamps `AgentID` on both runs).
|
|
316
|
+
* @param promptID The co-agent system prompt id, or `null` to skip the prompt run + run step.
|
|
317
|
+
* @param modelID The resolved realtime model id (stamps the prompt run's `ModelID`).
|
|
318
|
+
* @param userID Optional owning user id for the agent run.
|
|
319
|
+
* @param agentSessionID The session id grouping this voice session's runs.
|
|
320
|
+
* @param contextUser The calling user.
|
|
321
|
+
* @param provider The request-scoped metadata provider.
|
|
322
|
+
* @returns The `{ CoAgentRunID, PromptRunID, CoAgentRunStepID }` ids, or `null` when the agent run failed.
|
|
323
|
+
*/
|
|
324
|
+
async createCoAgentObservabilityRun(coAgent, promptID, modelID, vendorID, userID, agentSessionID, contextUser, provider, conversationID) {
|
|
325
|
+
const coAgentRunID = await this.createCoAgentRun(coAgent, userID, agentSessionID, conversationID, contextUser, provider);
|
|
326
|
+
if (!coAgentRunID) {
|
|
327
|
+
return null;
|
|
328
|
+
}
|
|
329
|
+
const promptRunID = await this.createCoAgentPromptRun(coAgent, promptID, modelID, vendorID, coAgentRunID, contextUser, provider);
|
|
330
|
+
const runStepID = await this.createCoAgentRunStep(coAgentRunID, promptID, promptRunID, contextUser, provider);
|
|
331
|
+
return { CoAgentRunID: coAgentRunID, PromptRunID: promptRunID ?? undefined, CoAgentRunStepID: runStepID ?? undefined };
|
|
332
|
+
}
|
|
333
|
+
/**
|
|
334
|
+
* Creates the co-agent `AIAgentRun` row (Status `Running`). Returns its id, or `null` (logging
|
|
335
|
+
* `CompleteMessage`) when the save fails.
|
|
336
|
+
*/
|
|
337
|
+
async createCoAgentRun(coAgent, userID, agentSessionID, conversationID, contextUser, provider) {
|
|
338
|
+
const run = await provider.GetEntityObject('MJ: AI Agent Runs', contextUser);
|
|
339
|
+
run.NewRecord();
|
|
340
|
+
run.AgentID = coAgent.ID;
|
|
341
|
+
run.Status = 'Running';
|
|
342
|
+
run.StartedAt = new Date();
|
|
343
|
+
// Only stamp AgentSessionID when we actually have one — `AgentSessionID` is a `uniqueidentifier` FK,
|
|
344
|
+
// so assigning '' (a surface that didn't thread a session id) makes the WHOLE run save fail and the
|
|
345
|
+
// co-agent observability silently vanishes. Degrade gracefully: log the run without session grouping
|
|
346
|
+
// rather than not at all. This keeps the core logging identical across surfaces regardless of input.
|
|
347
|
+
const sessionID = agentSessionID?.trim();
|
|
348
|
+
if (sessionID) {
|
|
349
|
+
run.AgentSessionID = sessionID;
|
|
350
|
+
}
|
|
351
|
+
if (conversationID) {
|
|
352
|
+
run.ConversationID = conversationID;
|
|
353
|
+
}
|
|
354
|
+
if (userID) {
|
|
355
|
+
run.UserID = userID;
|
|
356
|
+
}
|
|
357
|
+
if (await run.Save()) {
|
|
358
|
+
return run.ID;
|
|
359
|
+
}
|
|
360
|
+
LogError(`RealtimeClientSessionService.createCoAgentRun save failed: ${run.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
361
|
+
return null;
|
|
362
|
+
}
|
|
363
|
+
/**
|
|
364
|
+
* Creates the co-agent `AIPromptRun` row (Status `Running`) linked to the co-agent run via
|
|
365
|
+
* `AgentRunID` AND to the co-agent itself via `AgentID` — so the run shows up both on the
|
|
366
|
+
* prompt's run history (`PromptID`) and in agent-scoped prompt-run views. Returns its id, or
|
|
367
|
+
* `null` when `promptID` is absent (skipped) or the save fails (logged).
|
|
368
|
+
*/
|
|
369
|
+
async createCoAgentPromptRun(coAgent, promptID, modelID, vendorID, coAgentRunID, contextUser, provider) {
|
|
370
|
+
if (!promptID) {
|
|
371
|
+
return null;
|
|
372
|
+
}
|
|
373
|
+
const promptRun = await provider.GetEntityObject('MJ: AI Prompt Runs', contextUser);
|
|
374
|
+
promptRun.NewRecord();
|
|
375
|
+
promptRun.PromptID = promptID;
|
|
376
|
+
promptRun.ModelID = modelID;
|
|
377
|
+
// VendorID is required on AIPromptRun ("Vendor cannot be null") — without it the prompt run
|
|
378
|
+
// save fails and the whole co-agent observability chain (transcript/tool-turn/usage) is dropped.
|
|
379
|
+
if (vendorID) {
|
|
380
|
+
promptRun.VendorID = vendorID;
|
|
381
|
+
}
|
|
382
|
+
promptRun.AgentID = coAgent.ID;
|
|
383
|
+
promptRun.RunAt = new Date();
|
|
384
|
+
promptRun.RunType = 'Single';
|
|
385
|
+
promptRun.Status = 'Running';
|
|
386
|
+
promptRun.AgentRunID = coAgentRunID;
|
|
387
|
+
if (await promptRun.Save()) {
|
|
388
|
+
return promptRun.ID;
|
|
389
|
+
}
|
|
390
|
+
LogError(`RealtimeClientSessionService.createCoAgentPromptRun save failed: ${promptRun.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
391
|
+
return null;
|
|
392
|
+
}
|
|
393
|
+
/**
|
|
394
|
+
* Creates the single `MJ: AI Agent Run Steps` row for the co-agent observability run — the
|
|
395
|
+
* realtime session has no iterative loop, so its Timeline carries exactly one step
|
|
396
|
+
* representing the session's system prompt (StepNumber 1, StepType `Prompt`, Status `Running`,
|
|
397
|
+
* `TargetID` = the system `AIPrompt`, `TargetLogID` = the linked `AIPromptRun` when one was
|
|
398
|
+
* created). Skipped (returns `null`) when no system prompt resolved. Best-effort: a save
|
|
399
|
+
* failure is logged and returns `null` — it never breaks the session.
|
|
400
|
+
*/
|
|
401
|
+
async createCoAgentRunStep(coAgentRunID, promptID, promptRunID, contextUser, provider) {
|
|
402
|
+
if (!promptID) {
|
|
403
|
+
return null;
|
|
404
|
+
}
|
|
405
|
+
try {
|
|
406
|
+
const step = await provider.GetEntityObject('MJ: AI Agent Run Steps', contextUser);
|
|
407
|
+
step.NewRecord();
|
|
408
|
+
step.AgentRunID = coAgentRunID;
|
|
409
|
+
step.StepNumber = 1;
|
|
410
|
+
step.StepType = 'Prompt';
|
|
411
|
+
step.StepName = 'Realtime session system prompt';
|
|
412
|
+
step.TargetID = promptID;
|
|
413
|
+
step.TargetLogID = promptRunID;
|
|
414
|
+
step.Status = 'Running';
|
|
415
|
+
step.StartedAt = new Date();
|
|
416
|
+
if (await step.Save()) {
|
|
417
|
+
return step.ID;
|
|
418
|
+
}
|
|
419
|
+
LogError(`RealtimeClientSessionService.createCoAgentRunStep save failed: ${step.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
420
|
+
return null;
|
|
421
|
+
}
|
|
422
|
+
catch (error) {
|
|
423
|
+
LogError(`RealtimeClientSessionService.createCoAgentRunStep failed: ${error instanceof Error ? error.message : String(error)}`);
|
|
424
|
+
return null;
|
|
425
|
+
}
|
|
426
|
+
}
|
|
427
|
+
/**
|
|
428
|
+
* Finalizes the server-side co-agent observability records when a voice session ends. Loads
|
|
429
|
+
* each (when its id is supplied) and, **only if it is still `Running`**, sets it to `Completed`
|
|
430
|
+
* (or `Failed` when `success` is false) with a `CompletedAt` + `Success` stamp. Idempotent and
|
|
431
|
+
* tolerant: a missing/already-finalized record is a no-op; a load/save failure is logged,
|
|
432
|
+
* never thrown.
|
|
433
|
+
*
|
|
434
|
+
* @param coAgentRunID The co-agent run id, or `null` to skip.
|
|
435
|
+
* @param promptRunID The co-agent prompt run id, or `null` to skip.
|
|
436
|
+
* @param contextUser The calling user.
|
|
437
|
+
* @param provider The request-scoped metadata provider.
|
|
438
|
+
* @param success Whether the session ended successfully (controls Completed vs Failed).
|
|
439
|
+
* @param coAgentRunStepID The co-agent run's single `MJ: AI Agent Run Steps` row id, or `null` to skip.
|
|
440
|
+
*/
|
|
441
|
+
async FinalizeCoAgentRun(coAgentRunID, promptRunID, contextUser, provider, success = true, coAgentRunStepID = null) {
|
|
442
|
+
await this.finalizeAgentRun(coAgentRunID, contextUser, provider, success);
|
|
443
|
+
await this.finalizePromptRun(promptRunID, contextUser, provider, success);
|
|
444
|
+
await this.finalizeRunStep(coAgentRunStepID, contextUser, provider, success);
|
|
445
|
+
}
|
|
446
|
+
/**
|
|
447
|
+
* Finalizes the **co-agent observability run(s)** for an agent session that were left `Running` because
|
|
448
|
+
* the session was reaped WITHOUT a live in-memory handle — a prior-boot orphan or a cross-host teardown,
|
|
449
|
+
* where the `Close()`-wrapped finalizer never ran. This is the by-`AgentSessionID` analogue of
|
|
450
|
+
* {@link FinalizeCoAgentRun}: the same-process path already knows its run ids (no query), but here that
|
|
451
|
+
* state died with the prior process, so we locate the session's TOP-LEVEL co-agent run (delegated target
|
|
452
|
+
* runs nest under it and finalize on their own runner) and finalize it + its prompt run + step via the
|
|
453
|
+
* same idempotent helpers. A clean teardown already marked them `Completed`, so this finds nothing.
|
|
454
|
+
*
|
|
455
|
+
* `MJ: AI Agent Runs` is a high-volume transactional table no engine caches, so a narrow ids-only query
|
|
456
|
+
* is the right tool (not a cache reuse). Tolerant — never throws.
|
|
457
|
+
*
|
|
458
|
+
* @param agentSessionID The agent session whose dangling co-agent runs to finalize.
|
|
459
|
+
* @param success Mark them `Completed` (true) or `Failed` (false).
|
|
460
|
+
* @param contextUser The user the writes run as.
|
|
461
|
+
* @param provider The request-scoped metadata provider.
|
|
462
|
+
* @returns The number of co-agent runs finalized (0 when none were dangling).
|
|
463
|
+
*/
|
|
464
|
+
async FinalizeCoAgentRunsBySession(agentSessionID, success, contextUser, provider) {
|
|
465
|
+
const sessionID = agentSessionID?.trim();
|
|
466
|
+
if (!sessionID) {
|
|
467
|
+
return 0;
|
|
468
|
+
}
|
|
469
|
+
const rv = new RunView();
|
|
470
|
+
const found = await rv.RunView({
|
|
471
|
+
EntityName: 'MJ: AI Agent Runs',
|
|
472
|
+
ExtraFilter: `AgentSessionID='${this.escapeSqlLiteral(sessionID)}' AND Status='Running' AND ParentRunID IS NULL`,
|
|
473
|
+
Fields: ['ID'],
|
|
474
|
+
ResultType: 'simple',
|
|
475
|
+
}, contextUser);
|
|
476
|
+
if (!found.Success) {
|
|
477
|
+
LogError(`RealtimeClientSessionService.FinalizeCoAgentRunsBySession RunView failed: ${found.ErrorMessage}`);
|
|
478
|
+
return 0;
|
|
479
|
+
}
|
|
480
|
+
let finalized = 0;
|
|
481
|
+
for (const row of found.Results) {
|
|
482
|
+
const child = await this.findCoAgentChildLogIds(row.ID, contextUser);
|
|
483
|
+
await this.FinalizeCoAgentRun(row.ID, child.PromptRunID, contextUser, provider, success, child.StepID);
|
|
484
|
+
finalized++;
|
|
485
|
+
}
|
|
486
|
+
if (finalized > 0) {
|
|
487
|
+
LogStatus(`RealtimeClientSessionService: finalized ${finalized} orphaned co-agent run(s) for session ${sessionID}.`);
|
|
488
|
+
}
|
|
489
|
+
return finalized;
|
|
490
|
+
}
|
|
491
|
+
/** Finds the still-`Running` prompt-run + run-step ids for a co-agent run (orphan finalize path). */
|
|
492
|
+
async findCoAgentChildLogIds(coAgentRunID, contextUser) {
|
|
493
|
+
const rv = new RunView();
|
|
494
|
+
const results = await rv.RunViews([
|
|
495
|
+
{ EntityName: 'MJ: AI Prompt Runs', ExtraFilter: `AgentRunID='${this.escapeSqlLiteral(coAgentRunID)}' AND Status='Running'`, Fields: ['ID'], ResultType: 'simple' },
|
|
496
|
+
{ EntityName: 'MJ: AI Agent Run Steps', ExtraFilter: `AgentRunID='${this.escapeSqlLiteral(coAgentRunID)}' AND Status='Running'`, Fields: ['ID'], ResultType: 'simple' },
|
|
497
|
+
], contextUser);
|
|
498
|
+
const firstId = (r) => {
|
|
499
|
+
const first = r?.Success ? r.Results[0] : undefined;
|
|
500
|
+
return first?.ID ?? null;
|
|
501
|
+
};
|
|
502
|
+
return { PromptRunID: firstId(results[0]), StepID: firstId(results[1]) };
|
|
503
|
+
}
|
|
504
|
+
/** Escapes single quotes for safe embedding in an `ExtraFilter` literal. */
|
|
505
|
+
escapeSqlLiteral(value) {
|
|
506
|
+
return value.replace(/'/g, "''");
|
|
507
|
+
}
|
|
508
|
+
/** Loads + finalizes the co-agent `AIAgentRun` if still `Running`. Tolerant: logs, never throws. */
|
|
509
|
+
async finalizeAgentRun(coAgentRunID, contextUser, provider, success) {
|
|
510
|
+
if (!coAgentRunID) {
|
|
511
|
+
return;
|
|
512
|
+
}
|
|
513
|
+
const run = await provider.GetEntityObject('MJ: AI Agent Runs', contextUser);
|
|
514
|
+
if (!(await run.Load(coAgentRunID)) || run.Status !== 'Running') {
|
|
515
|
+
return;
|
|
516
|
+
}
|
|
517
|
+
run.Status = success ? 'Completed' : 'Failed';
|
|
518
|
+
run.CompletedAt = new Date();
|
|
519
|
+
run.Success = success;
|
|
520
|
+
if (!(await run.Save())) {
|
|
521
|
+
LogError(`RealtimeClientSessionService.finalizeAgentRun save failed: ${run.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
522
|
+
}
|
|
523
|
+
}
|
|
524
|
+
/**
|
|
525
|
+
* Loads + finalizes the co-agent run's single system-prompt `MJ: AI Agent Run Steps` row if
|
|
526
|
+
* still `Running` (Status `Completed`/`Failed`, `CompletedAt`, `Success`). Tolerant: a
|
|
527
|
+
* missing/already-finalized step is a no-op; a load/save failure is logged, never thrown.
|
|
528
|
+
*/
|
|
529
|
+
async finalizeRunStep(coAgentRunStepID, contextUser, provider, success) {
|
|
530
|
+
if (!coAgentRunStepID) {
|
|
531
|
+
return;
|
|
532
|
+
}
|
|
533
|
+
try {
|
|
534
|
+
const step = await provider.GetEntityObject('MJ: AI Agent Run Steps', contextUser);
|
|
535
|
+
if (!(await step.Load(coAgentRunStepID)) || step.Status !== 'Running') {
|
|
536
|
+
return;
|
|
537
|
+
}
|
|
538
|
+
step.Status = success ? 'Completed' : 'Failed';
|
|
539
|
+
step.CompletedAt = new Date();
|
|
540
|
+
step.Success = success;
|
|
541
|
+
if (!success) {
|
|
542
|
+
step.ErrorMessage = 'The realtime session ended in an error state.';
|
|
543
|
+
}
|
|
544
|
+
if (!(await step.Save())) {
|
|
545
|
+
LogError(`RealtimeClientSessionService.finalizeRunStep save failed: ${step.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
546
|
+
}
|
|
547
|
+
}
|
|
548
|
+
catch (error) {
|
|
549
|
+
LogError(`RealtimeClientSessionService.finalizeRunStep failed: ${error instanceof Error ? error.message : String(error)}`);
|
|
550
|
+
}
|
|
551
|
+
}
|
|
552
|
+
/** Loads + finalizes the co-agent `AIPromptRun` if still `Running`. Tolerant: logs, never throws. */
|
|
553
|
+
async finalizePromptRun(promptRunID, contextUser, provider, success) {
|
|
554
|
+
if (!promptRunID) {
|
|
555
|
+
return;
|
|
556
|
+
}
|
|
557
|
+
// Serialize the finalize against any in-flight message/usage writes so it can't race them — and so a
|
|
558
|
+
// late usage flush queued behind it sees the run already Completed.
|
|
559
|
+
await this.serializePromptRunWrite(promptRunID, async () => {
|
|
560
|
+
const run = await provider.GetEntityObject('MJ: AI Prompt Runs', contextUser);
|
|
561
|
+
if (!(await run.Load(promptRunID)) || run.Status !== 'Running') {
|
|
562
|
+
return false;
|
|
563
|
+
}
|
|
564
|
+
run.Status = success ? 'Completed' : 'Failed';
|
|
565
|
+
run.CompletedAt = new Date();
|
|
566
|
+
run.Success = success;
|
|
567
|
+
if (!(await run.Save())) {
|
|
568
|
+
LogError(`RealtimeClientSessionService.finalizePromptRun save failed: ${run.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
569
|
+
}
|
|
570
|
+
return true;
|
|
571
|
+
});
|
|
572
|
+
// Drop the per-run lock chain — no further writes are expected after finalize.
|
|
573
|
+
this.promptRunWriteChains.delete(promptRunID);
|
|
574
|
+
}
|
|
575
|
+
/**
|
|
576
|
+
* Appends (or replaces) one transcript turn onto the co-agent's long-lived `AIPromptRun.Messages`,
|
|
577
|
+
* so the realtime co-agent's conversation is captured on its run exactly like every other MJ agent
|
|
578
|
+
* run — closing the observability gap where the run held only token totals, never the turns. The
|
|
579
|
+
* run viewer can then show what the co-agent heard and said. Mirrors {@link accumulatePromptRunUsage}'s
|
|
580
|
+
* load/append/save pattern; best-effort and tolerant (logs, never throws).
|
|
581
|
+
*
|
|
582
|
+
* `replacePrevious` swaps the last same-role message instead of appending — the streaming-correction
|
|
583
|
+
* case (an interim assistant turn finalized into its full text). The stored shape is the standard
|
|
584
|
+
* chat-message array (`[{ role, content }, …]`) the rest of MJ already reads from `Messages`.
|
|
585
|
+
*
|
|
586
|
+
* NOTE: load-append-save carries the same benign race as usage accumulation; realtime turns are
|
|
587
|
+
* sequential per session so collisions are rare. A dedicated child turn-row entity would remove the
|
|
588
|
+
* race (and the blob rewrite) entirely — a future increment. Tool-call turns (the browser_ and
|
|
589
|
+
* Whiteboard_ channel tools) are a separate increment that requires the client to relay them.
|
|
590
|
+
*
|
|
591
|
+
* @returns `true` when the turn was persisted onto the prompt run.
|
|
592
|
+
*/
|
|
593
|
+
async AppendPromptRunMessage(promptRunID, role, content, replacePrevious, contextUser, provider) {
|
|
594
|
+
// Serialized against usage checkpoints on the same run so a concurrent usage save can't clobber
|
|
595
|
+
// the Messages we write here (and vice-versa). See promptRunWriteChains.
|
|
596
|
+
return this.serializePromptRunWrite(promptRunID, async () => {
|
|
597
|
+
try {
|
|
598
|
+
const promptRun = await provider.GetEntityObject('MJ: AI Prompt Runs', contextUser);
|
|
599
|
+
if (!(await promptRun.Load(promptRunID))) {
|
|
600
|
+
LogError(`AppendPromptRunMessage: co-agent prompt run ${promptRunID} not found — transcript turn dropped.`);
|
|
601
|
+
return false;
|
|
602
|
+
}
|
|
603
|
+
const messages = this.parsePromptRunMessages(promptRun.Messages);
|
|
604
|
+
const last = messages[messages.length - 1];
|
|
605
|
+
if (replacePrevious && last && last.role === role) {
|
|
606
|
+
last.content = content;
|
|
607
|
+
}
|
|
608
|
+
else {
|
|
609
|
+
messages.push({ role, content });
|
|
610
|
+
}
|
|
611
|
+
promptRun.Messages = JSON.stringify(messages);
|
|
612
|
+
if (!(await promptRun.Save())) {
|
|
613
|
+
LogError(`AppendPromptRunMessage: prompt run ${promptRunID} save failed: ${promptRun.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
614
|
+
return false;
|
|
615
|
+
}
|
|
616
|
+
return true;
|
|
617
|
+
}
|
|
618
|
+
catch (error) {
|
|
619
|
+
LogError(`AppendPromptRunMessage: append failed for prompt run ${promptRunID}: ${error.message}`);
|
|
620
|
+
return false;
|
|
621
|
+
}
|
|
622
|
+
});
|
|
623
|
+
}
|
|
624
|
+
/**
|
|
625
|
+
* Accumulates relayed usage DELTAS onto the co-agent `AIPromptRun`'s `TokensPrompt` / `TokensCompletion`
|
|
626
|
+
* (recomputing `TokensUsed`). Serialized against {@link AppendPromptRunMessage} on the same run so the
|
|
627
|
+
* high-frequency usage checkpoint never overwrites freshly-appended transcript turns (and vice-versa).
|
|
628
|
+
* Best-effort: load/save failures log and return `false`, never throw.
|
|
629
|
+
*
|
|
630
|
+
* @param promptRunID The co-agent observability prompt run.
|
|
631
|
+
* @param inputDelta Input-token delta to add (caller clamps to >= 0).
|
|
632
|
+
* @param outputDelta Output-token delta to add (caller clamps to >= 0).
|
|
633
|
+
* @returns `true` when the accumulated usage was persisted.
|
|
634
|
+
*/
|
|
635
|
+
async AccumulatePromptRunUsage(promptRunID, inputDelta, outputDelta, contextUser, provider) {
|
|
636
|
+
return this.serializePromptRunWrite(promptRunID, async () => {
|
|
637
|
+
try {
|
|
638
|
+
const promptRun = await provider.GetEntityObject('MJ: AI Prompt Runs', contextUser);
|
|
639
|
+
if (!(await promptRun.Load(promptRunID))) {
|
|
640
|
+
LogError(`AccumulatePromptRunUsage: co-agent prompt run ${promptRunID} not found — usage delta dropped.`);
|
|
641
|
+
return false;
|
|
642
|
+
}
|
|
643
|
+
promptRun.TokensPrompt = (promptRun.TokensPrompt ?? 0) + inputDelta;
|
|
644
|
+
promptRun.TokensCompletion = (promptRun.TokensCompletion ?? 0) + outputDelta;
|
|
645
|
+
promptRun.TokensUsed = (promptRun.TokensPrompt ?? 0) + (promptRun.TokensCompletion ?? 0);
|
|
646
|
+
if (!(await promptRun.Save())) {
|
|
647
|
+
LogError(`AccumulatePromptRunUsage: prompt run ${promptRunID} save failed: ${promptRun.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
648
|
+
return false;
|
|
649
|
+
}
|
|
650
|
+
return true;
|
|
651
|
+
}
|
|
652
|
+
catch (error) {
|
|
653
|
+
LogError(`AccumulatePromptRunUsage: usage accumulation failed for prompt run ${promptRunID}: ${error.message}`);
|
|
654
|
+
return false;
|
|
655
|
+
}
|
|
656
|
+
});
|
|
657
|
+
}
|
|
658
|
+
/** Parses the prompt run's `Messages` JSON into a mutable chat-message array (tolerant: `[]` on empty/malformed). */
|
|
659
|
+
parsePromptRunMessages(raw) {
|
|
660
|
+
if (!raw || !raw.trim()) {
|
|
661
|
+
return [];
|
|
662
|
+
}
|
|
663
|
+
try {
|
|
664
|
+
const parsed = JSON.parse(raw);
|
|
665
|
+
return Array.isArray(parsed) ? parsed : [];
|
|
666
|
+
}
|
|
667
|
+
catch {
|
|
668
|
+
return [];
|
|
669
|
+
}
|
|
670
|
+
}
|
|
671
|
+
/**
|
|
672
|
+
* Executes a single tool call relayed from the browser and returns its serialized result.
|
|
673
|
+
*
|
|
674
|
+
* Builds a {@link RealtimeToolBroker} whose `DelegateToTarget` runs the target agent (threading
|
|
675
|
+
* the abort signal, parent run, and session id) and whose `ExecuteTool` returns a structured
|
|
676
|
+
* "not available" result for non-target tools (action wiring is a later phase). The broker
|
|
677
|
+
* routes the call and always resolves with structured JSON — failures become `tool_response`
|
|
678
|
+
* errors the model can narrate rather than thrown exceptions.
|
|
679
|
+
*
|
|
680
|
+
* @param input The relayed tool call plus delegation linkage.
|
|
681
|
+
* @param contextUser The calling user (threaded into the delegated agent run).
|
|
682
|
+
* @param provider The request-scoped metadata provider (threaded into the delegated agent run).
|
|
683
|
+
* @returns `{ ResultJson, Success, PausedRunID?, Artifacts? }` — the serialized tool result for
|
|
684
|
+
* the browser to relay back, the paused run id when the delegated target agent paused awaiting
|
|
685
|
+
* feedback (so the resolver can persist it and resume that run on the next answer), and the
|
|
686
|
+
* artifacts the delegated run produced (so the resolver can junction-link them into the
|
|
687
|
+
* session's conversation history — the same info is embedded in `ResultJson` for the client).
|
|
688
|
+
*/
|
|
689
|
+
async ExecuteRelayedTool(input, contextUser, provider) {
|
|
690
|
+
// Register this call in the in-flight registry so CancelInFlightDelegations (the
|
|
691
|
+
// CancelRealtimeSessionTool mutation) can abort it mid-flight. The registry controller's
|
|
692
|
+
// signal is combined with any caller-supplied signal — either source cancels the run.
|
|
693
|
+
const controller = this.registerInFlightDelegation(input.AgentSessionID, input.Call.CallID);
|
|
694
|
+
const effectiveInput = {
|
|
695
|
+
...input,
|
|
696
|
+
AbortSignal: input.AbortSignal ? this.combineSignals(controller.signal, input.AbortSignal) : controller.signal
|
|
697
|
+
};
|
|
698
|
+
try {
|
|
699
|
+
const broker = this.buildToolBroker(effectiveInput, contextUser, provider);
|
|
700
|
+
const result = await broker.ExecuteToolCall(effectiveInput.Call);
|
|
701
|
+
return { ResultJson: result.ResultJson, Success: result.Success, PausedRunID: result.PausedRunID, Artifacts: result.Artifacts };
|
|
702
|
+
}
|
|
703
|
+
finally {
|
|
704
|
+
this.unregisterInFlightDelegation(input.AgentSessionID, input.Call.CallID, controller);
|
|
705
|
+
}
|
|
706
|
+
}
|
|
707
|
+
/**
|
|
708
|
+
* Aborts in-flight relayed delegations for a session — the server half of the client-direct
|
|
709
|
+
* CANCEL channel (see the registry note on {@link inFlightDelegations}).
|
|
710
|
+
*
|
|
711
|
+
* @param agentSessionID The session whose in-flight delegations to abort.
|
|
712
|
+
* @param callID When supplied, only the delegation for this specific call is aborted; when
|
|
713
|
+
* omitted, EVERY in-flight delegation for the session is aborted.
|
|
714
|
+
* @returns The number of in-flight delegations aborted. **Tolerant by design**: an unknown
|
|
715
|
+
* session, an unknown call id, or a session with nothing in flight returns `0` — never throws
|
|
716
|
+
* (the call the user wanted dead may simply have finished already, which is a fine outcome).
|
|
717
|
+
*/
|
|
718
|
+
CancelInFlightDelegations(agentSessionID, callID) {
|
|
719
|
+
const sessionKey = this.registryKey(agentSessionID);
|
|
720
|
+
const sessionMap = this.inFlightDelegations.get(sessionKey);
|
|
721
|
+
if (!sessionMap || sessionMap.size === 0) {
|
|
722
|
+
return 0;
|
|
723
|
+
}
|
|
724
|
+
let aborted = 0;
|
|
725
|
+
if (callID != null && callID.trim().length > 0) {
|
|
726
|
+
const callKey = this.registryKey(callID);
|
|
727
|
+
const controller = sessionMap.get(callKey);
|
|
728
|
+
if (controller) {
|
|
729
|
+
controller.abort();
|
|
730
|
+
sessionMap.delete(callKey);
|
|
731
|
+
aborted = 1;
|
|
732
|
+
}
|
|
733
|
+
}
|
|
734
|
+
else {
|
|
735
|
+
for (const controller of sessionMap.values()) {
|
|
736
|
+
controller.abort();
|
|
737
|
+
aborted++;
|
|
738
|
+
}
|
|
739
|
+
sessionMap.clear();
|
|
740
|
+
}
|
|
741
|
+
if (sessionMap.size === 0) {
|
|
742
|
+
this.inFlightDelegations.delete(sessionKey);
|
|
743
|
+
}
|
|
744
|
+
if (aborted > 0) {
|
|
745
|
+
LogStatus(`RealtimeClientSessionService: aborted ${aborted} in-flight delegation(s) for session ${agentSessionID}.`);
|
|
746
|
+
}
|
|
747
|
+
return aborted;
|
|
748
|
+
}
|
|
749
|
+
/** Normalized (trim + lowercase) registry key so UUID casing differences can't split entries. */
|
|
750
|
+
registryKey(id) {
|
|
751
|
+
return id.trim().toLowerCase();
|
|
752
|
+
}
|
|
753
|
+
/** Creates + registers the abort controller for one in-flight relayed call. */
|
|
754
|
+
registerInFlightDelegation(agentSessionID, callID) {
|
|
755
|
+
const sessionKey = this.registryKey(agentSessionID);
|
|
756
|
+
let sessionMap = this.inFlightDelegations.get(sessionKey);
|
|
757
|
+
if (!sessionMap) {
|
|
758
|
+
sessionMap = new Map();
|
|
759
|
+
this.inFlightDelegations.set(sessionKey, sessionMap);
|
|
760
|
+
}
|
|
761
|
+
const controller = new AbortController();
|
|
762
|
+
sessionMap.set(this.registryKey(callID), controller);
|
|
763
|
+
return controller;
|
|
764
|
+
}
|
|
765
|
+
/**
|
|
766
|
+
* Removes one call's registry entry on completion — but only when the stored controller is
|
|
767
|
+
* STILL the one this execution registered (a cancel may already have removed it, and a
|
|
768
|
+
* same-callID retry may have replaced it).
|
|
769
|
+
*/
|
|
770
|
+
unregisterInFlightDelegation(agentSessionID, callID, controller) {
|
|
771
|
+
const sessionKey = this.registryKey(agentSessionID);
|
|
772
|
+
const sessionMap = this.inFlightDelegations.get(sessionKey);
|
|
773
|
+
if (!sessionMap) {
|
|
774
|
+
return;
|
|
775
|
+
}
|
|
776
|
+
const callKey = this.registryKey(callID);
|
|
777
|
+
if (sessionMap.get(callKey) === controller) {
|
|
778
|
+
sessionMap.delete(callKey);
|
|
779
|
+
}
|
|
780
|
+
if (sessionMap.size === 0) {
|
|
781
|
+
this.inFlightDelegations.delete(sessionKey);
|
|
782
|
+
}
|
|
783
|
+
}
|
|
784
|
+
/**
|
|
785
|
+
* Ensures {@link AIEngine} metadata is loaded before resolution. **Overridable seam** so tests
|
|
786
|
+
* can skip the DB-backed config load.
|
|
787
|
+
*
|
|
788
|
+
* @param contextUser The calling user.
|
|
789
|
+
* @param provider The request-scoped metadata provider.
|
|
790
|
+
*/
|
|
791
|
+
async configureEngine(contextUser, provider) {
|
|
792
|
+
await AIEngine.Instance.Config(false, contextUser, provider);
|
|
793
|
+
}
|
|
794
|
+
/**
|
|
795
|
+
* Resolves the co-agent from either the supplied entity or its id (from cached metadata).
|
|
796
|
+
*
|
|
797
|
+
* @param input The prepare-session input.
|
|
798
|
+
* @returns The co-agent entity, or `null` when neither form resolves.
|
|
799
|
+
*/
|
|
800
|
+
resolveCoAgent(input) {
|
|
801
|
+
if (input.CoAgent) {
|
|
802
|
+
return input.CoAgent;
|
|
803
|
+
}
|
|
804
|
+
if (input.CoAgentID) {
|
|
805
|
+
return (AIEngine.Instance.Agents ?? []).find(a => UUIDsEqual(a.ID, input.CoAgentID)) ?? null;
|
|
806
|
+
}
|
|
807
|
+
return null;
|
|
808
|
+
}
|
|
809
|
+
/**
|
|
810
|
+
* Resolves the realtime model for a session, honoring an explicit user choice when present.
|
|
811
|
+
*
|
|
812
|
+
* - With {@link PrepareClientSessionInput.PreferredModelID}: resolve THAT model strictly via
|
|
813
|
+
* {@link resolvePreferredRealtimeModel} — failures return a specific reason and never fall
|
|
814
|
+
* back to another model (the user explicitly chose). (The transport layer has already
|
|
815
|
+
* authorization-gated a deviating explicit choice.)
|
|
816
|
+
* - Else, with an effective-config `realtime.modelPreference` (name or id): resolve it via
|
|
817
|
+
* {@link resolveConfiguredModelPreference}. METADATA preferences degrade gracefully — an
|
|
818
|
+
* unsatisfiable preference logs and FALLS THROUGH to the default (mirroring the co-agent
|
|
819
|
+
* resolution chain's tolerant metadata steps), it never breaks calls.
|
|
820
|
+
* - Without either: the existing default behavior via {@link resolveRealtimeModel}
|
|
821
|
+
* (highest-PowerRank active Realtime model), with the generic {@link noModelMessage} on failure.
|
|
822
|
+
*
|
|
823
|
+
* @param input The prepare-session input (carries the optional preferred model id).
|
|
824
|
+
* @param coAgent The resolved co-agent (threaded to the default-resolution seam).
|
|
825
|
+
* @param effectiveConfig The resolved effective configuration (carries `modelPreference`).
|
|
826
|
+
* @returns The resolution outcome (resolution or failure reason).
|
|
827
|
+
*/
|
|
828
|
+
async resolveModelForSession(input, coAgent, effectiveConfig) {
|
|
829
|
+
if (input.PreferredModelID) {
|
|
830
|
+
return this.resolvePreferredRealtimeModel(input.PreferredModelID);
|
|
831
|
+
}
|
|
832
|
+
const fromConfig = this.resolveConfiguredModelPreference(effectiveConfig);
|
|
833
|
+
if (fromConfig) {
|
|
834
|
+
return { Resolution: fromConfig };
|
|
835
|
+
}
|
|
836
|
+
const resolution = await this.resolveRealtimeModel(coAgent);
|
|
837
|
+
return resolution ? { Resolution: resolution } : { ErrorMessage: this.noModelMessage() };
|
|
838
|
+
}
|
|
839
|
+
/**
|
|
840
|
+
* Resolves the effective config's `realtime.modelPreference` (an `MJ: AI Models` Name OR ID)
|
|
841
|
+
* into a usable realtime model. TOLERANT by design — this is a METADATA preference, so any
|
|
842
|
+
* failure (unknown model, inactive, wrong type, no vendor/key) logs a warning and returns
|
|
843
|
+
* `null`, falling through to the default highest-PowerRank resolution. Contrast with the
|
|
844
|
+
* explicit runtime choice ({@link resolvePreferredRealtimeModel}), which fails loud.
|
|
845
|
+
*
|
|
846
|
+
* @param effectiveConfig The resolved effective configuration.
|
|
847
|
+
* @returns The resolution, or `null` when no preference is configured or it can't be satisfied.
|
|
848
|
+
*/
|
|
849
|
+
resolveConfiguredModelPreference(effectiveConfig) {
|
|
850
|
+
const preference = effectiveConfig?.realtime?.modelPreference;
|
|
851
|
+
if (!preference) {
|
|
852
|
+
return null;
|
|
853
|
+
}
|
|
854
|
+
const model = this.findModelByIDOrName(preference);
|
|
855
|
+
if (!model) {
|
|
856
|
+
LogError(`RealtimeClientSessionService: configured realtime model preference '${preference}' matches no model in ` +
|
|
857
|
+
'AI model metadata — falling through to default realtime model resolution.');
|
|
858
|
+
return null;
|
|
859
|
+
}
|
|
860
|
+
if (!model.IsActive || !this.isRealtimeModel(model)) {
|
|
861
|
+
LogError(`RealtimeClientSessionService: configured realtime model preference '${preference}' resolved to ` +
|
|
862
|
+
`'${model.Name}' but it is not an Active Realtime model — falling through to default resolution.`);
|
|
863
|
+
return null;
|
|
864
|
+
}
|
|
865
|
+
const resolution = this.resolveVendorAndInstantiate(model);
|
|
866
|
+
if (!resolution) {
|
|
867
|
+
LogError(`RealtimeClientSessionService: configured realtime model preference '${model.Name}' has no usable ` +
|
|
868
|
+
'vendor DriverClass/API key — falling through to default resolution.');
|
|
869
|
+
}
|
|
870
|
+
return resolution;
|
|
871
|
+
}
|
|
872
|
+
/**
|
|
873
|
+
* Looks up a model by ID (UUID-insensitive) or, failing that, by case/whitespace-insensitive
|
|
874
|
+
* Name in {@link AIEngine}'s cached models. **Overridable seam**; tolerant of an unloaded cache.
|
|
875
|
+
*
|
|
876
|
+
* @param preference The `MJ: AI Models` ID or Name.
|
|
877
|
+
* @returns The model entity, or `null`.
|
|
878
|
+
*/
|
|
879
|
+
findModelByIDOrName(preference) {
|
|
880
|
+
try {
|
|
881
|
+
const models = AIEngine.Instance.Models ?? [];
|
|
882
|
+
const wanted = preference.trim().toLowerCase();
|
|
883
|
+
return (models.find(m => UUIDsEqual(m.ID, preference)) ??
|
|
884
|
+
models.find(m => m.Name?.trim().toLowerCase() === wanted) ??
|
|
885
|
+
null);
|
|
886
|
+
}
|
|
887
|
+
catch {
|
|
888
|
+
return null;
|
|
889
|
+
}
|
|
890
|
+
}
|
|
891
|
+
/**
|
|
892
|
+
* Strictly resolves an EXPLICITLY requested realtime model. Each precondition failure returns
|
|
893
|
+
* a clear, user-facing reason naming the model — there is NO fallback to another model, because
|
|
894
|
+
* the caller's user explicitly chose this one.
|
|
895
|
+
*
|
|
896
|
+
* @param preferredModelID The `MJ: AI Models.ID` the user chose.
|
|
897
|
+
* @returns The resolution outcome (resolution or a specific failure reason).
|
|
898
|
+
*/
|
|
899
|
+
resolvePreferredRealtimeModel(preferredModelID) {
|
|
900
|
+
const model = this.findModelByID(preferredModelID);
|
|
901
|
+
if (!model) {
|
|
902
|
+
return { ErrorMessage: `The requested realtime model (id '${preferredModelID}') was not found in AI model metadata.` };
|
|
903
|
+
}
|
|
904
|
+
if (!model.IsActive) {
|
|
905
|
+
return { ErrorMessage: `The requested model '${model.Name}' is not active and cannot be used for a voice session.` };
|
|
906
|
+
}
|
|
907
|
+
if (!this.isRealtimeModel(model)) {
|
|
908
|
+
return { ErrorMessage: `The requested model '${model.Name}' is not a Realtime model (its type is '${model.AIModelType}').` };
|
|
909
|
+
}
|
|
910
|
+
const resolution = this.resolveVendorAndInstantiate(model);
|
|
911
|
+
if (!resolution) {
|
|
912
|
+
return {
|
|
913
|
+
ErrorMessage: `The requested model '${model.Name}' has no active vendor with a usable DriverClass/API key ` +
|
|
914
|
+
'(e.g. AI_VENDOR_API_KEY__<driver>), so the voice session could not be started with it.'
|
|
915
|
+
};
|
|
916
|
+
}
|
|
917
|
+
return { Resolution: resolution };
|
|
918
|
+
}
|
|
919
|
+
/**
|
|
920
|
+
* Looks up a model by id in {@link AIEngine}'s cached models. **Overridable seam** for tests.
|
|
921
|
+
*
|
|
922
|
+
* @param modelID The `MJ: AI Models.ID` to find.
|
|
923
|
+
* @returns The model entity, or `null` when not present.
|
|
924
|
+
*/
|
|
925
|
+
findModelByID(modelID) {
|
|
926
|
+
return (AIEngine.Instance.Models ?? []).find(m => UUIDsEqual(m.ID, modelID)) ?? null;
|
|
927
|
+
}
|
|
928
|
+
/** True when the model's denormalized `AIModelType` name is `Realtime` (case/whitespace-insensitive). */
|
|
929
|
+
isRealtimeModel(model) {
|
|
930
|
+
return typeof model.AIModelType === 'string' && model.AIModelType.trim().toLowerCase() === 'realtime';
|
|
931
|
+
}
|
|
932
|
+
/**
|
|
933
|
+
* Resolves the Realtime model + vendor driver + API key, mirroring `BaseAgent`'s server-bridged
|
|
934
|
+
* resolution: highest-power active model of AIModelType `Realtime`; highest-priority active
|
|
935
|
+
* vendor whose `DriverClass` has a resolvable API key; instantiated via the `ClassFactory`.
|
|
936
|
+
*
|
|
937
|
+
* **Overridable seam.** Test subclasses override this to return a mock model so the service can
|
|
938
|
+
* be exercised without provider SDKs or DB metadata. Returns `null` (never throws) when any
|
|
939
|
+
* step can't be satisfied.
|
|
940
|
+
*
|
|
941
|
+
* @param coAgent The co-agent being voiced (reserved for future per-agent model preference).
|
|
942
|
+
* @returns The resolved model + identifiers, or `null`.
|
|
943
|
+
*/
|
|
944
|
+
async resolveRealtimeModel(coAgent) {
|
|
945
|
+
// Walk candidates in descending PowerRank, returning the FIRST that fully resolves to a usable
|
|
946
|
+
// client-direct driver (active vendor + API key + ClassFactory driver + SupportsClientDirect).
|
|
947
|
+
// Single-pick dead-ended whenever the highest-power model lacked a key or client-direct support
|
|
948
|
+
// — e.g. a newly-seeded provider (Grok/Inworld) with no env key outranking GPT Realtime — and
|
|
949
|
+
// surfaced "No usable Realtime model" instead of falling through to a model that works.
|
|
950
|
+
const candidates = this.selectRealtimeModelCandidates(coAgent);
|
|
951
|
+
for (const model of candidates) {
|
|
952
|
+
const resolution = this.resolveVendorAndInstantiate(model);
|
|
953
|
+
if (resolution && resolution.Model.SupportsClientDirect) {
|
|
954
|
+
return resolution;
|
|
955
|
+
}
|
|
956
|
+
}
|
|
957
|
+
return null;
|
|
958
|
+
}
|
|
959
|
+
/**
|
|
960
|
+
* Shared tail of model resolution: picks the vendor (with a usable API key) for an
|
|
961
|
+
* already-chosen model entity and instantiates its realtime driver.
|
|
962
|
+
*
|
|
963
|
+
* @param model The chosen model entity.
|
|
964
|
+
* @returns The full resolution, or `null` when no vendor/key/driver can be satisfied.
|
|
965
|
+
*/
|
|
966
|
+
resolveVendorAndInstantiate(model) {
|
|
967
|
+
const vendor = this.selectRealtimeVendor(model.ID);
|
|
968
|
+
if (!vendor) {
|
|
969
|
+
return null;
|
|
970
|
+
}
|
|
971
|
+
const apiKey = this.getAPIKeyForDriver(vendor.DriverClass);
|
|
972
|
+
if (!apiKey) {
|
|
973
|
+
return null;
|
|
974
|
+
}
|
|
975
|
+
const instance = this.createModelInstance(vendor.DriverClass, apiKey);
|
|
976
|
+
if (!instance) {
|
|
977
|
+
return null;
|
|
978
|
+
}
|
|
979
|
+
return {
|
|
980
|
+
Model: instance,
|
|
981
|
+
ModelID: model.ID,
|
|
982
|
+
VendorID: vendor.VendorID,
|
|
983
|
+
APIName: vendor.APIName,
|
|
984
|
+
ModelName: model.Name,
|
|
985
|
+
DriverClass: vendor.DriverClass
|
|
986
|
+
};
|
|
987
|
+
}
|
|
988
|
+
/**
|
|
989
|
+
* Resolves the API key for a vendor driver class. **Overridable seam** (wraps the module-level
|
|
990
|
+
* {@link GetAIAPIKey}) so tests can simulate present/absent keys without environment setup.
|
|
991
|
+
*
|
|
992
|
+
* @param driverClass The vendor's `DriverClass`.
|
|
993
|
+
* @returns The API key, or a falsy value when none is configured.
|
|
994
|
+
*/
|
|
995
|
+
getAPIKeyForDriver(driverClass) {
|
|
996
|
+
return GetAIAPIKey(driverClass) || undefined;
|
|
997
|
+
}
|
|
998
|
+
/**
|
|
999
|
+
* Instantiates the realtime driver for a vendor driver class via the ClassFactory.
|
|
1000
|
+
* **Overridable seam** so tests can return a mock driver.
|
|
1001
|
+
*
|
|
1002
|
+
* @param driverClass The vendor's `DriverClass` (the ClassFactory key).
|
|
1003
|
+
* @param apiKey The resolved API key (constructor argument).
|
|
1004
|
+
* @returns The driver instance, or `null` when the factory cannot create one.
|
|
1005
|
+
*/
|
|
1006
|
+
createModelInstance(driverClass, apiKey) {
|
|
1007
|
+
return MJGlobal.Instance.ClassFactory.CreateInstance(BaseRealtimeModel, driverClass, apiKey) ?? null;
|
|
1008
|
+
}
|
|
1009
|
+
/**
|
|
1010
|
+
* The active models of AIModelType `Realtime`, sorted highest-PowerRank first — the candidate
|
|
1011
|
+
* list {@link resolveRealtimeModel} walks until one yields a usable client-direct driver.
|
|
1012
|
+
* Returns ALL candidates (not just the top pick) so a keyless or non-client-direct top model
|
|
1013
|
+
* falls through to the next usable one instead of dead-ending the whole resolution.
|
|
1014
|
+
*
|
|
1015
|
+
* @param coAgent The co-agent (reserved for future per-agent model preference).
|
|
1016
|
+
* @returns The candidate models in resolution order (empty array when none are active).
|
|
1017
|
+
*/
|
|
1018
|
+
selectRealtimeModelCandidates(coAgent) {
|
|
1019
|
+
return AIEngine.Instance.Models
|
|
1020
|
+
.filter(m => m.IsActive && this.isRealtimeModel(m))
|
|
1021
|
+
.sort((a, b) => (b.PowerRank ?? 0) - (a.PowerRank ?? 0));
|
|
1022
|
+
}
|
|
1023
|
+
/**
|
|
1024
|
+
* Selects the highest-priority active vendor for a model whose `DriverClass` has a resolvable
|
|
1025
|
+
* API key. Mirrors `BaseAgent.selectRealtimeVendor`.
|
|
1026
|
+
*
|
|
1027
|
+
* @param modelID The chosen model's id.
|
|
1028
|
+
* @returns The vendor driver/api identifiers, or `null` when none has a usable key.
|
|
1029
|
+
*/
|
|
1030
|
+
selectRealtimeVendor(modelID) {
|
|
1031
|
+
const vendors = AIEngine.Instance.ModelVendors
|
|
1032
|
+
.filter(mv => UUIDsEqual(mv.ModelID, modelID) && mv.Status === 'Active' && mv.DriverClass != null)
|
|
1033
|
+
.sort((a, b) => (b.Priority ?? 0) - (a.Priority ?? 0));
|
|
1034
|
+
for (const v of vendors) {
|
|
1035
|
+
if (this.getAPIKeyForDriver(v.DriverClass)) {
|
|
1036
|
+
return { VendorID: v.VendorID ?? '', DriverClass: v.DriverClass, APIName: v.APIName ?? '' };
|
|
1037
|
+
}
|
|
1038
|
+
}
|
|
1039
|
+
return null;
|
|
1040
|
+
}
|
|
1041
|
+
/**
|
|
1042
|
+
* Resolves the DB-driven progress-narration instruction template: the Active `MJ: AI Prompts`
|
|
1043
|
+
* row named {@link RealtimeClientSessionService.NarrationPromptName}, read from
|
|
1044
|
+
* {@link AIEngine}'s cached prompts. When the current name is absent, falls back to the
|
|
1045
|
+
* DEPRECATED {@link RealtimeClientSessionService.LegacyNarrationPromptName} (pre-rename seed)
|
|
1046
|
+
* with a deprecation log. **Tolerant**: returns `null` (never throws) when neither prompt is
|
|
1047
|
+
* present, the text is empty, or the engine cache is unavailable — clients fall back to their
|
|
1048
|
+
* built-in narration instruction text.
|
|
1049
|
+
*
|
|
1050
|
+
* @returns The template text (containing a `{{ progressMessage }}` placeholder), or `null`.
|
|
1051
|
+
*/
|
|
1052
|
+
resolveNarrationInstructionsTemplate() {
|
|
1053
|
+
return ResolveNarrationInstructionsTemplate();
|
|
1054
|
+
}
|
|
1055
|
+
/**
|
|
1056
|
+
* Builds the {@link RealtimeSessionParams} for the client-direct session: the companion system
|
|
1057
|
+
* prompt plus the stable, target-independent tool set.
|
|
1058
|
+
*
|
|
1059
|
+
* @param input The prepare-session input.
|
|
1060
|
+
* @param coAgent The resolved co-agent.
|
|
1061
|
+
* @param modelApiName The vendor API name of the resolved realtime model.
|
|
1062
|
+
* @param contextUser The calling user.
|
|
1063
|
+
* @param provider The request-scoped metadata provider.
|
|
1064
|
+
* @param effectiveConfig The resolved effective configuration (voice persona + provider settings).
|
|
1065
|
+
* @param driverClass The resolved vendor's DriverClass — matches per-provider voice settings.
|
|
1066
|
+
* @returns The assembled session params.
|
|
1067
|
+
*/
|
|
1068
|
+
async buildSessionParams(input, coAgent, modelApiName, contextUser, provider, effectiveConfig, driverClass) {
|
|
1069
|
+
const systemPrompt = await this.buildCompanionSystemPrompt(input, coAgent, contextUser, provider, effectiveConfig);
|
|
1070
|
+
const memoryContext = await this.assembleMemoryContext(input, coAgent, contextUser);
|
|
1071
|
+
const tools = this.buildStableToolSet(input.ExtraTools);
|
|
1072
|
+
// One line per mint: confirms which tools + whether the channel-direct framing actually reach
|
|
1073
|
+
// the model — settles "why does the co-agent delegate instead of calling browser_*" without
|
|
1074
|
+
// runtime guesswork (channelExceptionInPrompt=false ⇒ stale build; browser_* missing from
|
|
1075
|
+
// tools ⇒ the channel's tools never reached the mint).
|
|
1076
|
+
console.log(`[RealtimeCoAgent] mint model=${modelApiName} ` +
|
|
1077
|
+
`tools=[${tools.map(t => t.Name).join(', ')}] ` +
|
|
1078
|
+
`channelExceptionInPrompt=${systemPrompt.includes('interactive-surface')}`);
|
|
1079
|
+
return {
|
|
1080
|
+
Model: modelApiName,
|
|
1081
|
+
SystemPrompt: systemPrompt,
|
|
1082
|
+
Tools: tools,
|
|
1083
|
+
InitialContext: memoryContext || undefined,
|
|
1084
|
+
Config: this.buildSessionConfigBag(input, effectiveConfig, driverClass)
|
|
1085
|
+
};
|
|
1086
|
+
}
|
|
1087
|
+
/**
|
|
1088
|
+
* Builds the provider-pact `Config` bag for the session: the effective config's matching
|
|
1089
|
+
* per-provider voice settings (`realtime.voice.providers.<provider>`) merged UNDER any
|
|
1090
|
+
* caller-supplied {@link PrepareClientSessionInput.Config} (the runtime bag wins per key).
|
|
1091
|
+
* The settings objects are OPAQUE driver pacts — each server driver consumes its own keys
|
|
1092
|
+
* exactly as it consumes any other entry of the open config bag (OpenAI spreads it into
|
|
1093
|
+
* `session.update`, AssemblyAI reads `voice`, Gemini merges it last). Returns the original
|
|
1094
|
+
* `input.Config` (possibly `undefined`) when no provider settings match, preserving the
|
|
1095
|
+
* pre-config behavior byte-for-byte.
|
|
1096
|
+
*
|
|
1097
|
+
* @param input The prepare-session input (carries the runtime config bag).
|
|
1098
|
+
* @param effectiveConfig The resolved effective configuration.
|
|
1099
|
+
* @param driverClass The resolved vendor's DriverClass.
|
|
1100
|
+
* @returns The merged config bag, or `undefined` when nothing contributes.
|
|
1101
|
+
*/
|
|
1102
|
+
buildSessionConfigBag(input, effectiveConfig, driverClass) {
|
|
1103
|
+
const providerVoice = GetProviderVoiceSettings(effectiveConfig, driverClass ?? null);
|
|
1104
|
+
let bag = providerVoice
|
|
1105
|
+
? DeepMergeConfigs(providerVoice, input.Config)
|
|
1106
|
+
: input.Config;
|
|
1107
|
+
// Multi-agent meeting: carry the host-NEUTRAL disable-auto-response flag in the open config bag so
|
|
1108
|
+
// each provider translates it its own way (OpenAI → turn_detection.create_response=false) — the
|
|
1109
|
+
// bridge becomes the sole speech trigger. Absent ⇒ byte-for-byte the prior 1:1 behavior.
|
|
1110
|
+
if (input.DisableAutoResponse) {
|
|
1111
|
+
bag = { ...(bag ?? {}), disableAutoResponse: true };
|
|
1112
|
+
}
|
|
1113
|
+
return bag;
|
|
1114
|
+
}
|
|
1115
|
+
/**
|
|
1116
|
+
* Assembles the companion system prompt: the framing ("you are the voice for the target"), the
|
|
1117
|
+
* co-agent's own system prompt text, the TARGET agent's identity/capabilities (Name +
|
|
1118
|
+
* Description), the conversation history, and the same memory/context a loop agent assembles.
|
|
1119
|
+
*
|
|
1120
|
+
* When the effective configuration carries a voice persona (`realtime.voice.default`), a
|
|
1121
|
+
* short "Voice & manner" section (tone / speaking style) is appended after the co-agent's
|
|
1122
|
+
* own prompt so the model speaks in the configured manner.
|
|
1123
|
+
*
|
|
1124
|
+
* @param input The prepare-session input.
|
|
1125
|
+
* @param coAgent The resolved co-agent.
|
|
1126
|
+
* @param contextUser The calling user.
|
|
1127
|
+
* @param provider The request-scoped metadata provider.
|
|
1128
|
+
* @param effectiveConfig The resolved effective configuration (voice persona source).
|
|
1129
|
+
* @returns The concatenated system prompt (never empty — the framing is always present).
|
|
1130
|
+
*/
|
|
1131
|
+
async buildCompanionSystemPrompt(input, coAgent, contextUser, provider, effectiveConfig) {
|
|
1132
|
+
const target = this.resolveTargetAgent(input.TargetAgentID);
|
|
1133
|
+
const targetName = target?.Name ?? 'the configured target agent';
|
|
1134
|
+
// Identity framing comes from the ONE shared producer (see BuildRealtimeAgentFraming) so the agent
|
|
1135
|
+
// is the same agent on every host. The interactive-surface clause is host-specific (native chat's
|
|
1136
|
+
// browser/whiteboard); bridges pass none.
|
|
1137
|
+
const framing = BuildRealtimeAgentFraming(targetName, this.buildInteractiveSurfaceFraming(input.ExtraTools));
|
|
1138
|
+
const meetingFraming = this.buildMeetingFraming(input);
|
|
1139
|
+
const coAgentPrompt = this.getCoAgentSystemPromptText(coAgent);
|
|
1140
|
+
const voiceManner = BuildVoiceMannerSection(effectiveConfig);
|
|
1141
|
+
const targetIdentity = this.formatTargetIdentity(target);
|
|
1142
|
+
const priorTranscript = this.formatPriorTranscript(input.PriorTranscript);
|
|
1143
|
+
const history = this.formatConversationHistory(input.ConversationMessages);
|
|
1144
|
+
const memoryContext = await this.assembleMemoryContext(input, coAgent, contextUser);
|
|
1145
|
+
return [framing, meetingFraming, coAgentPrompt, voiceManner, targetIdentity, priorTranscript, history, memoryContext]
|
|
1146
|
+
.filter(part => part && part.trim().length > 0)
|
|
1147
|
+
.join('\n\n');
|
|
1148
|
+
}
|
|
1149
|
+
/**
|
|
1150
|
+
* Builds the **meeting-mode** discipline clause — present only for a multi-agent meeting session
|
|
1151
|
+
* ({@link PrepareClientSessionInput.DisableAutoResponse}). It tells the agent to hear the whole
|
|
1152
|
+
* conversation but speak only when addressed (named) or clearly called on, and never to talk over
|
|
1153
|
+
* others. This is the *prompt* half of "hear always, speak selectively"; the enforcement half is the
|
|
1154
|
+
* model's disabled auto-response + the bridge's addressing gate. Empty for a 1:1 call (prompt unchanged).
|
|
1155
|
+
* See `plans/realtime/multi-agent-meeting-turn-taking.md`.
|
|
1156
|
+
*
|
|
1157
|
+
* @param input The prepare-session input (carries the meeting flag + self names).
|
|
1158
|
+
* @returns The meeting clause, or `''` for a non-meeting session.
|
|
1159
|
+
*/
|
|
1160
|
+
buildMeetingFraming(input) {
|
|
1161
|
+
if (!input.DisableAutoResponse) {
|
|
1162
|
+
return '';
|
|
1163
|
+
}
|
|
1164
|
+
const names = (input.SelfNames ?? []).map(n => n.trim()).filter(n => n.length > 0);
|
|
1165
|
+
const addressed = names.length > 0
|
|
1166
|
+
? `You are addressed when someone says your name (${names.join(', ')}) or clearly directs a question at you.`
|
|
1167
|
+
: `You are addressed when someone clearly directs a question at you.`;
|
|
1168
|
+
return (`MEETING MODE: You are one of several participants (people and other agents) in a live meeting. ` +
|
|
1169
|
+
`LISTEN to the whole conversation, but do NOT respond to every utterance — speak only when it is your turn. ` +
|
|
1170
|
+
`${addressed} When you are not addressed, stay silent and keep listening; never talk over others or answer ` +
|
|
1171
|
+
`a question meant for someone else. Let people finish before you respond, and keep your replies brief.`);
|
|
1172
|
+
}
|
|
1173
|
+
/**
|
|
1174
|
+
* Builds the "interactive-surface tools" exception clause appended to the co-agent framing when
|
|
1175
|
+
* the client supplied channel tools (browser_*, Whiteboard_*, …) as ExtraTools. Without it the
|
|
1176
|
+
* co-agent — told to route ALL work through invoke-target-agent — delegates browser/whiteboard
|
|
1177
|
+
* requests to the target agent (which has no live channel of its own) instead of driving the
|
|
1178
|
+
* surface itself, then hallucinates a "missing session id". The tools ARE already in its set
|
|
1179
|
+
* ({@link buildStableToolSet} merges `[invokeTarget, ...extraTools]`); this clause tells the model
|
|
1180
|
+
* to USE them directly. Returns empty for pure-voice sessions (no ExtraTools), keeping that
|
|
1181
|
+
* framing untouched. Generic by design — it names browser_ and Whiteboard_ tools only as
|
|
1182
|
+
* examples, so any future client channel is covered automatically.
|
|
1183
|
+
*
|
|
1184
|
+
* @param extraTools The client-supplied channel tools, when any.
|
|
1185
|
+
* @returns The exception clause (leading space included), or '' when there are no extra tools.
|
|
1186
|
+
*/
|
|
1187
|
+
buildInteractiveSurfaceFraming(extraTools) {
|
|
1188
|
+
if (!extraTools || extraTools.length === 0) {
|
|
1189
|
+
return '';
|
|
1190
|
+
}
|
|
1191
|
+
return ` ONE EXCEPTION: besides '${INVOKE_TARGET_AGENT_TOOL_NAME}' you have been given ` +
|
|
1192
|
+
`interactive-surface tools (for example 'browser_*' to drive a LIVE web browser the user can ` +
|
|
1193
|
+
`watch, or 'Whiteboard_*' to draw on a shared board). Those surfaces are operated by YOU, ` +
|
|
1194
|
+
`directly — when the user asks to use one (e.g. "open/show a browser", "go to a site", "add ` +
|
|
1195
|
+
`to the whiteboard"), call the matching tool yourself immediately and narrate what you're ` +
|
|
1196
|
+
`doing. NEVER route an interactive-surface request through '${INVOKE_TARGET_AGENT_TOOL_NAME}', ` +
|
|
1197
|
+
`and never claim you lack a session — calling the tool is all that's needed.`;
|
|
1198
|
+
}
|
|
1199
|
+
/**
|
|
1200
|
+
* Frames the prior-leg transcript (when a session resumes via `lastSessionId`) as a clearly
|
|
1201
|
+
* labeled PRIOR-CONVERSATION section of the system prompt, so the model REMEMBERS the last
|
|
1202
|
+
* live session rather than greeting the user cold. The transport layer supplies the
|
|
1203
|
+
* already-capped, role-tagged lines (see {@link PrepareClientSessionInput.PriorTranscript});
|
|
1204
|
+
* this method only adds the framing. Empty/whitespace input yields an empty section.
|
|
1205
|
+
*
|
|
1206
|
+
* @param priorTranscript The role-tagged transcript lines, or undefined.
|
|
1207
|
+
* @returns The framed section, or empty string when there is nothing to frame.
|
|
1208
|
+
*/
|
|
1209
|
+
formatPriorTranscript(priorTranscript) {
|
|
1210
|
+
const text = priorTranscript?.trim() ?? '';
|
|
1211
|
+
if (text.length === 0) {
|
|
1212
|
+
return '';
|
|
1213
|
+
}
|
|
1214
|
+
return ('Earlier in this conversation (a previous live session that you are now resuming), ' +
|
|
1215
|
+
'you and the user discussed the following. Treat it as shared context you both remember:\n' +
|
|
1216
|
+
text);
|
|
1217
|
+
}
|
|
1218
|
+
/**
|
|
1219
|
+
* Resolves the target agent entity from cached metadata.
|
|
1220
|
+
*
|
|
1221
|
+
* @param targetAgentID The target agent id.
|
|
1222
|
+
* @returns The target agent entity, or `null` when not found.
|
|
1223
|
+
*/
|
|
1224
|
+
resolveTargetAgent(targetAgentID) {
|
|
1225
|
+
if (!targetAgentID) {
|
|
1226
|
+
return null;
|
|
1227
|
+
}
|
|
1228
|
+
return (AIEngine.Instance.Agents ?? []).find(a => UUIDsEqual(a.ID, targetAgentID)) ?? null;
|
|
1229
|
+
}
|
|
1230
|
+
/**
|
|
1231
|
+
* Reads the co-agent's own system prompt text from its highest-priority active agent prompt,
|
|
1232
|
+
* mirroring `BaseAgent.loadAgentConfiguration`'s child-prompt resolution.
|
|
1233
|
+
*
|
|
1234
|
+
* @param coAgent The resolved co-agent.
|
|
1235
|
+
* @returns The co-agent's system prompt template text, or empty string when none is configured.
|
|
1236
|
+
*/
|
|
1237
|
+
getCoAgentSystemPromptText(coAgent) {
|
|
1238
|
+
return this.resolveCoAgentSystemPrompt(coAgent).Text;
|
|
1239
|
+
}
|
|
1240
|
+
/**
|
|
1241
|
+
* Resolves the co-agent's highest-priority active system prompt, returning both its template
|
|
1242
|
+
* text and its prompt id. The id is surfaced so {@link PrepareClientSession} can create a linked
|
|
1243
|
+
* co-agent `AIPromptRun` for observability. Mirrors `BaseAgent.loadAgentConfiguration`'s
|
|
1244
|
+
* child-prompt resolution.
|
|
1245
|
+
*
|
|
1246
|
+
* @param coAgent The resolved co-agent.
|
|
1247
|
+
* @returns The prompt text + id, or `{ Text: '', PromptID: null }` when none is configured.
|
|
1248
|
+
*/
|
|
1249
|
+
resolveCoAgentSystemPrompt(coAgent) {
|
|
1250
|
+
const engine = AIEngine.Instance;
|
|
1251
|
+
const agentPrompt = (engine.AgentPrompts ?? [])
|
|
1252
|
+
.filter(ap => UUIDsEqual(ap.AgentID, coAgent.ID) && ap.Status === 'Active')
|
|
1253
|
+
.sort((a, b) => a.ExecutionOrder - b.ExecutionOrder)[0];
|
|
1254
|
+
if (!agentPrompt) {
|
|
1255
|
+
return { Text: '', PromptID: null };
|
|
1256
|
+
}
|
|
1257
|
+
const prompt = (engine.Prompts ?? []).find(p => UUIDsEqual(p.ID, agentPrompt.PromptID));
|
|
1258
|
+
return { Text: prompt?.TemplateText ?? '', PromptID: prompt?.ID ?? null };
|
|
1259
|
+
}
|
|
1260
|
+
/**
|
|
1261
|
+
* Formats the target agent's identity + capabilities block for the system prompt.
|
|
1262
|
+
*
|
|
1263
|
+
* @param target The target agent, or `null`.
|
|
1264
|
+
* @returns The formatted block, or empty string when no target resolved.
|
|
1265
|
+
*/
|
|
1266
|
+
formatTargetIdentity(target) {
|
|
1267
|
+
if (!target) {
|
|
1268
|
+
return '';
|
|
1269
|
+
}
|
|
1270
|
+
const description = target.Description?.trim() ? target.Description.trim() : 'No description provided.';
|
|
1271
|
+
return `Target agent you are voicing for:\nName: ${target.Name}\nCapabilities: ${description}`;
|
|
1272
|
+
}
|
|
1273
|
+
/**
|
|
1274
|
+
* Formats prior conversation history as a plain-text block for the system prompt.
|
|
1275
|
+
*
|
|
1276
|
+
* @param messages The conversation messages, or undefined.
|
|
1277
|
+
* @returns The formatted history block, or empty string when there is none.
|
|
1278
|
+
*/
|
|
1279
|
+
formatConversationHistory(messages) {
|
|
1280
|
+
if (!messages || messages.length === 0) {
|
|
1281
|
+
return '';
|
|
1282
|
+
}
|
|
1283
|
+
const lines = messages
|
|
1284
|
+
.map(m => {
|
|
1285
|
+
const text = typeof m.content === 'string' ? m.content : '';
|
|
1286
|
+
return text.trim().length > 0 ? `${m.role}: ${text}` : '';
|
|
1287
|
+
})
|
|
1288
|
+
.filter(line => line.length > 0);
|
|
1289
|
+
return lines.length > 0 ? `Conversation so far:\n${lines.join('\n')}` : '';
|
|
1290
|
+
}
|
|
1291
|
+
/**
|
|
1292
|
+
* Assembles the same memory/context block a loop agent injects, reusing
|
|
1293
|
+
* {@link AgentMemoryContextBuilder} so there is no duplicated retrieval logic. The builder
|
|
1294
|
+
* unshifts a system message onto a throwaway array, which we pull back out as plain text.
|
|
1295
|
+
*
|
|
1296
|
+
* @param input The prepare-session input.
|
|
1297
|
+
* @param coAgent The resolved co-agent.
|
|
1298
|
+
* @param contextUser The calling user.
|
|
1299
|
+
* @returns The concatenated context text (empty string when nothing was injected).
|
|
1300
|
+
*/
|
|
1301
|
+
async assembleMemoryContext(input, coAgent, contextUser) {
|
|
1302
|
+
const lastUserMessage = (input.ConversationMessages ?? []).filter(m => m.role === 'user').pop();
|
|
1303
|
+
const inputText = typeof lastUserMessage?.content === 'string' ? lastUserMessage.content : '';
|
|
1304
|
+
const scratch = [];
|
|
1305
|
+
const builder = new AgentMemoryContextBuilder();
|
|
1306
|
+
await builder.InjectContextMemory(inputText, coAgent, input.UserID || contextUser?.ID, input.CompanyID, contextUser, scratch, undefined, undefined, undefined, null);
|
|
1307
|
+
return scratch
|
|
1308
|
+
.map(m => (typeof m.content === 'string' ? m.content : ''))
|
|
1309
|
+
.filter(c => c.length > 0)
|
|
1310
|
+
.join('\n\n');
|
|
1311
|
+
}
|
|
1312
|
+
/**
|
|
1313
|
+
* Builds the stable, target-independent tool set every voice session exposes: the single
|
|
1314
|
+
* `invoke-target-agent` tool plus any caller-supplied extra tools. The target is a runtime
|
|
1315
|
+
* argument *inside* the call, never a per-target tool — this keeps the provider contract
|
|
1316
|
+
* identical across targets.
|
|
1317
|
+
*
|
|
1318
|
+
* @param extraTools Optional additional target-independent tools.
|
|
1319
|
+
* @returns The tools to register at session start.
|
|
1320
|
+
*/
|
|
1321
|
+
buildStableToolSet(extraTools) {
|
|
1322
|
+
const invokeTarget = {
|
|
1323
|
+
Name: INVOKE_TARGET_AGENT_TOOL_NAME,
|
|
1324
|
+
Description: 'Hand the user\'s request to the target agent to perform the actual work. Call this whenever ' +
|
|
1325
|
+
'real work (data lookup, analysis, actions) is required, then narrate progress while it runs.',
|
|
1326
|
+
ParametersSchema: {
|
|
1327
|
+
type: 'object',
|
|
1328
|
+
properties: {
|
|
1329
|
+
request: {
|
|
1330
|
+
type: 'string',
|
|
1331
|
+
description: 'The natural-language request to hand to the target agent.'
|
|
1332
|
+
}
|
|
1333
|
+
},
|
|
1334
|
+
required: ['request']
|
|
1335
|
+
}
|
|
1336
|
+
};
|
|
1337
|
+
return extraTools && extraTools.length > 0 ? [invokeTarget, ...extraTools] : [invokeTarget];
|
|
1338
|
+
}
|
|
1339
|
+
/**
|
|
1340
|
+
* Builds the {@link RealtimeToolBroker} for a relayed tool call, wiring `DelegateToTarget` to a
|
|
1341
|
+
* target-agent run and `ExecuteTool` to a structured "not available" placeholder.
|
|
1342
|
+
*
|
|
1343
|
+
* @param input The relayed tool input.
|
|
1344
|
+
* @param contextUser The calling user.
|
|
1345
|
+
* @param provider The request-scoped metadata provider.
|
|
1346
|
+
* @returns The constructed broker.
|
|
1347
|
+
*/
|
|
1348
|
+
buildToolBroker(input, contextUser, provider) {
|
|
1349
|
+
const deps = {
|
|
1350
|
+
DelegateToTarget: (request) => this.delegateToTarget(input, request, contextUser, provider),
|
|
1351
|
+
ExecuteTool: (call) => this.executeNonTargetTool(call)
|
|
1352
|
+
};
|
|
1353
|
+
return new RealtimeToolBroker(deps);
|
|
1354
|
+
}
|
|
1355
|
+
/**
|
|
1356
|
+
* Delegates an `invoke-target-agent` call to the target agent via {@link AgentRunner.RunAgent}.
|
|
1357
|
+
*
|
|
1358
|
+
* Threads the broker-owned abort signal (combined with any caller signal) into the child run's
|
|
1359
|
+
* `cancellationToken`, links the child run to the co-agent run via `parentRunID`, and propagates
|
|
1360
|
+
* `agentSessionID` so both runs group under the same session. Mirrors
|
|
1361
|
+
* `BaseAgent.delegateRealtimeToTarget`.
|
|
1362
|
+
*
|
|
1363
|
+
* @param input The relayed tool input (target id + linkage).
|
|
1364
|
+
* @param request The broker's delegation request (call id + arguments + abort signal).
|
|
1365
|
+
* @param contextUser The calling user.
|
|
1366
|
+
* @param provider The request-scoped metadata provider.
|
|
1367
|
+
* @returns The delegated result for the model's tool_response.
|
|
1368
|
+
*/
|
|
1369
|
+
async delegateToTarget(input, request, contextUser, provider) {
|
|
1370
|
+
const target = this.resolveTargetAgent(input.TargetAgentID);
|
|
1371
|
+
if (!target) {
|
|
1372
|
+
return {
|
|
1373
|
+
CallID: request.CallID,
|
|
1374
|
+
Success: false,
|
|
1375
|
+
Output: 'No target agent is configured for this voice session, so the request could not be performed.'
|
|
1376
|
+
};
|
|
1377
|
+
}
|
|
1378
|
+
try {
|
|
1379
|
+
const result = await this.runDelegatedAgent(input, request, target, contextUser, provider);
|
|
1380
|
+
const artifacts = await this.createDelegatedRunArtifacts(result, contextUser, provider);
|
|
1381
|
+
return this.buildDelegatedResult(request.CallID, result, artifacts);
|
|
1382
|
+
}
|
|
1383
|
+
catch (error) {
|
|
1384
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
1385
|
+
return { CallID: request.CallID, Success: false, Output: `Delegation failed: ${message}` };
|
|
1386
|
+
}
|
|
1387
|
+
}
|
|
1388
|
+
/**
|
|
1389
|
+
* Creates artifact(s) from a completed delegated run's payload — the voice-path equivalent of
|
|
1390
|
+
* the chat path's artifact step in `AgentRunner.RunAgentInConversation`. Delegated voice runs
|
|
1391
|
+
* execute via `AgentRunner.RunAgent` directly (no conversation detail), so without this step
|
|
1392
|
+
* they would never produce artifacts at all.
|
|
1393
|
+
*
|
|
1394
|
+
* Eligibility guards (all must hold, mirroring the chat path's `processArtifacts`):
|
|
1395
|
+
* - the run succeeded and did NOT pause awaiting feedback (a paused run has no deliverable yet);
|
|
1396
|
+
* - the run returned a non-empty payload.
|
|
1397
|
+
*
|
|
1398
|
+
* The DB work is delegated to {@link processRunArtifacts} (an overridable seam), which reuses
|
|
1399
|
+
* `AgentRunner.ProcessAgentArtifacts` — so ArtifactCreationMode, DefaultArtifactTypeID,
|
|
1400
|
+
* name extraction, and duplicate-version dedup all behave exactly as in chat. **Best-effort:**
|
|
1401
|
+
* any failure is logged and returns `undefined`; artifact surfacing never fails the delegation.
|
|
1402
|
+
*
|
|
1403
|
+
* @param result The delegated agent execution result.
|
|
1404
|
+
* @param contextUser The calling user.
|
|
1405
|
+
* @param provider The request-scoped metadata provider.
|
|
1406
|
+
* @returns The produced artifact descriptor(s), or `undefined` when none were created.
|
|
1407
|
+
*/
|
|
1408
|
+
async createDelegatedRunArtifacts(result, contextUser, provider) {
|
|
1409
|
+
const paused = result.agentRun?.Status === 'AwaitingFeedback';
|
|
1410
|
+
const payload = result.payload;
|
|
1411
|
+
const hasPayload = payload != null && Object.keys(payload).length > 0;
|
|
1412
|
+
if (!result.success || paused || !hasPayload) {
|
|
1413
|
+
return undefined;
|
|
1414
|
+
}
|
|
1415
|
+
try {
|
|
1416
|
+
return await this.processRunArtifacts(result, contextUser, provider);
|
|
1417
|
+
}
|
|
1418
|
+
catch (error) {
|
|
1419
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
1420
|
+
LogError(`RealtimeClientSessionService.createDelegatedRunArtifacts failed (delegation continues): ${message}`);
|
|
1421
|
+
return undefined;
|
|
1422
|
+
}
|
|
1423
|
+
}
|
|
1424
|
+
/**
|
|
1425
|
+
* The DB-backed artifact-creation seam: runs `AgentRunner.ProcessAgentArtifacts` WITHOUT a
|
|
1426
|
+
* conversation detail (the voice path has none — the artifact + version are created and the
|
|
1427
|
+
* junction link is skipped), then loads the artifact header for its display name.
|
|
1428
|
+
*
|
|
1429
|
+
* Artifacts whose Visibility resolved to `System Only` (the agent's ArtifactCreationMode) are
|
|
1430
|
+
* created but NOT surfaced to the overlay — matching how chat hides them from users.
|
|
1431
|
+
*
|
|
1432
|
+
* **Overridable seam** so tests can exercise {@link createDelegatedRunArtifacts}' eligibility
|
|
1433
|
+
* guards without a DB.
|
|
1434
|
+
*
|
|
1435
|
+
* @param result The delegated agent execution result (payload + agentRun).
|
|
1436
|
+
* @param contextUser The calling user.
|
|
1437
|
+
* @param provider The request-scoped metadata provider.
|
|
1438
|
+
* @returns The produced artifact descriptor(s), or `undefined`.
|
|
1439
|
+
*/
|
|
1440
|
+
async processRunArtifacts(result, contextUser, provider) {
|
|
1441
|
+
const runner = new AgentRunner(provider);
|
|
1442
|
+
const info = await runner.ProcessAgentArtifacts(result, undefined, undefined, contextUser, provider);
|
|
1443
|
+
if (!info) {
|
|
1444
|
+
return undefined;
|
|
1445
|
+
}
|
|
1446
|
+
const artifact = await provider.GetEntityObject('MJ: Artifacts', contextUser);
|
|
1447
|
+
if (!(await artifact.Load(info.artifactId))) {
|
|
1448
|
+
return undefined;
|
|
1449
|
+
}
|
|
1450
|
+
if (artifact.Visibility === 'System Only') {
|
|
1451
|
+
return undefined; // created for system purposes, never user-surfaced
|
|
1452
|
+
}
|
|
1453
|
+
return [{ ArtifactID: info.artifactId, ArtifactVersionID: info.versionId, Name: artifact.Name }];
|
|
1454
|
+
}
|
|
1455
|
+
/**
|
|
1456
|
+
* Runs (or resumes) the target agent for a delegation. Threads the combined abort signal, parent
|
|
1457
|
+
* run linkage, session id, and the `OnProgress` callback so the resolver can stream progress.
|
|
1458
|
+
* When {@link ExecuteRelayedToolInput.ResumeRunID} is set, resumes that paused run via
|
|
1459
|
+
* `lastRunId` + `autoPopulateLastRunPayload` (the user's answer continues the same interactive
|
|
1460
|
+
* run) instead of starting fresh.
|
|
1461
|
+
*
|
|
1462
|
+
* @param input The relayed tool input (linkage, progress callback, optional resume id).
|
|
1463
|
+
* @param request The broker's delegation request (call id + arguments + abort signal).
|
|
1464
|
+
* @param target The resolved target agent.
|
|
1465
|
+
* @param contextUser The calling user.
|
|
1466
|
+
* @param provider The request-scoped metadata provider.
|
|
1467
|
+
* @returns The agent execution result.
|
|
1468
|
+
*/
|
|
1469
|
+
async runDelegatedAgent(input, request, target, contextUser, provider) {
|
|
1470
|
+
const requestText = this.parseDelegateRequestText(request.Arguments);
|
|
1471
|
+
const parentRun = await this.loadParentRun(input.ParentRunID, contextUser, provider);
|
|
1472
|
+
const runner = new AgentRunner(provider);
|
|
1473
|
+
return runner.RunAgent({
|
|
1474
|
+
agent: target,
|
|
1475
|
+
conversationMessages: [{ role: 'user', content: requestText }],
|
|
1476
|
+
contextUser,
|
|
1477
|
+
provider,
|
|
1478
|
+
cancellationToken: this.combineSignals(request.AbortSignal, input.AbortSignal),
|
|
1479
|
+
parentRun: parentRun ?? undefined,
|
|
1480
|
+
agentSessionID: input.AgentSessionID,
|
|
1481
|
+
onProgress: input.OnProgress,
|
|
1482
|
+
lastRunId: input.ResumeRunID,
|
|
1483
|
+
autoPopulateLastRunPayload: input.ResumeRunID ? true : undefined
|
|
1484
|
+
});
|
|
1485
|
+
}
|
|
1486
|
+
/**
|
|
1487
|
+
* Maps an {@link ExecuteAgentResult} onto the broker's {@link DelegatedResult}, special-casing a
|
|
1488
|
+
* run that paused awaiting feedback. An `AwaitingFeedback` run is a valid intermediate outcome,
|
|
1489
|
+
* not an error: we return its clarifying QUESTION (the run's `Message`) as the tool Output —
|
|
1490
|
+
* phrased so the realtime model relays it as a question to the user — set `Success: true`, and
|
|
1491
|
+
* surface the paused run id so the resolver can resume that run on the user's next answer.
|
|
1492
|
+
*
|
|
1493
|
+
* @param callID The provider call id this result corresponds to.
|
|
1494
|
+
* @param result The agent execution result.
|
|
1495
|
+
* @param artifacts Artifacts the run produced (from {@link createDelegatedRunArtifacts}),
|
|
1496
|
+
* threaded into the result so the broker serializes them for the call overlay.
|
|
1497
|
+
* @returns The delegated result for the model's tool_response.
|
|
1498
|
+
*/
|
|
1499
|
+
buildDelegatedResult(callID, result, artifacts) {
|
|
1500
|
+
if (result.agentRun?.Status === 'AwaitingFeedback') {
|
|
1501
|
+
const question = result.agentRun.Message?.trim()
|
|
1502
|
+
|| 'The target agent needs more information to continue.';
|
|
1503
|
+
return {
|
|
1504
|
+
CallID: callID,
|
|
1505
|
+
Success: true,
|
|
1506
|
+
Output: `You need an answer from the user before you can continue this work. Ask them, in your own first-person voice: ${question}`,
|
|
1507
|
+
PausedRunID: result.agentRun.ID,
|
|
1508
|
+
RunID: result.agentRun.ID
|
|
1509
|
+
};
|
|
1510
|
+
}
|
|
1511
|
+
return {
|
|
1512
|
+
CallID: callID,
|
|
1513
|
+
Success: result.success,
|
|
1514
|
+
Output: result.success
|
|
1515
|
+
? (result.agentRun?.Message || 'The delegated work is complete. Share the outcome with the user in your own first-person voice.')
|
|
1516
|
+
: (result.agentRun?.ErrorMessage || 'The work could not be completed. Tell the user, in first person, that you hit a problem and offer a next step.'),
|
|
1517
|
+
RunID: result.agentRun?.ID,
|
|
1518
|
+
Artifacts: artifacts
|
|
1519
|
+
};
|
|
1520
|
+
}
|
|
1521
|
+
/**
|
|
1522
|
+
* Loads the co-agent run entity behind {@link ExecuteRelayedToolInput.ParentRunID} so the
|
|
1523
|
+
* delegated run can link to it via `parentRun` (→ `ParentRunID`). Returns `null` when no id was
|
|
1524
|
+
* supplied or the run cannot be loaded (delegation proceeds without parent linkage rather than
|
|
1525
|
+
* failing the whole call).
|
|
1526
|
+
*
|
|
1527
|
+
* @param parentRunID The co-agent run id, or undefined.
|
|
1528
|
+
* @param contextUser The calling user.
|
|
1529
|
+
* @param provider The request-scoped metadata provider.
|
|
1530
|
+
* @returns The loaded parent run entity, or `null`.
|
|
1531
|
+
*/
|
|
1532
|
+
async loadParentRun(parentRunID, contextUser, provider) {
|
|
1533
|
+
if (!parentRunID) {
|
|
1534
|
+
return null;
|
|
1535
|
+
}
|
|
1536
|
+
const run = await provider.GetEntityObject('MJ: AI Agent Runs', contextUser);
|
|
1537
|
+
return (await run.Load(parentRunID)) ? run : null;
|
|
1538
|
+
}
|
|
1539
|
+
/**
|
|
1540
|
+
* Routes a non-target tool call. For now this returns a structured "not available" result —
|
|
1541
|
+
* the richer client/UI/action routing is wired in a later phase. Documented minimal seam.
|
|
1542
|
+
*
|
|
1543
|
+
* @param call The non-target tool call.
|
|
1544
|
+
* @returns A failed {@link ToolExecutionResult} the model can narrate.
|
|
1545
|
+
*/
|
|
1546
|
+
async executeNonTargetTool(call) {
|
|
1547
|
+
return {
|
|
1548
|
+
CallID: call.CallID,
|
|
1549
|
+
Success: false,
|
|
1550
|
+
Output: `Tool '${call.ToolName}' is not available in this voice session.`
|
|
1551
|
+
};
|
|
1552
|
+
}
|
|
1553
|
+
/**
|
|
1554
|
+
* Parses the natural-language request text out of an `invoke-target-agent` call's arguments.
|
|
1555
|
+
* Falls back to the raw argument string when it is not the expected `{ request: string }` JSON.
|
|
1556
|
+
*
|
|
1557
|
+
* @param argumentsJson The raw arguments string emitted by the model.
|
|
1558
|
+
* @returns The request text to hand to the target agent.
|
|
1559
|
+
*/
|
|
1560
|
+
parseDelegateRequestText(argumentsJson) {
|
|
1561
|
+
try {
|
|
1562
|
+
const parsed = JSON.parse(argumentsJson);
|
|
1563
|
+
if (typeof parsed.request === 'string') {
|
|
1564
|
+
return parsed.request;
|
|
1565
|
+
}
|
|
1566
|
+
}
|
|
1567
|
+
catch {
|
|
1568
|
+
/* not JSON — fall through to raw */
|
|
1569
|
+
}
|
|
1570
|
+
return argumentsJson;
|
|
1571
|
+
}
|
|
1572
|
+
/**
|
|
1573
|
+
* Combines the broker's per-call abort signal with an optional caller-supplied signal so either
|
|
1574
|
+
* source can cancel the delegated run. Returns the broker signal alone when no caller signal is
|
|
1575
|
+
* present (the common case), avoiding an unnecessary controller.
|
|
1576
|
+
*
|
|
1577
|
+
* @param brokerSignal The broker-owned per-call abort signal (always present).
|
|
1578
|
+
* @param callerSignal An optional caller signal (e.g. a request-scoped barge-in).
|
|
1579
|
+
* @returns A single abort signal that fires when either source aborts.
|
|
1580
|
+
*/
|
|
1581
|
+
combineSignals(brokerSignal, callerSignal) {
|
|
1582
|
+
if (!callerSignal) {
|
|
1583
|
+
return brokerSignal;
|
|
1584
|
+
}
|
|
1585
|
+
const controller = new AbortController();
|
|
1586
|
+
const abort = () => controller.abort();
|
|
1587
|
+
if (brokerSignal.aborted || callerSignal.aborted) {
|
|
1588
|
+
controller.abort();
|
|
1589
|
+
}
|
|
1590
|
+
else {
|
|
1591
|
+
brokerSignal.addEventListener('abort', abort, { once: true });
|
|
1592
|
+
callerSignal.addEventListener('abort', abort, { once: true });
|
|
1593
|
+
}
|
|
1594
|
+
return controller.signal;
|
|
1595
|
+
}
|
|
1596
|
+
/**
|
|
1597
|
+
* The clear, actionable message returned when no usable Realtime model can be resolved.
|
|
1598
|
+
*
|
|
1599
|
+
* @returns The failure message.
|
|
1600
|
+
*/
|
|
1601
|
+
noModelMessage() {
|
|
1602
|
+
return ('No usable Realtime model could be resolved for the Realtime Co-Agent. Configure a model of ' +
|
|
1603
|
+
"AIModelType 'Realtime' with an active vendor DriverClass and a valid API key " +
|
|
1604
|
+
'(e.g. AI_VENDOR_API_KEY__<driver>).');
|
|
1605
|
+
}
|
|
1606
|
+
}
|
|
1607
|
+
//# sourceMappingURL=realtime-client-session-service.js.map
|