@voicelayer/sdk 0.6.2 → 0.6.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -14,11 +14,11 @@ import { LoggerProvider, BatchLogRecordProcessor } from '@opentelemetry/sdk-logs
14
14
  import { PeriodicExportingMetricReader } from '@opentelemetry/sdk-metrics';
15
15
  import { NodeSDK } from '@opentelemetry/sdk-node';
16
16
  import { ATTR_SERVICE_VERSION, ATTR_SERVICE_NAME } from '@opentelemetry/semantic-conventions';
17
+ import { llm, DEFAULT_API_CONNECT_OPTIONS, intervalForRetry, voice as voice$1, APIStatusError, APITimeoutError, APIConnectionError } from '@livekit/agents';
17
18
  import http from 'http';
18
19
  import https from 'https';
19
20
  import { Readable } from 'stream';
20
21
  import { lookup } from 'dns/promises';
21
- import { voice as voice$1, llm } from '@livekit/agents';
22
22
  import { getQuickJS, shouldInterruptAfterDeadline } from 'quickjs-emscripten';
23
23
  import { randomUUID, createHash } from 'crypto';
24
24
  import { setup, createActor } from 'xstate';
@@ -4596,32 +4596,116 @@ function acceptsReasoningEffort(model2) {
4596
4596
  const id = baseModelId(model2);
4597
4597
  return isReasoningModel(model2) && !/^o1-(mini|preview)/.test(id) && !/-chat(-|$)/.test(id);
4598
4598
  }
4599
+ function acceptsNoReasoningEffort(model2) {
4600
+ if (!acceptsReasoningEffort(model2))
4601
+ return false;
4602
+ const id = baseModelId(model2);
4603
+ if (/-pro\b/.test(id))
4604
+ return false;
4605
+ const gpt = /^gpt-(\d+)(?:\.(\d+))?/.exec(id);
4606
+ if (!gpt)
4607
+ return false;
4608
+ const major = Number(gpt[1]);
4609
+ const minor = gpt[2] !== void 0 ? Number(gpt[2]) : 0;
4610
+ return major > 5 || major === 5 && minor >= 1;
4611
+ }
4612
+ function forgetLearnedReasoningEfforts() {
4613
+ learnedEfforts.clear();
4614
+ }
4615
+ function chatReasoningEffort(model2, opts) {
4616
+ const learned = learnedEfforts.get(learnedKey(model2, opts.tools === true));
4617
+ if (learned !== void 0)
4618
+ return learned === "omit" ? void 0 : learned;
4619
+ if (!acceptsReasoningEffort(model2))
4620
+ return void 0;
4621
+ if (opts.tools === true && acceptsNoReasoningEffort(model2))
4622
+ return "none";
4623
+ return opts.requested;
4624
+ }
4599
4625
  function chatCompletionParams(model2, input) {
4626
+ const effort = chatReasoningEffort(model2, {
4627
+ ...input.tools !== void 0 ? { tools: input.tools } : {},
4628
+ ...input.reasoningEffort !== void 0 ? { requested: input.reasoningEffort } : {}
4629
+ });
4600
4630
  if (isReasoningModel(model2)) {
4601
4631
  return {
4602
4632
  ...input.maxTokens !== void 0 ? { max_completion_tokens: Math.max(input.maxTokens, REASONING_MIN_COMPLETION_TOKENS) } : {},
4603
- ...input.reasoningEffort !== void 0 && acceptsReasoningEffort(model2) ? { reasoning_effort: input.reasoningEffort } : {}
4633
+ ...effort !== void 0 ? { reasoning_effort: effort } : {}
4604
4634
  };
4605
4635
  }
4606
4636
  return {
4637
+ // only when a provider taught the process that this model (one we don't know as a reasoning model) takes an effort
4638
+ ...effort !== void 0 ? { reasoning_effort: effort } : {},
4607
4639
  ...input.maxTokens !== void 0 ? { max_tokens: input.maxTokens } : {},
4608
4640
  ...input.temperature !== void 0 ? { temperature: input.temperature } : {},
4609
4641
  ...input.topP !== void 0 ? { top_p: input.topP } : {}
4610
4642
  };
4611
4643
  }
4612
- function modelRejectionOf(err) {
4644
+ function acceptsSamplingParams(model2) {
4645
+ return !isReasoningModel(model2);
4646
+ }
4647
+ function providerErrorOf(err) {
4648
+ if (err === null || typeof err !== "object")
4649
+ return null;
4613
4650
  const e = err;
4614
- const status = typeof e?.status === "number" ? e.status : null;
4615
- if (status !== 400 && status !== 403 && status !== 404)
4651
+ const body = e["body"] !== null && typeof e["body"] === "object" ? e["body"] : {};
4652
+ const pick = (k) => e[k] !== void 0 && e[k] !== null ? e[k] : body[k];
4653
+ const statusRaw = typeof e["status"] === "number" ? e["status"] : e["statusCode"];
4654
+ const status = typeof statusRaw === "number" && statusRaw > 0 ? statusRaw : null;
4655
+ const code = pick("code");
4656
+ const type = pick("type");
4657
+ const param = pick("param");
4658
+ const message = typeof body["message"] === "string" ? body["message"] : typeof e["message"] === "string" ? e["message"] : "";
4659
+ return {
4660
+ status,
4661
+ code: typeof code === "string" ? code : typeof type === "string" ? type : "",
4662
+ param: typeof param === "string" ? param : null,
4663
+ message
4664
+ };
4665
+ }
4666
+ function reasoningEffortRejectionOf(err) {
4667
+ const f = providerErrorOf(err);
4668
+ if (!f || f.status !== 400)
4669
+ return null;
4670
+ if (f.param !== "reasoning_effort" && !/reasoning[_ ]effort/i.test(f.message))
4671
+ return null;
4672
+ return { status: f.status, message: f.message, suggestsNone: /reasoning[_ ]effort[^.]*\bnone\b/i.test(f.message) };
4673
+ }
4674
+ function learnReasoningEffort(model2, shape, err) {
4675
+ const rejection = reasoningEffortRejectionOf(err);
4676
+ if (!rejection)
4677
+ return false;
4678
+ const next = shape.sent !== "none" && (rejection.suggestsNone || shape.sent === void 0) ? "none" : "omit";
4679
+ if ((next === "omit" ? void 0 : next) === shape.sent)
4680
+ return false;
4681
+ learnedEfforts.set(learnedKey(model2, shape.tools), next);
4682
+ return true;
4683
+ }
4684
+ async function withReasoningEffortRetry(shape, call) {
4685
+ const sent = chatReasoningEffort(shape.model, {
4686
+ tools: shape.tools,
4687
+ ...shape.requested !== void 0 ? { requested: shape.requested } : {}
4688
+ });
4689
+ try {
4690
+ return await call();
4691
+ } catch (err) {
4692
+ if (!learnReasoningEffort(shape.model, { tools: shape.tools, sent }, err))
4693
+ throw err;
4694
+ return call();
4695
+ }
4696
+ }
4697
+ function modelRejectionOf(err) {
4698
+ const f = providerErrorOf(err);
4699
+ const status = f?.status ?? null;
4700
+ if (!f || status === null || status !== 400 && status !== 403 && status !== 404)
4616
4701
  return null;
4617
- const code = typeof e?.code === "string" ? e.code : typeof e?.type === "string" ? e.type : "";
4618
- const param = typeof e?.param === "string" ? e.param : null;
4619
- const rejected = REJECTION_CODES.has(code) || param !== null && MODEL_PARAMS.has(param);
4702
+ const { code, param } = f;
4703
+ const rejected = REJECTION_CODES.has(code) || param !== null && MODEL_PARAMS.has(param) || reasoningEffortRejectionOf(err) !== null;
4620
4704
  if (!rejected)
4621
4705
  return null;
4622
4706
  if (status === 403 && code !== "model_not_found")
4623
4707
  return null;
4624
- return { status, code: code || "invalid_request_error", message: typeof e?.message === "string" ? e.message : "" };
4708
+ return { status, code: code || "invalid_request_error", message: f.message };
4625
4709
  }
4626
4710
  async function withModelFallback(opts) {
4627
4711
  const fallbackModel = opts.fallbackModel ?? PLATFORM_DEFAULT_CHAT_MODEL;
@@ -4642,11 +4726,13 @@ async function withModelFallback(opts) {
4642
4726
  return opts.call(fallbackModel);
4643
4727
  }
4644
4728
  }
4645
- var PLATFORM_DEFAULT_CHAT_MODEL, REASONING_MIN_COMPLETION_TOKENS, REJECTION_CODES, MODEL_PARAMS, MODEL_REJECTION_TTL_MS, ModelRejectionCache;
4729
+ var PLATFORM_DEFAULT_CHAT_MODEL, REASONING_MIN_COMPLETION_TOKENS, learnedEfforts, learnedKey, REJECTION_CODES, MODEL_PARAMS, MODEL_REJECTION_TTL_MS, ModelRejectionCache;
4646
4730
  var init_chat_params = __esm({
4647
4731
  "../llm-client/dist/chat-params.js"() {
4648
4732
  PLATFORM_DEFAULT_CHAT_MODEL = "gpt-4o-mini";
4649
4733
  REASONING_MIN_COMPLETION_TOKENS = 2048;
4734
+ learnedEfforts = /* @__PURE__ */ new Map();
4735
+ learnedKey = (model2, tools) => `${baseModelId(model2)}|${tools ? "tools" : "plain"}`;
4650
4736
  REJECTION_CODES = /* @__PURE__ */ new Set(["unsupported_parameter", "unsupported_value", "model_not_found"]);
4651
4737
  MODEL_PARAMS = /* @__PURE__ */ new Set(["model", "max_tokens", "max_completion_tokens", "temperature", "top_p", "reasoning_effort"]);
4652
4738
  MODEL_REJECTION_TTL_MS = 5 * 60 * 1e3;
@@ -4674,6 +4760,30 @@ var init_chat_params = __esm({
4674
4760
  };
4675
4761
  }
4676
4762
  });
4763
+
4764
+ // ../llm-client/dist/index.js
4765
+ var dist_exports = {};
4766
+ __export(dist_exports, {
4767
+ MODEL_REJECTION_TTL_MS: () => MODEL_REJECTION_TTL_MS,
4768
+ ModelRejectionCache: () => ModelRejectionCache,
4769
+ PLATFORM_DEFAULT_CHAT_MODEL: () => PLATFORM_DEFAULT_CHAT_MODEL,
4770
+ REASONING_MIN_COMPLETION_TOKENS: () => REASONING_MIN_COMPLETION_TOKENS,
4771
+ acceptsNoReasoningEffort: () => acceptsNoReasoningEffort,
4772
+ acceptsReasoningEffort: () => acceptsReasoningEffort,
4773
+ acceptsSamplingParams: () => acceptsSamplingParams,
4774
+ chatCompletionParams: () => chatCompletionParams,
4775
+ chatReasoningEffort: () => chatReasoningEffort,
4776
+ createChatClient: () => createChatClient,
4777
+ forgetLearnedReasoningEfforts: () => forgetLearnedReasoningEfforts,
4778
+ hasChatKey: () => hasChatKey,
4779
+ isReasoningModel: () => isReasoningModel,
4780
+ learnReasoningEffort: () => learnReasoningEffort,
4781
+ modelRejectionOf: () => modelRejectionOf,
4782
+ providerErrorOf: () => providerErrorOf,
4783
+ reasoningEffortRejectionOf: () => reasoningEffortRejectionOf,
4784
+ withModelFallback: () => withModelFallback,
4785
+ withReasoningEffortRetry: () => withReasoningEffortRetry
4786
+ });
4677
4787
  function createChatClient(config = {}) {
4678
4788
  return new OpenAI({
4679
4789
  ...config.apiKey ? { apiKey: config.apiKey } : {},
@@ -4681,6 +4791,9 @@ function createChatClient(config = {}) {
4681
4791
  ...config.timeoutMs ? { timeout: config.timeoutMs } : {}
4682
4792
  });
4683
4793
  }
4794
+ function hasChatKey(config = {}) {
4795
+ return Boolean(config.apiKey || process.env["OPENAI_API_KEY"]);
4796
+ }
4684
4797
  var init_dist = __esm({
4685
4798
  "../llm-client/dist/index.js"() {
4686
4799
  init_chat_params();
@@ -5139,512 +5252,213 @@ var init_wrap = __esm({
5139
5252
  "src/providers/wrap.ts"() {
5140
5253
  }
5141
5254
  });
5142
-
5143
- // src/providers/index.ts
5144
- async function importOptional(spec, hint) {
5145
- try {
5146
- return await import(spec);
5147
- } catch {
5148
- throw new Error(`${hint}: "${spec}" is not installed. Add it with: pnpm add ${spec}`);
5255
+ function metricExportIntervalMs() {
5256
+ const raw = Number(process.env.OTEL_METRIC_EXPORT_INTERVAL);
5257
+ return Number.isFinite(raw) && raw > 0 ? raw : DEFAULT_METRIC_EXPORT_INTERVAL_MS;
5258
+ }
5259
+ function start(opts) {
5260
+ if (started)
5261
+ return;
5262
+ started = true;
5263
+ if (opts.debug) {
5264
+ diag.setLogger(new DiagConsoleLogger(), DiagLogLevel.INFO);
5265
+ }
5266
+ const authKey = process.env.VOICELAYER_TELEMETRY_KEY || process.env.VOICELAYER_API_KEY;
5267
+ if (!authKey) {
5268
+ return;
5269
+ }
5270
+ if (!process.env.OTEL_EXPORTER_OTLP_ENDPOINT) {
5271
+ process.env.OTEL_EXPORTER_OTLP_ENDPOINT = DEFAULT_OTLP_ENDPOINT;
5149
5272
  }
5273
+ process.env.OTEL_EXPORTER_OTLP_HEADERS = `Authorization=Bearer ${authKey}`;
5274
+ const resource = resourceFromAttributes({
5275
+ [ATTR_SERVICE_NAME]: opts.service,
5276
+ ...opts.version ? { [ATTR_SERVICE_VERSION]: opts.version } : {}
5277
+ });
5278
+ loggerProvider = new LoggerProvider({
5279
+ resource,
5280
+ processors: [new BatchLogRecordProcessor(new OTLPLogExporter())]
5281
+ });
5282
+ logs.setGlobalLoggerProvider(loggerProvider);
5283
+ metricReader = new PeriodicExportingMetricReader({
5284
+ exportIntervalMillis: metricExportIntervalMs(),
5285
+ // DELTA temporality: each export carries only the increment since the
5286
+ // previous one. The read side (apps/api openobserve-repository) totals a
5287
+ // metric with SUM(value) over the query window, which is correct only for
5288
+ // deltas. The OTLP exporter defaults to CUMULATIVE — it re-reports the
5289
+ // running total every export, so SUM double-counts every call that lives
5290
+ // longer than one export interval (inflating LLM/TTS/STT usage + cost).
5291
+ // Setting DELTA makes the existing SUM reads correct without touching them.
5292
+ exporter: new OTLPMetricExporter({
5293
+ temporalityPreference: AggregationTemporalityPreference.DELTA
5294
+ })
5295
+ });
5296
+ sdk = new NodeSDK({
5297
+ resource,
5298
+ traceExporter: new OTLPTraceExporter(),
5299
+ metricReader,
5300
+ instrumentations: [
5301
+ getNodeAutoInstrumentations({
5302
+ // fs instrumentation is extremely noisy and rarely useful for app traces.
5303
+ "@opentelemetry/instrumentation-fs": { enabled: false }
5304
+ })
5305
+ ]
5306
+ });
5307
+ sdk.start();
5308
+ const shutdown = async () => {
5309
+ try {
5310
+ await sdk?.shutdown();
5311
+ await loggerProvider?.shutdown();
5312
+ } catch {
5313
+ }
5314
+ };
5315
+ process.once("SIGTERM", () => void shutdown().then(() => process.exit(0)));
5316
+ process.once("SIGINT", () => void shutdown().then(() => process.exit(0)));
5150
5317
  }
5151
- function withCreds(opts, creds2) {
5152
- if (creds2?.apiKey) opts["apiKey"] = creds2.apiKey;
5153
- if (creds2?.baseURL) opts["baseURL"] = creds2.baseURL;
5154
- return opts;
5318
+ async function flushTelemetry() {
5319
+ const flushes = [];
5320
+ if (metricReader)
5321
+ flushes.push(metricReader.forceFlush());
5322
+ if (loggerProvider)
5323
+ flushes.push(loggerProvider.forceFlush());
5324
+ const proxied = trace.getTracerProvider();
5325
+ const tracerProvider = proxied.getDelegate?.() ?? proxied;
5326
+ if (typeof tracerProvider.forceFlush === "function") {
5327
+ flushes.push(tracerProvider.forceFlush());
5328
+ }
5329
+ await Promise.allSettled(flushes);
5155
5330
  }
5156
- var deepgram, openai, anthropic, cartesia, elevenlabs, assemblyai, google, silero, livekitTurn, connector;
5157
- var init_providers2 = __esm({
5158
- "src/providers/index.ts"() {
5159
- init_ssrf();
5160
- init_transport_callback();
5161
- init_connector_llm();
5162
- init_wrap();
5163
- deepgram = {
5164
- stt(options = {}) {
5165
- return async () => {
5166
- const dg = await importOptional("@voicelayer/agents-plugin-deepgram", "deepgram.stt");
5167
- return new dg.STT(
5168
- withCreds(
5169
- { model: options.model ?? "nova-3", language: options.language ?? "en-US" },
5170
- options
5171
- )
5172
- );
5173
- };
5174
- },
5175
- tts(options = {}) {
5176
- return async () => {
5177
- const dg = await importOptional("@voicelayer/agents-plugin-deepgram", "deepgram.tts");
5178
- return new dg.TTS(
5179
- withCreds({ model: options.model ?? "aura-2-harmonia-en" }, options)
5180
- );
5181
- };
5182
- }
5183
- };
5184
- openai = {
5185
- llm(options = {}) {
5186
- return async () => {
5187
- const oa = await importOptional("@voicelayer/agents-plugin-openai", "openai.llm");
5188
- return new oa.LLM(
5189
- withCreds({ model: options.model ?? "gpt-4o-mini" }, options)
5190
- );
5191
- };
5192
- },
5193
- tts(options = {}) {
5194
- return async () => {
5195
- const oa = await importOptional("@voicelayer/agents-plugin-openai", "openai.tts");
5196
- const opts = { model: options.model ?? "tts-1" };
5197
- if (options.voice) opts["voice"] = options.voice;
5198
- if (options.instructions) opts["instructions"] = options.instructions;
5199
- return new oa.TTS(withCreds(opts, options));
5200
- };
5201
- },
5202
- // Speech-to-speech via the OpenAI Realtime API. Returns a RealtimeProvider
5203
- // that the SDK uses in place of the stt/llm/tts pipeline.
5204
- realtime(options = {}) {
5205
- return async () => {
5206
- const oa = await importOptional("@voicelayer/agents-plugin-openai", "openai.realtime");
5207
- const opts = {};
5208
- if (options.model) opts["model"] = options.model;
5209
- if (options.voice) opts["voice"] = options.voice;
5210
- return new oa.realtime.RealtimeModel(withCreds(opts, options));
5211
- };
5212
- }
5213
- };
5214
- anthropic = {
5215
- llm(options = {}) {
5216
- return async () => {
5217
- const an = await importOptional("@livekit/agents-plugin-anthropic", "anthropic.llm");
5218
- return new an.LLM(
5219
- withCreds({ model: options.model ?? "claude-sonnet-5-5" }, options)
5220
- );
5221
- };
5222
- }
5223
- };
5224
- cartesia = {
5225
- tts(options = {}) {
5226
- return async () => {
5227
- const ct = await importOptional("@voicelayer/agents-plugin-cartesia", "cartesia.tts");
5228
- return new ct.TTS(options);
5229
- };
5230
- },
5231
- stt(options = {}) {
5232
- return async () => {
5233
- const ct = await importOptional("@voicelayer/agents-plugin-cartesia", "cartesia.stt");
5234
- const opts = {};
5235
- if (options.model) opts["model"] = options.model;
5236
- if (options.language) opts["language"] = options.language;
5237
- return new ct.STT(withCreds(opts, options));
5238
- };
5239
- }
5240
- };
5241
- elevenlabs = {
5242
- tts(options = {}) {
5243
- return async () => {
5244
- const el = await importOptional("@voicelayer/agents-plugin-elevenlabs", "elevenlabs.tts");
5245
- const opts = {};
5246
- if (options.voice) opts["voiceId"] = options.voice;
5247
- if (options.model) opts["model"] = options.model;
5248
- return new el.TTS(withCreds(opts, options));
5249
- };
5250
- }
5251
- };
5252
- assemblyai = {
5253
- stt(options = {}) {
5254
- return async () => {
5255
- const aai = await importOptional("@voicelayer/agents-plugin-assemblyai", "assemblyai.stt");
5256
- const opts = {};
5257
- if (options.language) opts["language"] = options.language;
5258
- return new aai.STT(withCreds(opts, options));
5259
- };
5260
- }
5261
- };
5262
- google = {
5263
- llm(options = {}) {
5264
- return async () => {
5265
- const g = await importOptional("@voicelayer/agents-plugin-google", "google.llm");
5266
- return new g.LLM(
5267
- withCreds({ model: options.model ?? "gemini-3.8-flash" }, options)
5268
- );
5269
- };
5270
- },
5271
- realtime(options = {}) {
5272
- return async () => {
5273
- const g = await importOptional("@voicelayer/agents-plugin-google", "google.realtime");
5274
- const opts = {};
5275
- if (options.model) opts["model"] = options.model;
5276
- if (options.voice) opts["voice"] = options.voice;
5277
- return new g.beta.realtime.RealtimeModel(withCreds(opts, options));
5278
- };
5279
- }
5280
- };
5281
- silero = {
5282
- vad() {
5283
- return async () => {
5284
- const sil = await importOptional("@voicelayer/agents-plugin-silero", "silero.vad");
5285
- return await sil.VAD.load();
5286
- };
5287
- }
5288
- };
5289
- livekitTurn = {
5290
- english() {
5291
- return async () => {
5292
- const lk = await importOptional("@voicelayer/agents-plugin-livekit", "livekitTurn.english");
5293
- return new lk.turnDetector.EnglishModel();
5294
- };
5295
- },
5296
- multilingual() {
5297
- return async () => {
5298
- const lk = await importOptional("@voicelayer/agents-plugin-livekit", "livekitTurn.multilingual");
5299
- return new lk.turnDetector.MultilingualModel();
5300
- };
5301
- }
5302
- };
5303
- connector = {
5304
- llm(config = {}) {
5305
- return async (call) => {
5306
- if (config.url) {
5307
- await assertPublicHttpsUrl(
5308
- config.url,
5309
- config.allowHosts ? { allowHosts: config.allowHosts } : {}
5310
- );
5311
- return openai.llm({
5312
- ...config.model ? { model: config.model } : {},
5313
- ...config.apiKey ? { apiKey: config.apiKey } : {},
5314
- baseURL: config.url
5315
- })(call);
5316
- }
5317
- const transport = config.transport ?? (config.onQuery ? callbackTransport(config.onQuery) : void 0);
5318
- if (!transport) {
5319
- throw new Error("connector.llm requires one of: url, onQuery, or transport");
5320
- }
5321
- const projectId = typeof call.metadata["projectId"] === "string" ? call.metadata["projectId"] : void 0;
5322
- const llmOptions = {
5323
- ...config.model ? { model: config.model } : {},
5324
- ...config.temperature !== void 0 ? { temperature: config.temperature } : {},
5325
- ...config.fallbackText ? { fallbackText: config.fallbackText } : {},
5326
- ...call.callId ? { callId: call.callId } : {},
5327
- ...projectId ? { projectId } : {}
5328
- };
5329
- return await createConnectorLLM(transport, llmOptions);
5330
- };
5331
- }
5331
+ function getLogger(name) {
5332
+ return logs.getLogger(name);
5333
+ }
5334
+ var DEFAULT_OTLP_ENDPOINT, DEFAULT_METRIC_EXPORT_INTERVAL_MS, started, sdk, loggerProvider, metricReader;
5335
+ var init_start = __esm({
5336
+ "../observability/dist/start.js"() {
5337
+ DEFAULT_OTLP_ENDPOINT = "https://otel.vlayers.ai/v1/otlp";
5338
+ DEFAULT_METRIC_EXPORT_INTERVAL_MS = 15e3;
5339
+ started = false;
5340
+ sdk = null;
5341
+ loggerProvider = null;
5342
+ metricReader = null;
5343
+ }
5344
+ });
5345
+
5346
+ // ../observability/dist/attributes.js
5347
+ function callContextToAttributes(ctx) {
5348
+ const out = {};
5349
+ if (ctx.projectId)
5350
+ out[ATTR.projectId] = ctx.projectId;
5351
+ if (ctx.callId)
5352
+ out[ATTR.callId] = ctx.callId;
5353
+ if (ctx.campaignId)
5354
+ out[ATTR.campaignId] = ctx.campaignId;
5355
+ if (ctx.room)
5356
+ out[ATTR.room] = ctx.room;
5357
+ if (ctx.agentId)
5358
+ out[ATTR.agentId] = ctx.agentId;
5359
+ if (ctx.phoneNumberId)
5360
+ out[ATTR.phoneNumberId] = ctx.phoneNumberId;
5361
+ if (ctx.bindingId)
5362
+ out[ATTR.bindingId] = ctx.bindingId;
5363
+ return out;
5364
+ }
5365
+ var ATTR;
5366
+ var init_attributes = __esm({
5367
+ "../observability/dist/attributes.js"() {
5368
+ ATTR = {
5369
+ projectId: "vl.project_id",
5370
+ callId: "vl.call_id",
5371
+ campaignId: "vl.campaign_id",
5372
+ room: "vl.room",
5373
+ agentId: "vl.agent_id",
5374
+ phoneNumberId: "vl.phone_number_id",
5375
+ bindingId: "vl.binding_id",
5376
+ source: "vl.source",
5377
+ kind: "vl.kind"
5332
5378
  };
5333
5379
  }
5334
5380
  });
5335
-
5336
- // src/providers/registry.ts
5337
- function entry(id, bits) {
5338
- const identity = providerIdentity(id);
5339
- if (!identity) throw new Error(`PROVIDER_REGISTRY: no provider identity in contracts for '${id}'`);
5340
- return { name: identity.id, capabilities: identity.capabilities, ...bits };
5381
+ async function withCallContext(opts, fn) {
5382
+ const tracer = trace.getTracer(TRACER_NAME);
5383
+ const ctx = {
5384
+ ...opts.projectId ? { projectId: opts.projectId } : {},
5385
+ ...opts.callId ? { callId: opts.callId } : {},
5386
+ ...opts.room ? { room: opts.room } : {},
5387
+ ...opts.agentId ? { agentId: opts.agentId } : {},
5388
+ ...opts.phoneNumberId ? { phoneNumberId: opts.phoneNumberId } : {},
5389
+ ...opts.bindingId ? { bindingId: opts.bindingId } : {}
5390
+ };
5391
+ return callContextStore.run(ctx, () => tracer.startActiveSpan(opts.kind, async (span) => {
5392
+ span.setAttributes(callContextToAttributes(opts));
5393
+ span.setAttribute(ATTR.kind, opts.kind);
5394
+ if (opts.source)
5395
+ span.setAttribute(ATTR.source, opts.source);
5396
+ if (opts.attributes)
5397
+ span.setAttributes(opts.attributes);
5398
+ try {
5399
+ return await fn(span);
5400
+ } catch (err) {
5401
+ span.recordException(err);
5402
+ span.setStatus({ code: SpanStatusCode.ERROR, message: err.message });
5403
+ throw err;
5404
+ } finally {
5405
+ span.end();
5406
+ }
5407
+ }));
5341
5408
  }
5342
- function providerEntry(name) {
5343
- if (!name) return void 0;
5344
- return PROVIDER_REGISTRY[PROVIDER_ALIASES[name] ?? name];
5409
+ function getCurrentCallContext() {
5410
+ return callContextStore.getStore();
5345
5411
  }
5346
- function resolveStt(name, o) {
5347
- return (providerEntry(name)?.stt ?? PROVIDER_REGISTRY["deepgram"].stt)(o);
5412
+ var TRACER_NAME, callContextStore;
5413
+ var init_call_context = __esm({
5414
+ "../observability/dist/call-context.js"() {
5415
+ init_attributes();
5416
+ TRACER_NAME = "@voicelayer/observability";
5417
+ callContextStore = new AsyncLocalStorage();
5418
+ }
5419
+ });
5420
+ var init_trace_propagation = __esm({
5421
+ "../observability/dist/trace-propagation.js"() {
5422
+ }
5423
+ });
5424
+ function meter() {
5425
+ return metrics.getMeter(METER_NAME, METER_VERSION);
5348
5426
  }
5349
- function resolveTts(name, o) {
5350
- return (providerEntry(name)?.tts ?? PROVIDER_REGISTRY["deepgram"].tts)(o);
5427
+ function attrs(a) {
5428
+ const out = {
5429
+ [ATTR.projectId]: a.projectId,
5430
+ [ATTR.callId]: a.callId,
5431
+ "vl.provider": a.provider
5432
+ };
5433
+ if (a.campaignId)
5434
+ out[ATTR.campaignId] = a.campaignId;
5435
+ if (a.model)
5436
+ out["vl.model"] = a.model;
5437
+ return out;
5351
5438
  }
5352
- function resolveLlm(name, o) {
5353
- return (providerEntry(name)?.llm ?? PROVIDER_REGISTRY["openai"].llm)(withCurrentModel(name, o));
5439
+ function llmInputTokens() {
5440
+ return _llmInputTokens ??= meter().createCounter(USAGE_METRIC_NAMES.llmInputTokens, {
5441
+ description: "Input tokens consumed by an LLM call",
5442
+ unit: "{tokens}"
5443
+ });
5354
5444
  }
5355
- function resolveRealtime(name, o) {
5356
- return (providerEntry(name)?.realtime ?? PROVIDER_REGISTRY["openai"].realtime)(withCurrentModel(name, o));
5445
+ function llmOutputTokens() {
5446
+ return _llmOutputTokens ??= meter().createCounter(USAGE_METRIC_NAMES.llmOutputTokens, {
5447
+ description: "Output tokens produced by an LLM call",
5448
+ unit: "{tokens}"
5449
+ });
5357
5450
  }
5358
- function withCurrentModel(name, o) {
5359
- if (!name || !o.model) return o;
5360
- const current = currentModelFor(name, o.model);
5361
- if (!current.retired) return o;
5362
- console.warn("[agent] the configured model is retired; running its replacement", {
5363
- provider: name,
5364
- model: current.retired,
5365
- replacement: current.model
5451
+ function ttsChars() {
5452
+ return _ttsChars ??= meter().createCounter(USAGE_METRIC_NAMES.ttsChars, {
5453
+ description: "Characters synthesized by TTS",
5454
+ unit: "{chars}"
5366
5455
  });
5367
- return { ...o, model: current.model };
5368
5456
  }
5369
- var model, lang, voice, creds, PROVIDER_REGISTRY;
5370
- var init_registry = __esm({
5371
- "src/providers/registry.ts"() {
5372
- init_providers2();
5373
- init_src();
5374
- model = (o) => o.model ? { model: o.model } : {};
5375
- lang = (o) => o.language ? { language: o.language } : {};
5376
- voice = (o) => o.voice ? { voice: o.voice } : {};
5377
- creds = (o) => o.creds ?? {};
5378
- PROVIDER_REGISTRY = {
5379
- deepgram: entry("deepgram", {
5380
- failureClass: "vendor-api",
5381
- engines: ["livekit"],
5382
- stt: (o) => deepgram.stt({ ...model(o), ...lang(o), ...creds(o) }),
5383
- // Deepgram folds the voice into the model id (aura-2-<name>-en): wire config
5384
- // passes a voice, code config passes a model — both land as `model`.
5385
- tts: (o) => {
5386
- const m = o.voice ?? o.model;
5387
- return deepgram.tts({ ...m ? { model: m } : {}, ...creds(o) });
5388
- }
5389
- }),
5390
- openai: entry("openai", {
5391
- failureClass: "vendor-api",
5392
- engines: ["livekit", "text"],
5393
- llm: (o) => openai.llm({ ...model(o), ...creds(o) }),
5394
- tts: (o) => openai.tts({ ...voice(o), ...model(o), ...creds(o) }),
5395
- realtime: (o) => openai.realtime({ ...model(o), ...voice(o), ...creds(o) })
5396
- }),
5397
- anthropic: entry("anthropic", {
5398
- failureClass: "vendor-api",
5399
- engines: ["livekit", "text"],
5400
- llm: (o) => anthropic.llm({ ...model(o), ...creds(o) })
5401
- }),
5402
- google: entry("google", {
5403
- failureClass: "vendor-api",
5404
- engines: ["livekit", "text"],
5405
- llm: (o) => google.llm({ ...model(o), ...creds(o) }),
5406
- realtime: (o) => google.realtime({ ...model(o), ...voice(o), ...creds(o) })
5407
- }),
5408
- cartesia: entry("cartesia", {
5409
- failureClass: "vendor-api",
5410
- engines: ["livekit"],
5411
- stt: (o) => cartesia.stt({ ...model(o), ...lang(o), ...creds(o) }),
5412
- tts: (o) => cartesia.tts({ ...voice(o), ...model(o), ...creds(o) })
5413
- }),
5414
- elevenlabs: entry("elevenlabs", {
5415
- failureClass: "vendor-api",
5416
- engines: ["livekit"],
5417
- tts: (o) => elevenlabs.tts({ ...voice(o), ...model(o), ...creds(o) })
5418
- }),
5419
- assemblyai: entry("assemblyai", {
5420
- failureClass: "vendor-api",
5421
- engines: ["livekit"],
5422
- stt: (o) => assemblyai.stt({ ...lang(o), ...creds(o) })
5423
- // no model knob
5424
- }),
5425
- silero: entry("silero", {
5426
- failureClass: "local",
5427
- engines: ["livekit"],
5428
- vad: () => silero.vad()
5429
- })
5430
- };
5431
- }
5432
- });
5433
-
5434
- // src/providers/llm.ts
5435
- var init_llm = __esm({
5436
- "src/providers/llm.ts"() {
5437
- init_openai_default();
5438
- init_registry();
5439
- }
5440
- });
5441
- function metricExportIntervalMs() {
5442
- const raw = Number(process.env.OTEL_METRIC_EXPORT_INTERVAL);
5443
- return Number.isFinite(raw) && raw > 0 ? raw : DEFAULT_METRIC_EXPORT_INTERVAL_MS;
5444
- }
5445
- function start(opts) {
5446
- if (started)
5447
- return;
5448
- started = true;
5449
- if (opts.debug) {
5450
- diag.setLogger(new DiagConsoleLogger(), DiagLogLevel.INFO);
5451
- }
5452
- const authKey = process.env.VOICELAYER_TELEMETRY_KEY || process.env.VOICELAYER_API_KEY;
5453
- if (!authKey) {
5454
- return;
5455
- }
5456
- if (!process.env.OTEL_EXPORTER_OTLP_ENDPOINT) {
5457
- process.env.OTEL_EXPORTER_OTLP_ENDPOINT = DEFAULT_OTLP_ENDPOINT;
5458
- }
5459
- process.env.OTEL_EXPORTER_OTLP_HEADERS = `Authorization=Bearer ${authKey}`;
5460
- const resource = resourceFromAttributes({
5461
- [ATTR_SERVICE_NAME]: opts.service,
5462
- ...opts.version ? { [ATTR_SERVICE_VERSION]: opts.version } : {}
5463
- });
5464
- loggerProvider = new LoggerProvider({
5465
- resource,
5466
- processors: [new BatchLogRecordProcessor(new OTLPLogExporter())]
5467
- });
5468
- logs.setGlobalLoggerProvider(loggerProvider);
5469
- metricReader = new PeriodicExportingMetricReader({
5470
- exportIntervalMillis: metricExportIntervalMs(),
5471
- // DELTA temporality: each export carries only the increment since the
5472
- // previous one. The read side (apps/api openobserve-repository) totals a
5473
- // metric with SUM(value) over the query window, which is correct only for
5474
- // deltas. The OTLP exporter defaults to CUMULATIVE — it re-reports the
5475
- // running total every export, so SUM double-counts every call that lives
5476
- // longer than one export interval (inflating LLM/TTS/STT usage + cost).
5477
- // Setting DELTA makes the existing SUM reads correct without touching them.
5478
- exporter: new OTLPMetricExporter({
5479
- temporalityPreference: AggregationTemporalityPreference.DELTA
5480
- })
5481
- });
5482
- sdk = new NodeSDK({
5483
- resource,
5484
- traceExporter: new OTLPTraceExporter(),
5485
- metricReader,
5486
- instrumentations: [
5487
- getNodeAutoInstrumentations({
5488
- // fs instrumentation is extremely noisy and rarely useful for app traces.
5489
- "@opentelemetry/instrumentation-fs": { enabled: false }
5490
- })
5491
- ]
5492
- });
5493
- sdk.start();
5494
- const shutdown = async () => {
5495
- try {
5496
- await sdk?.shutdown();
5497
- await loggerProvider?.shutdown();
5498
- } catch {
5499
- }
5500
- };
5501
- process.once("SIGTERM", () => void shutdown().then(() => process.exit(0)));
5502
- process.once("SIGINT", () => void shutdown().then(() => process.exit(0)));
5503
- }
5504
- async function flushTelemetry() {
5505
- const flushes = [];
5506
- if (metricReader)
5507
- flushes.push(metricReader.forceFlush());
5508
- if (loggerProvider)
5509
- flushes.push(loggerProvider.forceFlush());
5510
- const proxied = trace.getTracerProvider();
5511
- const tracerProvider = proxied.getDelegate?.() ?? proxied;
5512
- if (typeof tracerProvider.forceFlush === "function") {
5513
- flushes.push(tracerProvider.forceFlush());
5514
- }
5515
- await Promise.allSettled(flushes);
5516
- }
5517
- function getLogger(name) {
5518
- return logs.getLogger(name);
5519
- }
5520
- var DEFAULT_OTLP_ENDPOINT, DEFAULT_METRIC_EXPORT_INTERVAL_MS, started, sdk, loggerProvider, metricReader;
5521
- var init_start = __esm({
5522
- "../observability/dist/start.js"() {
5523
- DEFAULT_OTLP_ENDPOINT = "https://otel.vlayers.ai/v1/otlp";
5524
- DEFAULT_METRIC_EXPORT_INTERVAL_MS = 15e3;
5525
- started = false;
5526
- sdk = null;
5527
- loggerProvider = null;
5528
- metricReader = null;
5529
- }
5530
- });
5531
-
5532
- // ../observability/dist/attributes.js
5533
- function callContextToAttributes(ctx) {
5534
- const out = {};
5535
- if (ctx.projectId)
5536
- out[ATTR.projectId] = ctx.projectId;
5537
- if (ctx.callId)
5538
- out[ATTR.callId] = ctx.callId;
5539
- if (ctx.campaignId)
5540
- out[ATTR.campaignId] = ctx.campaignId;
5541
- if (ctx.room)
5542
- out[ATTR.room] = ctx.room;
5543
- if (ctx.agentId)
5544
- out[ATTR.agentId] = ctx.agentId;
5545
- if (ctx.phoneNumberId)
5546
- out[ATTR.phoneNumberId] = ctx.phoneNumberId;
5547
- if (ctx.bindingId)
5548
- out[ATTR.bindingId] = ctx.bindingId;
5549
- return out;
5550
- }
5551
- var ATTR;
5552
- var init_attributes = __esm({
5553
- "../observability/dist/attributes.js"() {
5554
- ATTR = {
5555
- projectId: "vl.project_id",
5556
- callId: "vl.call_id",
5557
- campaignId: "vl.campaign_id",
5558
- room: "vl.room",
5559
- agentId: "vl.agent_id",
5560
- phoneNumberId: "vl.phone_number_id",
5561
- bindingId: "vl.binding_id",
5562
- source: "vl.source",
5563
- kind: "vl.kind"
5564
- };
5565
- }
5566
- });
5567
- async function withCallContext(opts, fn) {
5568
- const tracer = trace.getTracer(TRACER_NAME);
5569
- const ctx = {
5570
- ...opts.projectId ? { projectId: opts.projectId } : {},
5571
- ...opts.callId ? { callId: opts.callId } : {},
5572
- ...opts.room ? { room: opts.room } : {},
5573
- ...opts.agentId ? { agentId: opts.agentId } : {},
5574
- ...opts.phoneNumberId ? { phoneNumberId: opts.phoneNumberId } : {},
5575
- ...opts.bindingId ? { bindingId: opts.bindingId } : {}
5576
- };
5577
- return callContextStore.run(ctx, () => tracer.startActiveSpan(opts.kind, async (span) => {
5578
- span.setAttributes(callContextToAttributes(opts));
5579
- span.setAttribute(ATTR.kind, opts.kind);
5580
- if (opts.source)
5581
- span.setAttribute(ATTR.source, opts.source);
5582
- if (opts.attributes)
5583
- span.setAttributes(opts.attributes);
5584
- try {
5585
- return await fn(span);
5586
- } catch (err) {
5587
- span.recordException(err);
5588
- span.setStatus({ code: SpanStatusCode.ERROR, message: err.message });
5589
- throw err;
5590
- } finally {
5591
- span.end();
5592
- }
5593
- }));
5594
- }
5595
- function getCurrentCallContext() {
5596
- return callContextStore.getStore();
5597
- }
5598
- var TRACER_NAME, callContextStore;
5599
- var init_call_context = __esm({
5600
- "../observability/dist/call-context.js"() {
5601
- init_attributes();
5602
- TRACER_NAME = "@voicelayer/observability";
5603
- callContextStore = new AsyncLocalStorage();
5604
- }
5605
- });
5606
- var init_trace_propagation = __esm({
5607
- "../observability/dist/trace-propagation.js"() {
5608
- }
5609
- });
5610
- function meter() {
5611
- return metrics.getMeter(METER_NAME, METER_VERSION);
5612
- }
5613
- function attrs(a) {
5614
- const out = {
5615
- [ATTR.projectId]: a.projectId,
5616
- [ATTR.callId]: a.callId,
5617
- "vl.provider": a.provider
5618
- };
5619
- if (a.campaignId)
5620
- out[ATTR.campaignId] = a.campaignId;
5621
- if (a.model)
5622
- out["vl.model"] = a.model;
5623
- return out;
5624
- }
5625
- function llmInputTokens() {
5626
- return _llmInputTokens ??= meter().createCounter(USAGE_METRIC_NAMES.llmInputTokens, {
5627
- description: "Input tokens consumed by an LLM call",
5628
- unit: "{tokens}"
5629
- });
5630
- }
5631
- function llmOutputTokens() {
5632
- return _llmOutputTokens ??= meter().createCounter(USAGE_METRIC_NAMES.llmOutputTokens, {
5633
- description: "Output tokens produced by an LLM call",
5634
- unit: "{tokens}"
5635
- });
5636
- }
5637
- function ttsChars() {
5638
- return _ttsChars ??= meter().createCounter(USAGE_METRIC_NAMES.ttsChars, {
5639
- description: "Characters synthesized by TTS",
5640
- unit: "{chars}"
5641
- });
5642
- }
5643
- function sttSeconds() {
5644
- return _sttSeconds ??= meter().createCounter(USAGE_METRIC_NAMES.sttSeconds, {
5645
- description: "Audio seconds transcribed by STT",
5646
- unit: "s"
5647
- });
5457
+ function sttSeconds() {
5458
+ return _sttSeconds ??= meter().createCounter(USAGE_METRIC_NAMES.sttSeconds, {
5459
+ description: "Audio seconds transcribed by STT",
5460
+ unit: "s"
5461
+ });
5648
5462
  }
5649
5463
  function firstTokenMs() {
5650
5464
  return _firstTokenMs ??= meter().createHistogram("vl.turn.first_token_ms", {
@@ -5719,157 +5533,710 @@ function recordEouLatency(args) {
5719
5533
  eouOnTurnCompletedDelayMs().record(args.onUserTurnCompletedDelayMs, base);
5720
5534
  }
5721
5535
  }
5722
- function interruptionTotalMs() {
5723
- return _interruptionTotalMs ??= meter().createHistogram("vl.interruption.total_ms", {
5724
- description: "RTT for interruption detection inference",
5725
- unit: "ms"
5726
- });
5727
- }
5728
- function interruptionPredictionMs() {
5729
- return _interruptionPredictionMs ??= meter().createHistogram("vl.interruption.prediction_ms", {
5730
- description: "Model-side time for interruption prediction",
5731
- unit: "ms"
5732
- });
5733
- }
5734
- function interruptionDetectionDelayMs() {
5735
- return _interruptionDetectionDelayMs ??= meter().createHistogram("vl.interruption.detection_delay_ms", {
5736
- description: "ms from speech onset to final interruption prediction",
5737
- unit: "ms"
5738
- });
5739
- }
5740
- function interruptionCountCounter() {
5741
- return _interruptionCount ??= meter().createCounter("vl.interruption.count", {
5742
- description: "Interruptions detected",
5743
- unit: "{interruptions}"
5744
- });
5536
+ function interruptionTotalMs() {
5537
+ return _interruptionTotalMs ??= meter().createHistogram("vl.interruption.total_ms", {
5538
+ description: "RTT for interruption detection inference",
5539
+ unit: "ms"
5540
+ });
5541
+ }
5542
+ function interruptionPredictionMs() {
5543
+ return _interruptionPredictionMs ??= meter().createHistogram("vl.interruption.prediction_ms", {
5544
+ description: "Model-side time for interruption prediction",
5545
+ unit: "ms"
5546
+ });
5547
+ }
5548
+ function interruptionDetectionDelayMs() {
5549
+ return _interruptionDetectionDelayMs ??= meter().createHistogram("vl.interruption.detection_delay_ms", {
5550
+ description: "ms from speech onset to final interruption prediction",
5551
+ unit: "ms"
5552
+ });
5553
+ }
5554
+ function interruptionCountCounter() {
5555
+ return _interruptionCount ??= meter().createCounter("vl.interruption.count", {
5556
+ description: "Interruptions detected",
5557
+ unit: "{interruptions}"
5558
+ });
5559
+ }
5560
+ function backchannelCountCounter() {
5561
+ return _backchannelCount ??= meter().createCounter("vl.interruption.backchannel_count", {
5562
+ description: "Backchannel utterances detected",
5563
+ unit: "{backchannels}"
5564
+ });
5565
+ }
5566
+ function recordInterruptionMetrics(args) {
5567
+ const base = {
5568
+ [ATTR.projectId]: args.projectId,
5569
+ [ATTR.callId]: args.callId
5570
+ };
5571
+ if (args.campaignId)
5572
+ base[ATTR.campaignId] = args.campaignId;
5573
+ if (args.totalDurationMs !== void 0 && args.totalDurationMs > 0) {
5574
+ interruptionTotalMs().record(args.totalDurationMs, base);
5575
+ }
5576
+ if (args.predictionDurationMs !== void 0 && args.predictionDurationMs > 0) {
5577
+ interruptionPredictionMs().record(args.predictionDurationMs, base);
5578
+ }
5579
+ if (args.detectionDelayMs !== void 0 && args.detectionDelayMs > 0) {
5580
+ interruptionDetectionDelayMs().record(args.detectionDelayMs, base);
5581
+ }
5582
+ if (args.numInterruptions !== void 0 && args.numInterruptions > 0) {
5583
+ interruptionCountCounter().add(args.numInterruptions, base);
5584
+ }
5585
+ if (args.numBackchannels !== void 0 && args.numBackchannels > 0) {
5586
+ backchannelCountCounter().add(args.numBackchannels, base);
5587
+ }
5588
+ }
5589
+ function realtimeSessionDurationMs() {
5590
+ return _realtimeSessionDurationMs ??= meter().createCounter("vl.realtime.session_duration_ms", {
5591
+ description: "Realtime model session duration for session-billed providers",
5592
+ unit: "ms"
5593
+ });
5594
+ }
5595
+ function recordRealtimeSessionDuration(args) {
5596
+ if (args.sessionDurationMs > 0) {
5597
+ realtimeSessionDurationMs().add(args.sessionDurationMs, attrs(args));
5598
+ }
5599
+ }
5600
+ function modelFallbacks() {
5601
+ return _modelFallbacks ??= meter().createCounter("vl.llm.model_fallback", {
5602
+ description: "Model calls the provider rejected that fell back to the platform default model",
5603
+ unit: "{fallbacks}"
5604
+ });
5605
+ }
5606
+ function recordModelFallback(args) {
5607
+ const out = {
5608
+ "vl.model": args.model,
5609
+ "vl.fallback_model": args.fallbackModel,
5610
+ "vl.error_code": args.code
5611
+ };
5612
+ if (args.projectId)
5613
+ out[ATTR.projectId] = args.projectId;
5614
+ if (args.agentId)
5615
+ out[ATTR.agentId] = args.agentId;
5616
+ if (args.surface)
5617
+ out["vl.surface"] = args.surface;
5618
+ modelFallbacks().add(1, out);
5619
+ }
5620
+ var METER_NAME, METER_VERSION, USAGE_METRIC_NAMES, _llmInputTokens, _llmOutputTokens, _ttsChars, _sttSeconds, _firstTokenMs, _ttsStartMs, _eouDelayMs, _eouTranscriptionDelayMs, _eouOnTurnCompletedDelayMs, _interruptionTotalMs, _interruptionPredictionMs, _interruptionDetectionDelayMs, _interruptionCount, _backchannelCount, _realtimeSessionDurationMs, _modelFallbacks;
5621
+ var init_metrics = __esm({
5622
+ "../observability/dist/metrics.js"() {
5623
+ init_attributes();
5624
+ METER_NAME = "voicelayer";
5625
+ METER_VERSION = "0.1.0";
5626
+ USAGE_METRIC_NAMES = {
5627
+ llmInputTokens: "vl.llm.input_tokens",
5628
+ llmOutputTokens: "vl.llm.output_tokens",
5629
+ ttsChars: "vl.tts.chars",
5630
+ sttSeconds: "vl.stt.seconds"
5631
+ };
5632
+ _llmInputTokens = null;
5633
+ _llmOutputTokens = null;
5634
+ _ttsChars = null;
5635
+ _sttSeconds = null;
5636
+ _firstTokenMs = null;
5637
+ _ttsStartMs = null;
5638
+ _eouDelayMs = null;
5639
+ _eouTranscriptionDelayMs = null;
5640
+ _eouOnTurnCompletedDelayMs = null;
5641
+ _interruptionTotalMs = null;
5642
+ _interruptionPredictionMs = null;
5643
+ _interruptionDetectionDelayMs = null;
5644
+ _interruptionCount = null;
5645
+ _backchannelCount = null;
5646
+ _realtimeSessionDurationMs = null;
5647
+ _modelFallbacks = null;
5648
+ }
5649
+ });
5650
+ function recordTurnLatencySpan(args) {
5651
+ if (!Number.isFinite(args.latencyMs) || args.latencyMs < 0)
5652
+ return;
5653
+ if (args.latencyMs === 0 && !args.allowZero)
5654
+ return;
5655
+ const attributes = {
5656
+ [ATTR.projectId]: args.projectId,
5657
+ [ATTR.callId]: args.callId,
5658
+ [LATENCY_STAGE_ATTR]: args.stage,
5659
+ [LATENCY_MS_ATTR]: args.latencyMs,
5660
+ [LATENCY_REALTIME_ATTR]: args.realtime ? 1 : 0
5661
+ };
5662
+ if (args.campaignId)
5663
+ attributes[ATTR.campaignId] = args.campaignId;
5664
+ trace.getTracer(TRACER_NAME2).startSpan(LATENCY_SPAN_NAME, { attributes }).end();
5665
+ }
5666
+ var TRACER_NAME2, LATENCY_SPAN_NAME, LATENCY_STAGE_ATTR, LATENCY_MS_ATTR, LATENCY_REALTIME_ATTR;
5667
+ var init_latency_span = __esm({
5668
+ "../observability/dist/latency-span.js"() {
5669
+ init_attributes();
5670
+ TRACER_NAME2 = "@voicelayer/observability";
5671
+ LATENCY_SPAN_NAME = "vl.turn_latency";
5672
+ LATENCY_STAGE_ATTR = "vl.stage";
5673
+ LATENCY_MS_ATTR = "vl.latency_ms";
5674
+ LATENCY_REALTIME_ATTR = "vl.realtime";
5675
+ }
5676
+ });
5677
+
5678
+ // ../observability/dist/index.js
5679
+ var init_dist2 = __esm({
5680
+ "../observability/dist/index.js"() {
5681
+ init_start();
5682
+ init_call_context();
5683
+ init_trace_propagation();
5684
+ init_attributes();
5685
+ init_metrics();
5686
+ init_latency_span();
5687
+ }
5688
+ });
5689
+
5690
+ // src/providers/resilient-llm.ts
5691
+ var resilient_llm_exports = {};
5692
+ __export(resilient_llm_exports, {
5693
+ LLM_APOLOGY: () => LLM_APOLOGY,
5694
+ ResilientLLM: () => ResilientLLM,
5695
+ isResilientLLM: () => isResilientLLM
5696
+ });
5697
+ function isResilientLLM(value) {
5698
+ return typeof value === "object" && value !== null && value[RESILIENT] === true;
5699
+ }
5700
+ function transient(error) {
5701
+ if (error instanceof APIStatusError) {
5702
+ const s = error.statusCode;
5703
+ return s === 408 || s === 429 || s < 0 || s >= 500;
5704
+ }
5705
+ return error instanceof APITimeoutError || error instanceof APIConnectionError;
5706
+ }
5707
+ var LLM_APOLOGY, RESILIENT, ResilientLLM, sameModel, ServedLLM, ResilientLLMStream;
5708
+ var init_resilient_llm = __esm({
5709
+ "src/providers/resilient-llm.ts"() {
5710
+ init_dist();
5711
+ init_dist2();
5712
+ LLM_APOLOGY = "Sorry, I'm having trouble right now. Could you give me a moment and say that again?";
5713
+ RESILIENT = /* @__PURE__ */ Symbol.for("voicelayer.resilientLLM");
5714
+ ResilientLLM = class extends llm.LLM {
5715
+ [RESILIENT] = true;
5716
+ #opts;
5717
+ #fallback = null;
5718
+ #listeners = /* @__PURE__ */ new Set();
5719
+ /** Models the provider rejected on this call (a call's pipeline is built per call): later turns go straight to the fallback. */
5720
+ rejections = new ModelRejectionCache();
5721
+ constructor(opts) {
5722
+ super();
5723
+ this.#opts = opts;
5724
+ }
5725
+ label() {
5726
+ return this.#opts.primary.label();
5727
+ }
5728
+ get model() {
5729
+ return this.#opts.primary.model;
5730
+ }
5731
+ get provider() {
5732
+ return this.#opts.primary.provider;
5733
+ }
5734
+ get apology() {
5735
+ return this.#opts.apology ?? LLM_APOLOGY;
5736
+ }
5737
+ get reasoningParams() {
5738
+ return this.#opts.reasoningParams === true;
5739
+ }
5740
+ /** The fallback LLM, built on first need; null when there is none or it is the agent's own model. */
5741
+ fallbackLLM() {
5742
+ if (!this.#opts.fallback) return null;
5743
+ this.#fallback ??= this.#opts.fallback();
5744
+ return sameModel(this.#fallback.model, this.model) ? null : this.#fallback;
5745
+ }
5746
+ /** Hear about every recovery (and every apology). Returns the unsubscribe. */
5747
+ onIncident(listener) {
5748
+ this.#listeners.add(listener);
5749
+ return () => this.#listeners.delete(listener);
5750
+ }
5751
+ /** @internal */
5752
+ report(incident) {
5753
+ const log = incident.kind === "apology" ? console.error : console.warn;
5754
+ log(`[voicelayer] llm ${incident.kind.replace(/_/g, " ")}`, incident);
5755
+ if (incident.kind === "model_fallback") {
5756
+ const call = getCurrentCallContext();
5757
+ recordModelFallback({
5758
+ model: incident.model,
5759
+ fallbackModel: incident.fallbackModel,
5760
+ code: incident.code,
5761
+ surface: "voice",
5762
+ ...call?.projectId ? { projectId: call.projectId } : {},
5763
+ ...call?.agentId ? { agentId: call.agentId } : {}
5764
+ });
5765
+ }
5766
+ for (const listener of this.#listeners) {
5767
+ try {
5768
+ listener(incident);
5769
+ } catch {
5770
+ }
5771
+ }
5772
+ }
5773
+ chat(args) {
5774
+ return new ResilientLLMStream(this, args);
5775
+ }
5776
+ prewarm() {
5777
+ this.#opts.primary.prewarm();
5778
+ }
5779
+ async aclose() {
5780
+ await Promise.all([this.#opts.primary.aclose(), this.#fallback?.aclose()]);
5781
+ }
5782
+ /** @internal */
5783
+ get primary() {
5784
+ return this.#opts.primary;
5785
+ }
5786
+ };
5787
+ sameModel = (a, b) => a.trim().toLowerCase() === b.trim().toLowerCase();
5788
+ ServedLLM = class extends llm.LLM {
5789
+ constructor(owner, served) {
5790
+ super();
5791
+ this.owner = owner;
5792
+ this.served = served;
5793
+ }
5794
+ owner;
5795
+ served;
5796
+ label() {
5797
+ return this.owner.label();
5798
+ }
5799
+ get model() {
5800
+ return this.served();
5801
+ }
5802
+ get provider() {
5803
+ return this.owner.provider;
5804
+ }
5805
+ chat(args) {
5806
+ return this.owner.chat(args);
5807
+ }
5808
+ emit(event, ...args) {
5809
+ return this.owner.emit(event, ...args);
5810
+ }
5811
+ };
5812
+ ResilientLLMStream = class extends llm.LLMStream {
5813
+ #owner;
5814
+ #args;
5815
+ #conn;
5816
+ /** The model answering this stream — what its metrics report. */
5817
+ #served;
5818
+ #current = null;
5819
+ constructor(owner, args) {
5820
+ const conn = args.connOptions ?? DEFAULT_API_CONNECT_OPTIONS;
5821
+ const served = { model: owner.model };
5822
+ super(new ServedLLM(owner, () => served.model), {
5823
+ chatCtx: args.chatCtx,
5824
+ ...args.toolCtx ? { toolCtx: args.toolCtx } : {},
5825
+ connOptions: { ...conn, maxRetry: 0 }
5826
+ });
5827
+ this.#owner = owner;
5828
+ this.#args = args;
5829
+ this.#conn = conn;
5830
+ this.#served = served;
5831
+ this.abortController.signal.addEventListener("abort", () => this.#current?.close());
5832
+ }
5833
+ get #hasTools() {
5834
+ return this.#args.toolCtx !== void 0 && Object.keys(this.#args.toolCtx).length > 0;
5835
+ }
5836
+ #extraKwargs(model2) {
5837
+ const base = this.#args.extraKwargs;
5838
+ if (!this.#owner.reasoningParams) return base;
5839
+ const effort = chatReasoningEffort(model2, { tools: this.#hasTools });
5840
+ if (effort === void 0) {
5841
+ if (!base || !("reasoning_effort" in base)) return base;
5842
+ const { reasoning_effort: _dropped, ...rest } = base;
5843
+ return rest;
5844
+ }
5845
+ return { ...base, reasoning_effort: effort };
5846
+ }
5847
+ /** One request on `target`, forwarding its chunks. Its failure is the inner LLM's 'error' event. */
5848
+ async #attempt(target) {
5849
+ let failure2 = null;
5850
+ const onError = (ev) => {
5851
+ failure2 ??= ev.error;
5852
+ };
5853
+ target.on("error", onError);
5854
+ let started2 = false;
5855
+ try {
5856
+ const extraKwargs = this.#extraKwargs(target.model);
5857
+ const stream = target.chat({
5858
+ ...this.#args,
5859
+ connOptions: { ...this.#conn, maxRetry: 0 },
5860
+ ...extraKwargs !== void 0 ? { extraKwargs } : {}
5861
+ });
5862
+ this.#current = stream;
5863
+ for await (const chunk of stream) {
5864
+ if (this.abortController.signal.aborted) break;
5865
+ started2 = true;
5866
+ this.queue.put(chunk);
5867
+ }
5868
+ } catch (err) {
5869
+ failure2 ??= err instanceof Error ? err : new Error(String(err));
5870
+ } finally {
5871
+ target.off("error", onError);
5872
+ this.#current = null;
5873
+ }
5874
+ return failure2 ? { ok: false, error: failure2, started: started2 } : { ok: true };
5875
+ }
5876
+ /** Run `target` until it answers, or a failure that retrying it won't fix. */
5877
+ async #run(target) {
5878
+ let effortRetried = false;
5879
+ for (let retries = 0; ; ) {
5880
+ const sent = this.#owner.reasoningParams ? chatReasoningEffort(target.model, { tools: this.#hasTools }) : void 0;
5881
+ const result = await this.#attempt(target);
5882
+ if (result.ok || result.started || this.abortController.signal.aborted) return result;
5883
+ if (this.#owner.reasoningParams && !effortRetried && learnReasoningEffort(target.model, { tools: this.#hasTools, sent }, result.error)) {
5884
+ effortRetried = true;
5885
+ this.#owner.report({
5886
+ kind: "reasoning_effort_adapted",
5887
+ model: target.model,
5888
+ message: providerErrorOf(result.error)?.message ?? result.error.message
5889
+ });
5890
+ continue;
5891
+ }
5892
+ if (!transient(result.error) || retries >= this.#conn.maxRetry) return result;
5893
+ const wait = intervalForRetry(this.#conn, retries);
5894
+ retries += 1;
5895
+ if (wait > 0) await new Promise((r) => setTimeout(r, wait));
5896
+ if (this.abortController.signal.aborted) return result;
5897
+ }
5898
+ }
5899
+ async run() {
5900
+ const owner = this.#owner;
5901
+ const model2 = owner.model;
5902
+ const fallback = owner.fallbackLLM();
5903
+ const known = fallback ? owner.rejections.get(model2) : null;
5904
+ let failure2;
5905
+ if (known && fallback) {
5906
+ owner.report({ kind: "model_fallback", model: model2, fallbackModel: fallback.model, code: known.code, message: known.message, cached: true });
5907
+ failure2 = new Error(known.message);
5908
+ } else {
5909
+ const first = await this.#run(owner.primary);
5910
+ if (first.ok || first.started || this.abortController.signal.aborted) return;
5911
+ failure2 = first.error;
5912
+ const rejection = modelRejectionOf(first.error);
5913
+ if (rejection) owner.rejections.set(model2, rejection);
5914
+ if (fallback) {
5915
+ const f = providerErrorOf(first.error);
5916
+ owner.report({
5917
+ kind: "model_fallback",
5918
+ model: model2,
5919
+ fallbackModel: fallback.model,
5920
+ code: rejection?.code ?? (f?.status ? String(f.status) : "error"),
5921
+ message: f?.message || first.error.message,
5922
+ cached: false
5923
+ });
5924
+ }
5925
+ }
5926
+ if (fallback) {
5927
+ this.#served.model = fallback.model;
5928
+ const second = await this.#run(fallback);
5929
+ if (second.ok || second.started || this.abortController.signal.aborted) return;
5930
+ failure2 = second.error;
5931
+ }
5932
+ owner.report({ kind: "apology", model: this.#served.model, message: providerErrorOf(failure2)?.message || failure2.message });
5933
+ this.queue.put({ id: `vl-apology-${Date.now()}`, delta: { role: "assistant", content: owner.apology } });
5934
+ }
5935
+ };
5936
+ }
5937
+ });
5938
+
5939
+ // src/providers/index.ts
5940
+ async function importOptional(spec, hint) {
5941
+ try {
5942
+ return await import(spec);
5943
+ } catch {
5944
+ throw new Error(`${hint}: "${spec}" is not installed. Add it with: pnpm add ${spec}`);
5945
+ }
5946
+ }
5947
+ function withCreds(opts, creds2) {
5948
+ if (creds2?.apiKey) opts["apiKey"] = creds2.apiKey;
5949
+ if (creds2?.baseURL) opts["baseURL"] = creds2.baseURL;
5950
+ return opts;
5951
+ }
5952
+ var deepgram, openai, anthropic, cartesia, elevenlabs, assemblyai, google, silero, livekitTurn, connector;
5953
+ var init_providers2 = __esm({
5954
+ "src/providers/index.ts"() {
5955
+ init_ssrf();
5956
+ init_transport_callback();
5957
+ init_connector_llm();
5958
+ init_wrap();
5959
+ deepgram = {
5960
+ stt(options = {}) {
5961
+ return async () => {
5962
+ const dg = await importOptional("@voicelayer/agents-plugin-deepgram", "deepgram.stt");
5963
+ return new dg.STT(
5964
+ withCreds(
5965
+ { model: options.model ?? "nova-3", language: options.language ?? "en-US" },
5966
+ options
5967
+ )
5968
+ );
5969
+ };
5970
+ },
5971
+ tts(options = {}) {
5972
+ return async () => {
5973
+ const dg = await importOptional("@voicelayer/agents-plugin-deepgram", "deepgram.tts");
5974
+ return new dg.TTS(
5975
+ withCreds({ model: options.model ?? "aura-2-harmonia-en" }, options)
5976
+ );
5977
+ };
5978
+ }
5979
+ };
5980
+ openai = {
5981
+ llm(options = {}) {
5982
+ return async () => {
5983
+ const oa = await importOptional("@voicelayer/agents-plugin-openai", "openai.llm");
5984
+ const { ResilientLLM: ResilientLLM2 } = await Promise.resolve().then(() => (init_resilient_llm(), resilient_llm_exports));
5985
+ const { PLATFORM_DEFAULT_CHAT_MODEL: PLATFORM_DEFAULT_CHAT_MODEL2 } = await Promise.resolve().then(() => (init_dist(), dist_exports));
5986
+ return new ResilientLLM2({
5987
+ primary: new oa.LLM(withCreds({ model: options.model ?? "gpt-4o-mini" }, options)),
5988
+ fallback: () => new oa.LLM(withCreds({ model: PLATFORM_DEFAULT_CHAT_MODEL2 }, options)),
5989
+ reasoningParams: true
5990
+ });
5991
+ };
5992
+ },
5993
+ tts(options = {}) {
5994
+ return async () => {
5995
+ const oa = await importOptional("@voicelayer/agents-plugin-openai", "openai.tts");
5996
+ const opts = { model: options.model ?? "tts-1" };
5997
+ if (options.voice) opts["voice"] = options.voice;
5998
+ if (options.instructions) opts["instructions"] = options.instructions;
5999
+ return new oa.TTS(withCreds(opts, options));
6000
+ };
6001
+ },
6002
+ // Speech-to-speech via the OpenAI Realtime API. Returns a RealtimeProvider
6003
+ // that the SDK uses in place of the stt/llm/tts pipeline.
6004
+ realtime(options = {}) {
6005
+ return async () => {
6006
+ const oa = await importOptional("@voicelayer/agents-plugin-openai", "openai.realtime");
6007
+ const opts = {};
6008
+ if (options.model) opts["model"] = options.model;
6009
+ if (options.voice) opts["voice"] = options.voice;
6010
+ return new oa.realtime.RealtimeModel(withCreds(opts, options));
6011
+ };
6012
+ }
6013
+ };
6014
+ anthropic = {
6015
+ llm(options = {}) {
6016
+ return async () => {
6017
+ const an = await importOptional("@livekit/agents-plugin-anthropic", "anthropic.llm");
6018
+ return new an.LLM(
6019
+ withCreds({ model: options.model ?? "claude-sonnet-5-5" }, options)
6020
+ );
6021
+ };
6022
+ }
6023
+ };
6024
+ cartesia = {
6025
+ tts(options = {}) {
6026
+ return async () => {
6027
+ const ct = await importOptional("@voicelayer/agents-plugin-cartesia", "cartesia.tts");
6028
+ return new ct.TTS(options);
6029
+ };
6030
+ },
6031
+ stt(options = {}) {
6032
+ return async () => {
6033
+ const ct = await importOptional("@voicelayer/agents-plugin-cartesia", "cartesia.stt");
6034
+ const opts = {};
6035
+ if (options.model) opts["model"] = options.model;
6036
+ if (options.language) opts["language"] = options.language;
6037
+ return new ct.STT(withCreds(opts, options));
6038
+ };
6039
+ }
6040
+ };
6041
+ elevenlabs = {
6042
+ tts(options = {}) {
6043
+ return async () => {
6044
+ const el = await importOptional("@voicelayer/agents-plugin-elevenlabs", "elevenlabs.tts");
6045
+ const opts = {};
6046
+ if (options.voice) opts["voiceId"] = options.voice;
6047
+ if (options.model) opts["model"] = options.model;
6048
+ return new el.TTS(withCreds(opts, options));
6049
+ };
6050
+ }
6051
+ };
6052
+ assemblyai = {
6053
+ stt(options = {}) {
6054
+ return async () => {
6055
+ const aai = await importOptional("@voicelayer/agents-plugin-assemblyai", "assemblyai.stt");
6056
+ const opts = {};
6057
+ if (options.language) opts["language"] = options.language;
6058
+ return new aai.STT(withCreds(opts, options));
6059
+ };
6060
+ }
6061
+ };
6062
+ google = {
6063
+ llm(options = {}) {
6064
+ return async () => {
6065
+ const g = await importOptional("@voicelayer/agents-plugin-google", "google.llm");
6066
+ const { ResilientLLM: ResilientLLM2 } = await Promise.resolve().then(() => (init_resilient_llm(), resilient_llm_exports));
6067
+ return new ResilientLLM2({
6068
+ primary: new g.LLM(withCreds({ model: options.model ?? "gemini-3.8-flash" }, options))
6069
+ });
6070
+ };
6071
+ },
6072
+ realtime(options = {}) {
6073
+ return async () => {
6074
+ const g = await importOptional("@voicelayer/agents-plugin-google", "google.realtime");
6075
+ const opts = {};
6076
+ if (options.model) opts["model"] = options.model;
6077
+ if (options.voice) opts["voice"] = options.voice;
6078
+ return new g.beta.realtime.RealtimeModel(withCreds(opts, options));
6079
+ };
6080
+ }
6081
+ };
6082
+ silero = {
6083
+ vad() {
6084
+ return async () => {
6085
+ const sil = await importOptional("@voicelayer/agents-plugin-silero", "silero.vad");
6086
+ return await sil.VAD.load();
6087
+ };
6088
+ }
6089
+ };
6090
+ livekitTurn = {
6091
+ english() {
6092
+ return async () => {
6093
+ const lk = await importOptional("@voicelayer/agents-plugin-livekit", "livekitTurn.english");
6094
+ return new lk.turnDetector.EnglishModel();
6095
+ };
6096
+ },
6097
+ multilingual() {
6098
+ return async () => {
6099
+ const lk = await importOptional("@voicelayer/agents-plugin-livekit", "livekitTurn.multilingual");
6100
+ return new lk.turnDetector.MultilingualModel();
6101
+ };
6102
+ }
6103
+ };
6104
+ connector = {
6105
+ llm(config = {}) {
6106
+ return async (call) => {
6107
+ if (config.url) {
6108
+ await assertPublicHttpsUrl(
6109
+ config.url,
6110
+ config.allowHosts ? { allowHosts: config.allowHosts } : {}
6111
+ );
6112
+ return openai.llm({
6113
+ ...config.model ? { model: config.model } : {},
6114
+ ...config.apiKey ? { apiKey: config.apiKey } : {},
6115
+ baseURL: config.url
6116
+ })(call);
6117
+ }
6118
+ const transport = config.transport ?? (config.onQuery ? callbackTransport(config.onQuery) : void 0);
6119
+ if (!transport) {
6120
+ throw new Error("connector.llm requires one of: url, onQuery, or transport");
6121
+ }
6122
+ const projectId = typeof call.metadata["projectId"] === "string" ? call.metadata["projectId"] : void 0;
6123
+ const llmOptions = {
6124
+ ...config.model ? { model: config.model } : {},
6125
+ ...config.temperature !== void 0 ? { temperature: config.temperature } : {},
6126
+ ...config.fallbackText ? { fallbackText: config.fallbackText } : {},
6127
+ ...call.callId ? { callId: call.callId } : {},
6128
+ ...projectId ? { projectId } : {}
6129
+ };
6130
+ return await createConnectorLLM(transport, llmOptions);
6131
+ };
6132
+ }
6133
+ };
6134
+ }
6135
+ });
6136
+
6137
+ // src/providers/registry.ts
6138
+ function entry(id, bits) {
6139
+ const identity = providerIdentity(id);
6140
+ if (!identity) throw new Error(`PROVIDER_REGISTRY: no provider identity in contracts for '${id}'`);
6141
+ return { name: identity.id, capabilities: identity.capabilities, ...bits };
6142
+ }
6143
+ function providerEntry(name) {
6144
+ if (!name) return void 0;
6145
+ return PROVIDER_REGISTRY[PROVIDER_ALIASES[name] ?? name];
5745
6146
  }
5746
- function backchannelCountCounter() {
5747
- return _backchannelCount ??= meter().createCounter("vl.interruption.backchannel_count", {
5748
- description: "Backchannel utterances detected",
5749
- unit: "{backchannels}"
5750
- });
6147
+ function resolveStt(name, o) {
6148
+ return (providerEntry(name)?.stt ?? PROVIDER_REGISTRY["deepgram"].stt)(o);
5751
6149
  }
5752
- function recordInterruptionMetrics(args) {
5753
- const base = {
5754
- [ATTR.projectId]: args.projectId,
5755
- [ATTR.callId]: args.callId
5756
- };
5757
- if (args.campaignId)
5758
- base[ATTR.campaignId] = args.campaignId;
5759
- if (args.totalDurationMs !== void 0 && args.totalDurationMs > 0) {
5760
- interruptionTotalMs().record(args.totalDurationMs, base);
5761
- }
5762
- if (args.predictionDurationMs !== void 0 && args.predictionDurationMs > 0) {
5763
- interruptionPredictionMs().record(args.predictionDurationMs, base);
5764
- }
5765
- if (args.detectionDelayMs !== void 0 && args.detectionDelayMs > 0) {
5766
- interruptionDetectionDelayMs().record(args.detectionDelayMs, base);
5767
- }
5768
- if (args.numInterruptions !== void 0 && args.numInterruptions > 0) {
5769
- interruptionCountCounter().add(args.numInterruptions, base);
5770
- }
5771
- if (args.numBackchannels !== void 0 && args.numBackchannels > 0) {
5772
- backchannelCountCounter().add(args.numBackchannels, base);
5773
- }
6150
+ function resolveTts(name, o) {
6151
+ return (providerEntry(name)?.tts ?? PROVIDER_REGISTRY["deepgram"].tts)(o);
5774
6152
  }
5775
- function realtimeSessionDurationMs() {
5776
- return _realtimeSessionDurationMs ??= meter().createCounter("vl.realtime.session_duration_ms", {
5777
- description: "Realtime model session duration for session-billed providers",
5778
- unit: "ms"
5779
- });
6153
+ function resolveLlm(name, o) {
6154
+ return (providerEntry(name)?.llm ?? PROVIDER_REGISTRY["openai"].llm)(withCurrentModel(name, o));
5780
6155
  }
5781
- function recordRealtimeSessionDuration(args) {
5782
- if (args.sessionDurationMs > 0) {
5783
- realtimeSessionDurationMs().add(args.sessionDurationMs, attrs(args));
5784
- }
6156
+ function resolveRealtime(name, o) {
6157
+ return (providerEntry(name)?.realtime ?? PROVIDER_REGISTRY["openai"].realtime)(withCurrentModel(name, o));
5785
6158
  }
5786
- function modelFallbacks() {
5787
- return _modelFallbacks ??= meter().createCounter("vl.llm.model_fallback", {
5788
- description: "Model calls the provider rejected that fell back to the platform default model",
5789
- unit: "{fallbacks}"
6159
+ function withCurrentModel(name, o) {
6160
+ if (!name || !o.model) return o;
6161
+ const current = currentModelFor(name, o.model);
6162
+ if (!current.retired) return o;
6163
+ console.warn("[agent] the configured model is retired; running its replacement", {
6164
+ provider: name,
6165
+ model: current.retired,
6166
+ replacement: current.model
5790
6167
  });
6168
+ return { ...o, model: current.model };
5791
6169
  }
5792
- function recordModelFallback(args) {
5793
- const out = {
5794
- "vl.model": args.model,
5795
- "vl.fallback_model": args.fallbackModel,
5796
- "vl.error_code": args.code
5797
- };
5798
- if (args.projectId)
5799
- out[ATTR.projectId] = args.projectId;
5800
- if (args.agentId)
5801
- out[ATTR.agentId] = args.agentId;
5802
- if (args.surface)
5803
- out["vl.surface"] = args.surface;
5804
- modelFallbacks().add(1, out);
5805
- }
5806
- var METER_NAME, METER_VERSION, USAGE_METRIC_NAMES, _llmInputTokens, _llmOutputTokens, _ttsChars, _sttSeconds, _firstTokenMs, _ttsStartMs, _eouDelayMs, _eouTranscriptionDelayMs, _eouOnTurnCompletedDelayMs, _interruptionTotalMs, _interruptionPredictionMs, _interruptionDetectionDelayMs, _interruptionCount, _backchannelCount, _realtimeSessionDurationMs, _modelFallbacks;
5807
- var init_metrics = __esm({
5808
- "../observability/dist/metrics.js"() {
5809
- init_attributes();
5810
- METER_NAME = "voicelayer";
5811
- METER_VERSION = "0.1.0";
5812
- USAGE_METRIC_NAMES = {
5813
- llmInputTokens: "vl.llm.input_tokens",
5814
- llmOutputTokens: "vl.llm.output_tokens",
5815
- ttsChars: "vl.tts.chars",
5816
- sttSeconds: "vl.stt.seconds"
6170
+ var model, lang, voice, creds, PROVIDER_REGISTRY;
6171
+ var init_registry = __esm({
6172
+ "src/providers/registry.ts"() {
6173
+ init_providers2();
6174
+ init_src();
6175
+ model = (o) => o.model ? { model: o.model } : {};
6176
+ lang = (o) => o.language ? { language: o.language } : {};
6177
+ voice = (o) => o.voice ? { voice: o.voice } : {};
6178
+ creds = (o) => o.creds ?? {};
6179
+ PROVIDER_REGISTRY = {
6180
+ deepgram: entry("deepgram", {
6181
+ failureClass: "vendor-api",
6182
+ engines: ["livekit"],
6183
+ stt: (o) => deepgram.stt({ ...model(o), ...lang(o), ...creds(o) }),
6184
+ // Deepgram folds the voice into the model id (aura-2-<name>-en): wire config
6185
+ // passes a voice, code config passes a model — both land as `model`.
6186
+ tts: (o) => {
6187
+ const m = o.voice ?? o.model;
6188
+ return deepgram.tts({ ...m ? { model: m } : {}, ...creds(o) });
6189
+ }
6190
+ }),
6191
+ openai: entry("openai", {
6192
+ failureClass: "vendor-api",
6193
+ engines: ["livekit", "text"],
6194
+ llm: (o) => openai.llm({ ...model(o), ...creds(o) }),
6195
+ tts: (o) => openai.tts({ ...voice(o), ...model(o), ...creds(o) }),
6196
+ realtime: (o) => openai.realtime({ ...model(o), ...voice(o), ...creds(o) })
6197
+ }),
6198
+ anthropic: entry("anthropic", {
6199
+ failureClass: "vendor-api",
6200
+ engines: ["livekit", "text"],
6201
+ llm: (o) => anthropic.llm({ ...model(o), ...creds(o) })
6202
+ }),
6203
+ google: entry("google", {
6204
+ failureClass: "vendor-api",
6205
+ engines: ["livekit", "text"],
6206
+ llm: (o) => google.llm({ ...model(o), ...creds(o) }),
6207
+ realtime: (o) => google.realtime({ ...model(o), ...voice(o), ...creds(o) })
6208
+ }),
6209
+ cartesia: entry("cartesia", {
6210
+ failureClass: "vendor-api",
6211
+ engines: ["livekit"],
6212
+ stt: (o) => cartesia.stt({ ...model(o), ...lang(o), ...creds(o) }),
6213
+ tts: (o) => cartesia.tts({ ...voice(o), ...model(o), ...creds(o) })
6214
+ }),
6215
+ elevenlabs: entry("elevenlabs", {
6216
+ failureClass: "vendor-api",
6217
+ engines: ["livekit"],
6218
+ tts: (o) => elevenlabs.tts({ ...voice(o), ...model(o), ...creds(o) })
6219
+ }),
6220
+ assemblyai: entry("assemblyai", {
6221
+ failureClass: "vendor-api",
6222
+ engines: ["livekit"],
6223
+ stt: (o) => assemblyai.stt({ ...lang(o), ...creds(o) })
6224
+ // no model knob
6225
+ }),
6226
+ silero: entry("silero", {
6227
+ failureClass: "local",
6228
+ engines: ["livekit"],
6229
+ vad: () => silero.vad()
6230
+ })
5817
6231
  };
5818
- _llmInputTokens = null;
5819
- _llmOutputTokens = null;
5820
- _ttsChars = null;
5821
- _sttSeconds = null;
5822
- _firstTokenMs = null;
5823
- _ttsStartMs = null;
5824
- _eouDelayMs = null;
5825
- _eouTranscriptionDelayMs = null;
5826
- _eouOnTurnCompletedDelayMs = null;
5827
- _interruptionTotalMs = null;
5828
- _interruptionPredictionMs = null;
5829
- _interruptionDetectionDelayMs = null;
5830
- _interruptionCount = null;
5831
- _backchannelCount = null;
5832
- _realtimeSessionDurationMs = null;
5833
- _modelFallbacks = null;
5834
- }
5835
- });
5836
- function recordTurnLatencySpan(args) {
5837
- if (!Number.isFinite(args.latencyMs) || args.latencyMs < 0)
5838
- return;
5839
- if (args.latencyMs === 0 && !args.allowZero)
5840
- return;
5841
- const attributes = {
5842
- [ATTR.projectId]: args.projectId,
5843
- [ATTR.callId]: args.callId,
5844
- [LATENCY_STAGE_ATTR]: args.stage,
5845
- [LATENCY_MS_ATTR]: args.latencyMs,
5846
- [LATENCY_REALTIME_ATTR]: args.realtime ? 1 : 0
5847
- };
5848
- if (args.campaignId)
5849
- attributes[ATTR.campaignId] = args.campaignId;
5850
- trace.getTracer(TRACER_NAME2).startSpan(LATENCY_SPAN_NAME, { attributes }).end();
5851
- }
5852
- var TRACER_NAME2, LATENCY_SPAN_NAME, LATENCY_STAGE_ATTR, LATENCY_MS_ATTR, LATENCY_REALTIME_ATTR;
5853
- var init_latency_span = __esm({
5854
- "../observability/dist/latency-span.js"() {
5855
- init_attributes();
5856
- TRACER_NAME2 = "@voicelayer/observability";
5857
- LATENCY_SPAN_NAME = "vl.turn_latency";
5858
- LATENCY_STAGE_ATTR = "vl.stage";
5859
- LATENCY_MS_ATTR = "vl.latency_ms";
5860
- LATENCY_REALTIME_ATTR = "vl.realtime";
5861
6232
  }
5862
6233
  });
5863
6234
 
5864
- // ../observability/dist/index.js
5865
- var init_dist2 = __esm({
5866
- "../observability/dist/index.js"() {
5867
- init_start();
5868
- init_call_context();
5869
- init_trace_propagation();
5870
- init_attributes();
5871
- init_metrics();
5872
- init_latency_span();
6235
+ // src/providers/llm.ts
6236
+ var init_llm = __esm({
6237
+ "src/providers/llm.ts"() {
6238
+ init_openai_default();
6239
+ init_registry();
5873
6240
  }
5874
6241
  });
5875
6242
 
@@ -9237,14 +9604,23 @@ function defaultComplete() {
9237
9604
  model: model2,
9238
9605
  fallbackModel: RUNNER_MODEL,
9239
9606
  helper: "agent runner",
9240
- call: (m) => client.chat.completions.create(
9241
- {
9242
- model: m,
9243
- ...chatCompletionParams(m, { ...temperature !== void 0 ? { temperature } : {}, reasoningEffort: "low" }),
9244
- messages,
9245
- ...tools.length ? { tools } : {}
9246
- },
9247
- { timeout: RUNNER_TIMEOUT_MS }
9607
+ // With tools, a model that takes effort 'none' gets 'none' (OpenAI refuses tools at any other effort on chat
9608
+ // completions for GPT-5.2 and later), and a reasoning_effort 400 is learned and retried once (burn-down G-45).
9609
+ call: (m) => withReasoningEffortRetry(
9610
+ { model: m, tools: tools.length > 0, requested: "low" },
9611
+ () => client.chat.completions.create(
9612
+ {
9613
+ model: m,
9614
+ ...chatCompletionParams(m, {
9615
+ ...temperature !== void 0 ? { temperature } : {},
9616
+ reasoningEffort: "low",
9617
+ tools: tools.length > 0
9618
+ }),
9619
+ messages,
9620
+ ...tools.length ? { tools } : {}
9621
+ },
9622
+ { timeout: RUNNER_TIMEOUT_MS }
9623
+ )
9248
9624
  ),
9249
9625
  warn: graphWarn
9250
9626
  });
@@ -12897,7 +13273,7 @@ async function resolvePipeline(cfg, call, preloadedVad) {
12897
13273
  () => language.startsWith("en") ? livekitTurn.english() : livekitTurn.multilingual(),
12898
13274
  () => language.startsWith("en") ? livekitTurn.english() : livekitTurn.multilingual()
12899
13275
  ) : null;
12900
- const [stt, llm3, tts, vad] = await Promise.all([
13276
+ const [stt, llm4, tts, vad] = await Promise.all([
12901
13277
  sttFactory2(call),
12902
13278
  llmFactory2(call),
12903
13279
  ttsFactory2(call),
@@ -12913,7 +13289,7 @@ async function resolvePipeline(cfg, call, preloadedVad) {
12913
13289
  });
12914
13290
  }
12915
13291
  }
12916
- return { stt, llm: llm3, tts, vad, turnDetector };
13292
+ return { stt, llm: llm4, tts, vad, turnDetector };
12917
13293
  }
12918
13294
  var DEFAULT_REQUIRED_ENV = [
12919
13295
  "LIVEKIT_URL",
@@ -16417,7 +16793,7 @@ async function answerTextTurn(deps) {
16417
16793
  const { turn, config } = deps;
16418
16794
  const agentId = turn.agentId;
16419
16795
  const now = deps.now ?? Date.now;
16420
- const { voice: voice3, llm: llm3, initializeLogger, loggerOptions } = await import('@livekit/agents');
16796
+ const { voice: voice3, llm: llm4, initializeLogger, loggerOptions } = await import('@livekit/agents');
16421
16797
  if (loggerOptions() === void 0) initializeLogger({ pretty: false, level: "warn" });
16422
16798
  let effective = config;
16423
16799
  let pipeline = null;
@@ -16459,7 +16835,7 @@ async function answerTextTurn(deps) {
16459
16835
  getCtx
16460
16836
  });
16461
16837
  const instructions = processRt.augmentPrompt(composeSystemPrompt(effective.prompt, effective.routingInstructions));
16462
- const chatCtx = llm3.ChatContext.empty();
16838
+ const chatCtx = llm4.ChatContext.empty();
16463
16839
  let skipReply = false;
16464
16840
  for (const h of turn.history) {
16465
16841
  if (h.role === "user" && security && security.inputGuard.check(h.content).action === "block") {
@@ -19040,6 +19416,7 @@ function toolAuditObserver(emit) {
19040
19416
  }
19041
19417
 
19042
19418
  // src/agent.ts
19419
+ init_resilient_llm();
19043
19420
  init_helper_models();
19044
19421
  var FLOW_HANGUP_GRACE_MS = Number(process.env["VL_FLOW_HANGUP_GRACE_MS"] ?? "1200");
19045
19422
  function isLiveKitChildProcess() {
@@ -19372,6 +19749,9 @@ var Agent = class {
19372
19749
  void trackWrite(sdkClient.calls.appendEvent(opEventCallId, { kind, payload })).catch(() => {
19373
19750
  });
19374
19751
  };
19752
+ if (isResilientLLM(pipeline.llm)) {
19753
+ pipeline.llm.onIncident((incident) => emitOpEvent(`engine.llm.${incident.kind}`, { ...incident }));
19754
+ }
19375
19755
  const graphEvents = graphMode ? withOpEventFanout(createGraphEvents(processRt), emitOpEvent) : null;
19376
19756
  const isOutbound = stringValue(dispatchMdEarly["direction"]) === "outbound";
19377
19757
  const perCallAmd = stringValue(dispatchMdEarly["amd"]);