@voicelayer/sdk 0.6.1 → 0.6.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -14,11 +14,11 @@ import { LoggerProvider, BatchLogRecordProcessor } from '@opentelemetry/sdk-logs
14
14
  import { PeriodicExportingMetricReader } from '@opentelemetry/sdk-metrics';
15
15
  import { NodeSDK } from '@opentelemetry/sdk-node';
16
16
  import { ATTR_SERVICE_VERSION, ATTR_SERVICE_NAME } from '@opentelemetry/semantic-conventions';
17
+ import { llm, DEFAULT_API_CONNECT_OPTIONS, intervalForRetry, voice as voice$1, APIStatusError, APITimeoutError, APIConnectionError } from '@livekit/agents';
17
18
  import http from 'http';
18
19
  import https from 'https';
19
20
  import { Readable } from 'stream';
20
21
  import { lookup } from 'dns/promises';
21
- import { voice as voice$1, llm } from '@livekit/agents';
22
22
  import { getQuickJS, shouldInterruptAfterDeadline } from 'quickjs-emscripten';
23
23
  import { randomUUID, createHash } from 'crypto';
24
24
  import { setup, createActor } from 'xstate';
@@ -750,7 +750,7 @@ var init_consultation = __esm({
750
750
  });
751
751
  }
752
752
  });
753
- var CALL_CONTROL_PUBSUB_CHANNEL, CALL_CONTROL_ACK_CHANNEL, CALL_CONTROL_READY_KEY, CallEventKind, OP_CALL_EVENT_KIND, OpCallEventKind, commandId, SayCommand, HangupCommand, DtmfCommand, InstructCommand, InjectContextCommand, CallControlCommand;
753
+ var CALL_CONTROL_PUBSUB_CHANNEL, CALL_CONTROL_ACK_CHANNEL, CALL_CONTROL_READY_KEY, CallEventKind, OP_CALL_EVENT_KIND, OpCallEventKind, LatencyMs, TURN_LATENCY_MAX_PER_CALL, commandId, SayCommand, HangupCommand, DtmfCommand, InstructCommand, InjectContextCommand, CallControlCommand;
754
754
  var init_call_events = __esm({
755
755
  "../contracts/src/call-events.ts"() {
756
756
  init_consultation();
@@ -779,7 +779,9 @@ var init_call_events = __esm({
779
779
  "dtmf.sent",
780
780
  // A host's runtime intervention (burn-down G-6): call_say / call_send_guidance / call_inject_context / call_instruct,
781
781
  // with masked args and the API key — written by the API only, never through the worker's /events endpoint.
782
- "mcp.interaction"
782
+ "mcp.interaction",
783
+ // One per caller turn the agent answered (burn-down G-44): where that turn's wait went — see TurnLatencyPayload.
784
+ "turn.latency"
783
785
  ]);
784
786
  OP_CALL_EVENT_KIND = /^(tool|handoff|lookup|record|notify|engine)\.[a-z0-9_]{1,48}(\.[a-z0-9_]{1,48})?$/;
785
787
  OpCallEventKind = z.string().regex(OP_CALL_EVENT_KIND);
@@ -806,6 +808,18 @@ var init_call_events = __esm({
806
808
  code: z.number().int().min(0).max(15),
807
809
  participantId: z.string().optional()
808
810
  });
811
+ LatencyMs = z.number().int().min(0).max(6e5).nullable();
812
+ TURN_LATENCY_MAX_PER_CALL = 500;
813
+ z.object({
814
+ turn: z.number().int().min(1),
815
+ atMs: z.number().int().min(0),
816
+ endpointingMs: LatencyMs,
817
+ sttMs: LatencyMs,
818
+ llmMs: LatencyMs,
819
+ ttsMs: LatencyMs,
820
+ endToEndMs: LatencyMs,
821
+ realtime: z.boolean().optional()
822
+ });
809
823
  z.object({
810
824
  totalUsd: z.string(),
811
825
  breakdown: z.object({
@@ -924,9 +938,68 @@ var init_text_turns = __esm({
924
938
  ]);
925
939
  }
926
940
  });
941
+ var PLATFORM_MAX_CALL_MINUTES, SIP_LEG_MAX_CALL_DURATION_SECONDS, INBOUND_REFUSED_MESSAGE, Message, CallerRateLimit, CallLimits, CallerChannel, CallDurationLimits, INBOUND_REFUSED_END_REASON_PREFIX, CallAdmissionResponse;
942
+ var init_call_limits = __esm({
943
+ "../contracts/src/call-limits.ts"() {
944
+ PLATFORM_MAX_CALL_MINUTES = 240;
945
+ SIP_LEG_MAX_CALL_DURATION_SECONDS = (PLATFORM_MAX_CALL_MINUTES + 15) * 60;
946
+ INBOUND_REFUSED_MESSAGE = "Sorry, we can't take your call right now. Please try again later. Goodbye.";
947
+ Message = z.string().trim().min(1).max(300);
948
+ CallerRateLimit = z.object({
949
+ enabled: z.boolean().optional(),
950
+ perTenMinutes: z.number().int().min(1).max(100).optional(),
951
+ perDay: z.number().int().min(1).max(1e3).optional(),
952
+ withheldPerTenMinutes: z.number().int().min(1).max(100).optional(),
953
+ withheldPerDay: z.number().int().min(1).max(1e3).optional()
954
+ });
955
+ CallLimits = z.object({
956
+ /** The agent's own ceiling in minutes; clamped to the workspace's plan ceiling. Absent ⇒ DEFAULT_MAX_CALL_MINUTES. */
957
+ maxDurationMinutes: z.number().int().min(1).max(PLATFORM_MAX_CALL_MINUTES).optional(),
958
+ /** Seconds before the limit the wrap-up line is spoken; 0 ⇒ no warning. */
959
+ wrapUpWarningSeconds: z.number().int().min(0).max(300).optional(),
960
+ wrapUpMessage: Message.optional(),
961
+ goodbyeMessage: Message.optional(),
962
+ callerRateLimit: CallerRateLimit.optional(),
963
+ rateLimitedMessage: Message.optional()
964
+ });
965
+ CallerChannel = z.enum(["phone", "web"]);
966
+ z.object({
967
+ roomName: z.string().min(1).max(256),
968
+ /** From the dispatch metadata: lets the API admit (gate) a room whose call row doesn't exist yet. */
969
+ agentId: z.string().max(128).optional(),
970
+ phoneNumberId: z.string().max(128).optional(),
971
+ caller: z.object({
972
+ channel: CallerChannel,
973
+ /** The caller's number as the carrier presented it; null / absent / not E.164 ⇒ the withheld bucket. */
974
+ number: z.string().max(64).nullable().optional()
975
+ })
976
+ });
977
+ CallDurationLimits = z.object({
978
+ /** The agent's own ceiling (already clamped to the plan), or null when it set none. */
979
+ agentMaxSeconds: z.number().int().min(1).nullable(),
980
+ /** The plan ceiling — the platform bound no call outlives. */
981
+ ceilingSeconds: z.number().int().min(1),
982
+ wrapUpWarningSeconds: z.number().int().min(0),
983
+ wrapUpMessage: z.string(),
984
+ goodbyeMessage: z.string()
985
+ });
986
+ INBOUND_REFUSED_END_REASON_PREFIX = "inbound_refused:";
987
+ CallAdmissionResponse = z.discriminatedUnion("admitted", [
988
+ z.object({ admitted: z.literal(true), limits: CallDurationLimits }),
989
+ z.object({
990
+ admitted: z.literal(false),
991
+ /** 'rate_limited:caller' | 'inbound_refused:<code>' — already recorded as the call's end reason. */
992
+ endReason: z.string(),
993
+ /** The fixed line to speak before hanging up. */
994
+ message: z.string()
995
+ })
996
+ ]);
997
+ }
998
+ });
927
999
  var VoiceConfig, ModelConfig, SttConfig, RealtimeConfig, PipelineMode, ToolBinding, ProcessFieldType, ProcessFieldSpec, ProcessCompleteWhen, ProcessBackendAckSpec, ProcessSpec, TriggerConditionSpec, TriggerActionSpec, TriggerSpec, RequiredInfoField, AgentConfig;
928
1000
  var init_agent_config = __esm({
929
1001
  "../contracts/src/agent-config.ts"() {
1002
+ init_call_limits();
930
1003
  init_consultation();
931
1004
  VoiceConfig = z.object({
932
1005
  // Finite, vetted list — platform-level enum.
@@ -1090,6 +1163,9 @@ var init_agent_config = __esm({
1090
1163
  // LLM slot through the connector instead of a first-party model. Null/absent →
1091
1164
  // first-party pipeline (unchanged default). See docs/specs/voice-brain-connector.
1092
1165
  brainConnectorId: z.string().uuid().nullable().optional(),
1166
+ // Per-agent call bounds: caller rate limit, max duration and their spoken lines (burn-down G-41, G-43). Absent ⇒
1167
+ // every default (call-limits.ts) — the rate limit is ON by default.
1168
+ callLimits: CallLimits.optional(),
1093
1169
  // Provenance: 'sdk' (hand-coded defineAgent worker), 'flow' (canvas deploy),
1094
1170
  // 'playbook' (agent-mode deploy), or 'connector' (brain-connector agent).
1095
1171
  // DERIVED from the linked agents row's deploy tags (flowId / playbookId /
@@ -1445,6 +1521,7 @@ var init_agent_versions = __esm({
1445
1521
  "../contracts/src/agent-versions.ts"() {
1446
1522
  init_agent_config();
1447
1523
  init_agents();
1524
+ init_call_limits();
1448
1525
  init_consultation();
1449
1526
  z.object({
1450
1527
  systemPrompt: z.string().max(64e3),
@@ -1459,6 +1536,8 @@ var init_agent_versions = __esm({
1459
1536
  consultation: ConsultationPolicy,
1460
1537
  tags: z.array(z.string().max(64)).max(32).nullable(),
1461
1538
  brainConnectorId: z.string().uuid().nullable(),
1539
+ // Versions published before call limits existed carry none: read as null (every default).
1540
+ callLimits: CallLimits.nullable().default(null),
1462
1541
  flowVersion: z.number().int().nullable(),
1463
1542
  processSchema: ProcessSchemaDTO.nullable()
1464
1543
  });
@@ -1494,6 +1573,7 @@ var init_agent_versions = __esm({
1494
1573
  "language",
1495
1574
  "tools",
1496
1575
  "brainConnectorId",
1576
+ "callLimits",
1497
1577
  "consultation",
1498
1578
  "tags",
1499
1579
  "flowVersion",
@@ -1635,6 +1715,9 @@ var init_model_catalog = __esm({
1635
1715
  CostEstimateBilledPerMinute = z.object({
1636
1716
  platformFee: z.number().nonnegative(),
1637
1717
  providerPassthrough: z.number().nonnegative(),
1718
+ // The flat own-key fee per minute (burn-down B-26); 0 when every provider runs on the platform's keys. Optional
1719
+ // only for responses from an API older than the fee.
1720
+ ownKeyFee: z.number().nonnegative().optional(),
1638
1721
  telephony: z.number().nonnegative(),
1639
1722
  allIn: z.number().nonnegative()
1640
1723
  });
@@ -1900,9 +1983,46 @@ var init_call_review = __esm({
1900
1983
  });
1901
1984
  }
1902
1985
  });
1986
+ var CAPTURED_FIELDS_MAX, CAPTURED_STRING_MAX_CHARS, CAPTURED_ARRAY_MAX_ITEMS, CapturedFieldValue, CapturedField, StructuredOutputs;
1987
+ var init_call_outcome = __esm({
1988
+ "../contracts/src/call-outcome.ts"() {
1989
+ init_agent_config();
1990
+ CAPTURED_FIELDS_MAX = 200;
1991
+ CAPTURED_STRING_MAX_CHARS = 8e3;
1992
+ CAPTURED_ARRAY_MAX_ITEMS = 100;
1993
+ CapturedFieldValue = z.union([
1994
+ z.string().max(CAPTURED_STRING_MAX_CHARS),
1995
+ z.number().finite(),
1996
+ z.boolean(),
1997
+ z.array(z.string().max(CAPTURED_STRING_MAX_CHARS)).max(CAPTURED_ARRAY_MAX_ITEMS),
1998
+ z.null()
1999
+ ]);
2000
+ CapturedField = z.object({
2001
+ name: z.string().min(1).max(64),
2002
+ type: ProcessFieldType,
2003
+ required: z.boolean(),
2004
+ // false ⇒ the call never captured it; `value` is then null
2005
+ filled: z.boolean(),
2006
+ value: CapturedFieldValue
2007
+ });
2008
+ StructuredOutputs = z.record(CapturedFieldValue);
2009
+ z.object({
2010
+ // Every declared field, in declaration order; empty when the agent declares none.
2011
+ fields: z.array(CapturedField).max(CAPTURED_FIELDS_MAX)
2012
+ }).superRefine((body, ctx) => {
2013
+ const seen = /* @__PURE__ */ new Set();
2014
+ for (const f of body.fields) {
2015
+ if (seen.has(f.name)) ctx.addIssue({ code: z.ZodIssueCode.custom, message: `duplicate field ${f.name}` });
2016
+ seen.add(f.name);
2017
+ if (!f.filled && f.value !== null) ctx.addIssue({ code: z.ZodIssueCode.custom, message: `${f.name}: an unfilled field has no value` });
2018
+ }
2019
+ });
2020
+ }
2021
+ });
1903
2022
  var TranscriptTurn, ConsultationAudit, ToolCallAudit, McpInteractionAudit;
1904
2023
  var init_call_result = __esm({
1905
2024
  "../contracts/src/call-result.ts"() {
2025
+ init_call_outcome();
1906
2026
  TranscriptTurn = z.object({
1907
2027
  speaker: z.enum(["caller", "agent"]),
1908
2028
  text: z.string(),
@@ -1951,9 +2071,13 @@ var init_call_result = __esm({
1951
2071
  startedAt: z.string().datetime(),
1952
2072
  endedAt: z.string().datetime(),
1953
2073
  durationMs: z.number().int().min(0),
2074
+ // Final segments only, in call order.
1954
2075
  transcript: z.array(TranscriptTurn),
1955
- // Whatever the agent's process-schema captured. No prescribed shape.
1956
- structuredOutputs: z.record(z.unknown()).optional(),
2076
+ // The filled fields of the agent's process / flow, keyed by name (burn-down G-20). Absent when the agent declares no
2077
+ // fields, or its worker never reported them.
2078
+ structuredOutputs: StructuredOutputs.optional(),
2079
+ // Every declared field with what the call captured and whether it was filled — the same list `call.ended` carries.
2080
+ fields: z.array(CapturedField).optional(),
1957
2081
  recordingUrl: z.string().url().optional(),
1958
2082
  cost: z.object({
1959
2083
  totalUsd: z.string(),
@@ -1965,9 +2089,11 @@ var init_call_result = __esm({
1965
2089
  });
1966
2090
  }
1967
2091
  });
1968
- var WEBHOOK_EVENT_TYPES, CallWebhookDirection, CallWebhookOutcome, CallWebhookAttributes, Party;
2092
+ var WEBHOOK_EVENT_TYPES, CallWebhookDirection, CallWebhookOutcome, CallWebhookAttributes, Party, CALL_ENDED_PAYLOAD_VERSION, CallEndedTranscript;
1969
2093
  var init_webhook_events = __esm({
1970
2094
  "../contracts/src/webhook-events.ts"() {
2095
+ init_call_result();
2096
+ init_call_outcome();
1971
2097
  WEBHOOK_EVENT_TYPES = ["call.started", "call.ended", "recording.ready"];
1972
2098
  z.enum(WEBHOOK_EVENT_TYPES);
1973
2099
  CallWebhookDirection = z.enum(["inbound", "outbound", "web"]);
@@ -1985,7 +2111,18 @@ var init_webhook_events = __esm({
1985
2111
  // set, it is the same value `call.ended` carries; `call.ended` always has one.
1986
2112
  direction: CallWebhookDirection.nullable()
1987
2113
  });
2114
+ CALL_ENDED_PAYLOAD_VERSION = 2;
2115
+ CallEndedTranscript = z.object({
2116
+ // Final segments only (what was said, never an interim STT guess), in call order. Empty when `truncated`.
2117
+ turns: z.array(TranscriptTurn),
2118
+ // How many final turns the call has — the length of `turns` unless truncated.
2119
+ turnCount: z.number().int().min(0),
2120
+ truncated: z.boolean(),
2121
+ // Where the whole transcript is: GET /v1/calls/:callId/result (its `transcript`). Always set.
2122
+ fetchPath: z.string().regex(/^\/v1\/calls\/[0-9a-f-]{36}\/result$/)
2123
+ });
1988
2124
  z.object({
2125
+ version: z.literal(CALL_ENDED_PAYLOAD_VERSION),
1989
2126
  callId: z.string().uuid(),
1990
2127
  durationMs: z.number().int().min(0),
1991
2128
  endedAt: z.string().datetime(),
@@ -1996,7 +2133,20 @@ var init_webhook_events = __esm({
1996
2133
  toE164: Party,
1997
2134
  direction: CallWebhookDirection,
1998
2135
  outcome: CallWebhookOutcome,
1999
- attributes: CallWebhookAttributes
2136
+ attributes: CallWebhookAttributes,
2137
+ // Every field the agent declared (its process, or a flow's slots), in declaration order, with what the call captured
2138
+ // and whether it was filled. Empty when the agent declares none.
2139
+ fields: z.array(CapturedField),
2140
+ // The filled fields as one object keyed by name (as GET /v1/calls/:id/result returns them). Absent when the agent
2141
+ // declares no fields.
2142
+ structuredOutputs: StructuredOutputs.optional(),
2143
+ transcript: CallEndedTranscript,
2144
+ // true: the agent's worker confirmed its last transcript segment and captured field had landed before this payload
2145
+ // was built. false: built without that word (a worker that crashed or runs an SDK before 0.6.2, or one that didn't
2146
+ // answer within the hold) — the transcript and fields are what had arrived; GET /v1/calls/:id/result has any later.
2147
+ finalized: z.boolean(),
2148
+ // true: the payload would have passed CALL_ENDED_PAYLOAD_MAX_BYTES, so values were left out (see the cap) — fetch them.
2149
+ valuesOmitted: z.boolean()
2000
2150
  });
2001
2151
  z.object({
2002
2152
  callId: z.string().uuid(),
@@ -2942,166 +3092,6 @@ var init_telephony = __esm({
2942
3092
  });
2943
3093
  }
2944
3094
  });
2945
- var SHORTENER_HOSTS, PLACEHOLDER_HOSTS, PublicUrl, A2pBusinessType, A2pBusinessInfo, A2pOptInType, A2pUseCase, A2pMessagingProfile, A2pState;
2946
- var init_a2p = __esm({
2947
- "../contracts/src/a2p.ts"() {
2948
- SHORTENER_HOSTS = [
2949
- "bit.ly",
2950
- "tinyurl.com",
2951
- "t.co",
2952
- "goo.gl",
2953
- "ow.ly",
2954
- "is.gd",
2955
- "buff.ly",
2956
- "rebrand.ly",
2957
- "cutt.ly",
2958
- "shorturl.at",
2959
- "tiny.cc"
2960
- ];
2961
- PLACEHOLDER_HOSTS = ["acme.com", "example.com", "example.org", "test.com", "localhost"];
2962
- PublicUrl = z.string().url().refine((v) => v.startsWith("https://"), "must be https").refine((v) => {
2963
- try {
2964
- const host = new URL(v).hostname.toLowerCase().replace(/^www\./, "");
2965
- return !SHORTENER_HOSTS.includes(host);
2966
- } catch {
2967
- return false;
2968
- }
2969
- }, "public URL shorteners are rejected by carriers \u2014 use your own domain").refine((v) => {
2970
- try {
2971
- const host = new URL(v).hostname.toLowerCase().replace(/^www\./, "");
2972
- return !PLACEHOLDER_HOSTS.some((p) => host === p || host.endsWith(`.${p}`));
2973
- } catch {
2974
- return false;
2975
- }
2976
- }, "placeholder domain \u2014 reviewers will follow this link and reject the campaign");
2977
- A2pBusinessType = z.enum([
2978
- "Sole Proprietorship",
2979
- "Partnership",
2980
- "Corporation",
2981
- "Co-operative",
2982
- "Limited Liability Corporation",
2983
- "Non-profit Corporation"
2984
- ]);
2985
- A2pBusinessInfo = z.object({
2986
- legalName: z.string().trim().min(2).max(200),
2987
- /** EIN (US) or equivalent registration number. */
2988
- registrationNumber: z.string().trim().min(4).max(50),
2989
- businessType: A2pBusinessType,
2990
- /** Publicly reachable production site — not staging, not a 404. */
2991
- website: PublicUrl,
2992
- industry: z.string().trim().min(2).max(60),
2993
- address: z.object({
2994
- street: z.string().trim().min(2).max(200),
2995
- city: z.string().trim().min(1).max(100),
2996
- region: z.string().trim().min(1).max(100),
2997
- postalCode: z.string().trim().min(2).max(20),
2998
- isoCountry: z.string().trim().length(2)
2999
- }),
3000
- authorizedRep: z.object({
3001
- firstName: z.string().trim().min(1).max(100),
3002
- lastName: z.string().trim().min(1).max(100),
3003
- email: z.string().trim().email(),
3004
- phone: z.string().trim().regex(/^\+[1-9]\d{6,14}$/, "must be E.164"),
3005
- jobTitle: z.string().trim().min(2).max(100)
3006
- })
3007
- });
3008
- A2pOptInType = z.enum(["WEB_FORM", "PAPER_FORM", "VERBAL", "VIA_TEXT", "MOBILE_QR_CODE"]);
3009
- A2pUseCase = z.enum([
3010
- "MIXED",
3011
- "CUSTOMER_CARE",
3012
- "MARKETING",
3013
- "ACCOUNT_NOTIFICATION",
3014
- "2FA",
3015
- "DELIVERY_NOTIFICATION",
3016
- "HIGHER_EDUCATION",
3017
- "POLLING_VOTING",
3018
- "PUBLIC_SERVICE_ANNOUNCEMENT",
3019
- "LOW_VOLUME"
3020
- ]);
3021
- A2pMessagingProfile = z.object({
3022
- useCase: A2pUseCase,
3023
- /** Specific, not generic. "We send texts" gets rejected; describe the actual
3024
- * messages and when they are sent. */
3025
- description: z.string().trim().min(40).max(4096),
3026
- /**
3027
- * The single most-rejected field. Must describe HOW people opt in, state the
3028
- * message frequency, include the "message and data rates may apply"
3029
- * disclosure, and link to publicly reachable evidence. Twilio's API bounds it
3030
- * to 40–2049 characters.
3031
- */
3032
- messageFlow: z.string().trim().min(40).max(2049),
3033
- optInType: A2pOptInType,
3034
- /** Publicly accessible screenshots/pages showing the opt-in. Reviewers open
3035
- * these; anything behind a login fails. */
3036
- optInEvidenceUrls: z.array(PublicUrl).min(1).max(5),
3037
- /** Real messages the customer will send. Must reflect the declared use case
3038
- * and carry opt-out language. */
3039
- messageSamples: z.array(z.string().trim().min(10).max(1024)).min(2).max(5),
3040
- privacyPolicyUrl: PublicUrl,
3041
- termsAndConditionsUrl: PublicUrl,
3042
- hasEmbeddedLinks: z.boolean().default(false),
3043
- hasEmbeddedPhone: z.boolean().default(false)
3044
- });
3045
- z.object({
3046
- business: A2pBusinessInfo,
3047
- messaging: A2pMessagingProfile,
3048
- /** Register against Twilio's mock endpoints — exercises the full pipeline
3049
- * with no fees and no real carrier submission. Used in CI and staging. */
3050
- mock: z.boolean().default(false)
3051
- }).superRefine((v, ctx) => {
3052
- const flow = v.messaging.messageFlow.toLowerCase();
3053
- if (!/(msg|message)\s*(&|and)\s*data rates/.test(flow)) {
3054
- ctx.addIssue({
3055
- code: z.ZodIssueCode.custom,
3056
- path: ["messaging", "messageFlow"],
3057
- message: 'must include a "Message and data rates may apply" disclosure \u2014 carriers reject without it'
3058
- });
3059
- }
3060
- if (!/\d/.test(flow) || !/(msg|message|text)/.test(flow)) {
3061
- ctx.addIssue({
3062
- code: z.ZodIssueCode.custom,
3063
- path: ["messaging", "messageFlow"],
3064
- message: 'must state message frequency, e.g. "Up to 4 msgs/month"'
3065
- });
3066
- }
3067
- const hasOptOut = v.messaging.messageSamples.some((s) => /stop/i.test(s));
3068
- if (!hasOptOut) {
3069
- ctx.addIssue({
3070
- code: z.ZodIssueCode.custom,
3071
- path: ["messaging", "messageSamples"],
3072
- message: 'at least one sample must include opt-out language (e.g. "Reply STOP to opt out")'
3073
- });
3074
- }
3075
- });
3076
- A2pState = z.enum([
3077
- "none",
3078
- "profile_pending",
3079
- "profile_approved",
3080
- "profile_failed",
3081
- "brand_pending",
3082
- "brand_approved",
3083
- "brand_failed",
3084
- "campaign_pending",
3085
- "messaging_ready",
3086
- "campaign_failed"
3087
- ]);
3088
- z.object({
3089
- state: A2pState,
3090
- customerProfileSid: z.string().nullable(),
3091
- trustProductSid: z.string().nullable(),
3092
- brandSid: z.string().nullable(),
3093
- messagingServiceSid: z.string().nullable(),
3094
- campaignSid: z.string().nullable(),
3095
- mock: z.boolean(),
3096
- /** Carrier/Twilio rejection details, surfaced verbatim so the customer can
3097
- * fix the specific field rather than guess. */
3098
- failures: z.array(z.object({ code: z.number().nullable(), field: z.string().nullable(), message: z.string() })).default([]),
3099
- /** Plain-language next step for the dashboard. */
3100
- nextAction: z.string().nullable(),
3101
- updatedAt: z.string().nullable()
3102
- });
3103
- }
3104
- });
3105
3095
  var ConnectorMode, ConnectorStatus, ConnectorNormalize;
3106
3096
  var init_connector = __esm({
3107
3097
  "../contracts/src/connector.ts"() {
@@ -3218,7 +3208,7 @@ var init_connector_stream = __esm({
3218
3208
  ]);
3219
3209
  }
3220
3210
  });
3221
- var FLOW_BOOT_FAILURES;
3211
+ var FLOW_BOOT_FAILURES, AGENT_END_REASONS;
3222
3212
  var init_call_end_reason = __esm({
3223
3213
  "../contracts/src/call-end-reason.ts"() {
3224
3214
  FLOW_BOOT_FAILURES = [
@@ -3229,9 +3219,11 @@ var init_call_end_reason = __esm({
3229
3219
  "worker_identity_refused",
3230
3220
  "provider_unavailable"
3231
3221
  ];
3232
- z.enum(
3233
- FLOW_BOOT_FAILURES.map((cause) => `flow_boot:${cause}`)
3234
- );
3222
+ AGENT_END_REASONS = [
3223
+ "max_duration",
3224
+ ...FLOW_BOOT_FAILURES.map((cause) => `flow_boot:${cause}`)
3225
+ ];
3226
+ z.enum(AGENT_END_REASONS);
3235
3227
  }
3236
3228
  });
3237
3229
  var EnvironmentSpec;
@@ -4427,6 +4419,23 @@ var init_wallet = __esm({
4427
4419
  });
4428
4420
  }
4429
4421
  });
4422
+ var PipelineSlotKey;
4423
+ var init_pipeline_keys = __esm({
4424
+ "../contracts/src/pipeline-keys.ts"() {
4425
+ PipelineSlotKey = z.object({
4426
+ /** The provider as the slot reports it (LiveKit's label: `openai`, `api.openai.com`, `Deepgram`, …). */
4427
+ provider: z.string().min(1).max(200),
4428
+ /** true = the workspace's own key built it; false = the platform's. */
4429
+ byok: z.boolean()
4430
+ });
4431
+ z.object({
4432
+ llm: PipelineSlotKey.optional(),
4433
+ stt: PipelineSlotKey.optional(),
4434
+ tts: PipelineSlotKey.optional(),
4435
+ realtime: PipelineSlotKey.optional()
4436
+ });
4437
+ }
4438
+ });
4430
4439
 
4431
4440
  // ../contracts/src/index.ts
4432
4441
  var init_src = __esm({
@@ -4456,6 +4465,7 @@ var init_src = __esm({
4456
4465
  init_tests();
4457
4466
  init_call_review();
4458
4467
  init_call_result();
4468
+ init_call_outcome();
4459
4469
  init_webhook_events();
4460
4470
  init_numbers();
4461
4471
  init_project_limits();
@@ -4471,12 +4481,12 @@ var init_src = __esm({
4471
4481
  init_connection();
4472
4482
  init_secret_headers();
4473
4483
  init_telephony();
4474
- init_a2p();
4475
4484
  init_connector();
4476
4485
  init_connector_stream();
4477
4486
  init_flow_compile();
4478
4487
  init_flow_compile_check();
4479
4488
  init_call_end_reason();
4489
+ init_call_limits();
4480
4490
  init_flow_enrichment();
4481
4491
  init_environments();
4482
4492
  init_variables();
@@ -4492,6 +4502,7 @@ var init_src = __esm({
4492
4502
  init_wallet();
4493
4503
  init_network_address();
4494
4504
  init_dispatch_metadata();
4505
+ init_pipeline_keys();
4495
4506
  }
4496
4507
  });
4497
4508
 
@@ -4585,32 +4596,116 @@ function acceptsReasoningEffort(model2) {
4585
4596
  const id = baseModelId(model2);
4586
4597
  return isReasoningModel(model2) && !/^o1-(mini|preview)/.test(id) && !/-chat(-|$)/.test(id);
4587
4598
  }
4599
+ function acceptsNoReasoningEffort(model2) {
4600
+ if (!acceptsReasoningEffort(model2))
4601
+ return false;
4602
+ const id = baseModelId(model2);
4603
+ if (/-pro\b/.test(id))
4604
+ return false;
4605
+ const gpt = /^gpt-(\d+)(?:\.(\d+))?/.exec(id);
4606
+ if (!gpt)
4607
+ return false;
4608
+ const major = Number(gpt[1]);
4609
+ const minor = gpt[2] !== void 0 ? Number(gpt[2]) : 0;
4610
+ return major > 5 || major === 5 && minor >= 1;
4611
+ }
4612
+ function forgetLearnedReasoningEfforts() {
4613
+ learnedEfforts.clear();
4614
+ }
4615
+ function chatReasoningEffort(model2, opts) {
4616
+ const learned = learnedEfforts.get(learnedKey(model2, opts.tools === true));
4617
+ if (learned !== void 0)
4618
+ return learned === "omit" ? void 0 : learned;
4619
+ if (!acceptsReasoningEffort(model2))
4620
+ return void 0;
4621
+ if (opts.tools === true && acceptsNoReasoningEffort(model2))
4622
+ return "none";
4623
+ return opts.requested;
4624
+ }
4588
4625
  function chatCompletionParams(model2, input) {
4626
+ const effort = chatReasoningEffort(model2, {
4627
+ ...input.tools !== void 0 ? { tools: input.tools } : {},
4628
+ ...input.reasoningEffort !== void 0 ? { requested: input.reasoningEffort } : {}
4629
+ });
4589
4630
  if (isReasoningModel(model2)) {
4590
4631
  return {
4591
4632
  ...input.maxTokens !== void 0 ? { max_completion_tokens: Math.max(input.maxTokens, REASONING_MIN_COMPLETION_TOKENS) } : {},
4592
- ...input.reasoningEffort !== void 0 && acceptsReasoningEffort(model2) ? { reasoning_effort: input.reasoningEffort } : {}
4633
+ ...effort !== void 0 ? { reasoning_effort: effort } : {}
4593
4634
  };
4594
4635
  }
4595
4636
  return {
4637
+ // only when a provider taught the process that this model (one we don't know as a reasoning model) takes an effort
4638
+ ...effort !== void 0 ? { reasoning_effort: effort } : {},
4596
4639
  ...input.maxTokens !== void 0 ? { max_tokens: input.maxTokens } : {},
4597
4640
  ...input.temperature !== void 0 ? { temperature: input.temperature } : {},
4598
4641
  ...input.topP !== void 0 ? { top_p: input.topP } : {}
4599
4642
  };
4600
4643
  }
4601
- function modelRejectionOf(err) {
4644
+ function acceptsSamplingParams(model2) {
4645
+ return !isReasoningModel(model2);
4646
+ }
4647
+ function providerErrorOf(err) {
4648
+ if (err === null || typeof err !== "object")
4649
+ return null;
4602
4650
  const e = err;
4603
- const status = typeof e?.status === "number" ? e.status : null;
4604
- if (status !== 400 && status !== 403 && status !== 404)
4651
+ const body = e["body"] !== null && typeof e["body"] === "object" ? e["body"] : {};
4652
+ const pick = (k) => e[k] !== void 0 && e[k] !== null ? e[k] : body[k];
4653
+ const statusRaw = typeof e["status"] === "number" ? e["status"] : e["statusCode"];
4654
+ const status = typeof statusRaw === "number" && statusRaw > 0 ? statusRaw : null;
4655
+ const code = pick("code");
4656
+ const type = pick("type");
4657
+ const param = pick("param");
4658
+ const message = typeof body["message"] === "string" ? body["message"] : typeof e["message"] === "string" ? e["message"] : "";
4659
+ return {
4660
+ status,
4661
+ code: typeof code === "string" ? code : typeof type === "string" ? type : "",
4662
+ param: typeof param === "string" ? param : null,
4663
+ message
4664
+ };
4665
+ }
4666
+ function reasoningEffortRejectionOf(err) {
4667
+ const f = providerErrorOf(err);
4668
+ if (!f || f.status !== 400)
4669
+ return null;
4670
+ if (f.param !== "reasoning_effort" && !/reasoning[_ ]effort/i.test(f.message))
4671
+ return null;
4672
+ return { status: f.status, message: f.message, suggestsNone: /reasoning[_ ]effort[^.]*\bnone\b/i.test(f.message) };
4673
+ }
4674
+ function learnReasoningEffort(model2, shape, err) {
4675
+ const rejection = reasoningEffortRejectionOf(err);
4676
+ if (!rejection)
4677
+ return false;
4678
+ const next = shape.sent !== "none" && (rejection.suggestsNone || shape.sent === void 0) ? "none" : "omit";
4679
+ if ((next === "omit" ? void 0 : next) === shape.sent)
4680
+ return false;
4681
+ learnedEfforts.set(learnedKey(model2, shape.tools), next);
4682
+ return true;
4683
+ }
4684
+ async function withReasoningEffortRetry(shape, call) {
4685
+ const sent = chatReasoningEffort(shape.model, {
4686
+ tools: shape.tools,
4687
+ ...shape.requested !== void 0 ? { requested: shape.requested } : {}
4688
+ });
4689
+ try {
4690
+ return await call();
4691
+ } catch (err) {
4692
+ if (!learnReasoningEffort(shape.model, { tools: shape.tools, sent }, err))
4693
+ throw err;
4694
+ return call();
4695
+ }
4696
+ }
4697
+ function modelRejectionOf(err) {
4698
+ const f = providerErrorOf(err);
4699
+ const status = f?.status ?? null;
4700
+ if (!f || status === null || status !== 400 && status !== 403 && status !== 404)
4605
4701
  return null;
4606
- const code = typeof e?.code === "string" ? e.code : typeof e?.type === "string" ? e.type : "";
4607
- const param = typeof e?.param === "string" ? e.param : null;
4608
- const rejected = REJECTION_CODES.has(code) || param !== null && MODEL_PARAMS.has(param);
4702
+ const { code, param } = f;
4703
+ const rejected = REJECTION_CODES.has(code) || param !== null && MODEL_PARAMS.has(param) || reasoningEffortRejectionOf(err) !== null;
4609
4704
  if (!rejected)
4610
4705
  return null;
4611
4706
  if (status === 403 && code !== "model_not_found")
4612
4707
  return null;
4613
- return { status, code: code || "invalid_request_error", message: typeof e?.message === "string" ? e.message : "" };
4708
+ return { status, code: code || "invalid_request_error", message: f.message };
4614
4709
  }
4615
4710
  async function withModelFallback(opts) {
4616
4711
  const fallbackModel = opts.fallbackModel ?? PLATFORM_DEFAULT_CHAT_MODEL;
@@ -4631,11 +4726,13 @@ async function withModelFallback(opts) {
4631
4726
  return opts.call(fallbackModel);
4632
4727
  }
4633
4728
  }
4634
- var PLATFORM_DEFAULT_CHAT_MODEL, REASONING_MIN_COMPLETION_TOKENS, REJECTION_CODES, MODEL_PARAMS, MODEL_REJECTION_TTL_MS, ModelRejectionCache;
4729
+ var PLATFORM_DEFAULT_CHAT_MODEL, REASONING_MIN_COMPLETION_TOKENS, learnedEfforts, learnedKey, REJECTION_CODES, MODEL_PARAMS, MODEL_REJECTION_TTL_MS, ModelRejectionCache;
4635
4730
  var init_chat_params = __esm({
4636
4731
  "../llm-client/dist/chat-params.js"() {
4637
4732
  PLATFORM_DEFAULT_CHAT_MODEL = "gpt-4o-mini";
4638
4733
  REASONING_MIN_COMPLETION_TOKENS = 2048;
4734
+ learnedEfforts = /* @__PURE__ */ new Map();
4735
+ learnedKey = (model2, tools) => `${baseModelId(model2)}|${tools ? "tools" : "plain"}`;
4639
4736
  REJECTION_CODES = /* @__PURE__ */ new Set(["unsupported_parameter", "unsupported_value", "model_not_found"]);
4640
4737
  MODEL_PARAMS = /* @__PURE__ */ new Set(["model", "max_tokens", "max_completion_tokens", "temperature", "top_p", "reasoning_effort"]);
4641
4738
  MODEL_REJECTION_TTL_MS = 5 * 60 * 1e3;
@@ -4663,6 +4760,30 @@ var init_chat_params = __esm({
4663
4760
  };
4664
4761
  }
4665
4762
  });
4763
+
4764
+ // ../llm-client/dist/index.js
4765
+ var dist_exports = {};
4766
+ __export(dist_exports, {
4767
+ MODEL_REJECTION_TTL_MS: () => MODEL_REJECTION_TTL_MS,
4768
+ ModelRejectionCache: () => ModelRejectionCache,
4769
+ PLATFORM_DEFAULT_CHAT_MODEL: () => PLATFORM_DEFAULT_CHAT_MODEL,
4770
+ REASONING_MIN_COMPLETION_TOKENS: () => REASONING_MIN_COMPLETION_TOKENS,
4771
+ acceptsNoReasoningEffort: () => acceptsNoReasoningEffort,
4772
+ acceptsReasoningEffort: () => acceptsReasoningEffort,
4773
+ acceptsSamplingParams: () => acceptsSamplingParams,
4774
+ chatCompletionParams: () => chatCompletionParams,
4775
+ chatReasoningEffort: () => chatReasoningEffort,
4776
+ createChatClient: () => createChatClient,
4777
+ forgetLearnedReasoningEfforts: () => forgetLearnedReasoningEfforts,
4778
+ hasChatKey: () => hasChatKey,
4779
+ isReasoningModel: () => isReasoningModel,
4780
+ learnReasoningEffort: () => learnReasoningEffort,
4781
+ modelRejectionOf: () => modelRejectionOf,
4782
+ providerErrorOf: () => providerErrorOf,
4783
+ reasoningEffortRejectionOf: () => reasoningEffortRejectionOf,
4784
+ withModelFallback: () => withModelFallback,
4785
+ withReasoningEffortRetry: () => withReasoningEffortRetry
4786
+ });
4666
4787
  function createChatClient(config = {}) {
4667
4788
  return new OpenAI({
4668
4789
  ...config.apiKey ? { apiKey: config.apiKey } : {},
@@ -4670,6 +4791,9 @@ function createChatClient(config = {}) {
4670
4791
  ...config.timeoutMs ? { timeout: config.timeoutMs } : {}
4671
4792
  });
4672
4793
  }
4794
+ function hasChatKey(config = {}) {
4795
+ return Boolean(config.apiKey || process.env["OPENAI_API_KEY"]);
4796
+ }
4673
4797
  var init_dist = __esm({
4674
4798
  "../llm-client/dist/index.js"() {
4675
4799
  init_chat_params();
@@ -5128,505 +5252,206 @@ var init_wrap = __esm({
5128
5252
  "src/providers/wrap.ts"() {
5129
5253
  }
5130
5254
  });
5131
-
5132
- // src/providers/index.ts
5133
- async function importOptional(spec, hint) {
5134
- try {
5135
- return await import(spec);
5136
- } catch {
5137
- throw new Error(`${hint}: "${spec}" is not installed. Add it with: pnpm add ${spec}`);
5255
+ function metricExportIntervalMs() {
5256
+ const raw = Number(process.env.OTEL_METRIC_EXPORT_INTERVAL);
5257
+ return Number.isFinite(raw) && raw > 0 ? raw : DEFAULT_METRIC_EXPORT_INTERVAL_MS;
5258
+ }
5259
+ function start(opts) {
5260
+ if (started)
5261
+ return;
5262
+ started = true;
5263
+ if (opts.debug) {
5264
+ diag.setLogger(new DiagConsoleLogger(), DiagLogLevel.INFO);
5265
+ }
5266
+ const authKey = process.env.VOICELAYER_TELEMETRY_KEY || process.env.VOICELAYER_API_KEY;
5267
+ if (!authKey) {
5268
+ return;
5269
+ }
5270
+ if (!process.env.OTEL_EXPORTER_OTLP_ENDPOINT) {
5271
+ process.env.OTEL_EXPORTER_OTLP_ENDPOINT = DEFAULT_OTLP_ENDPOINT;
5272
+ }
5273
+ process.env.OTEL_EXPORTER_OTLP_HEADERS = `Authorization=Bearer ${authKey}`;
5274
+ const resource = resourceFromAttributes({
5275
+ [ATTR_SERVICE_NAME]: opts.service,
5276
+ ...opts.version ? { [ATTR_SERVICE_VERSION]: opts.version } : {}
5277
+ });
5278
+ loggerProvider = new LoggerProvider({
5279
+ resource,
5280
+ processors: [new BatchLogRecordProcessor(new OTLPLogExporter())]
5281
+ });
5282
+ logs.setGlobalLoggerProvider(loggerProvider);
5283
+ metricReader = new PeriodicExportingMetricReader({
5284
+ exportIntervalMillis: metricExportIntervalMs(),
5285
+ // DELTA temporality: each export carries only the increment since the
5286
+ // previous one. The read side (apps/api openobserve-repository) totals a
5287
+ // metric with SUM(value) over the query window, which is correct only for
5288
+ // deltas. The OTLP exporter defaults to CUMULATIVE — it re-reports the
5289
+ // running total every export, so SUM double-counts every call that lives
5290
+ // longer than one export interval (inflating LLM/TTS/STT usage + cost).
5291
+ // Setting DELTA makes the existing SUM reads correct without touching them.
5292
+ exporter: new OTLPMetricExporter({
5293
+ temporalityPreference: AggregationTemporalityPreference.DELTA
5294
+ })
5295
+ });
5296
+ sdk = new NodeSDK({
5297
+ resource,
5298
+ traceExporter: new OTLPTraceExporter(),
5299
+ metricReader,
5300
+ instrumentations: [
5301
+ getNodeAutoInstrumentations({
5302
+ // fs instrumentation is extremely noisy and rarely useful for app traces.
5303
+ "@opentelemetry/instrumentation-fs": { enabled: false }
5304
+ })
5305
+ ]
5306
+ });
5307
+ sdk.start();
5308
+ const shutdown = async () => {
5309
+ try {
5310
+ await sdk?.shutdown();
5311
+ await loggerProvider?.shutdown();
5312
+ } catch {
5313
+ }
5314
+ };
5315
+ process.once("SIGTERM", () => void shutdown().then(() => process.exit(0)));
5316
+ process.once("SIGINT", () => void shutdown().then(() => process.exit(0)));
5317
+ }
5318
+ async function flushTelemetry() {
5319
+ const flushes = [];
5320
+ if (metricReader)
5321
+ flushes.push(metricReader.forceFlush());
5322
+ if (loggerProvider)
5323
+ flushes.push(loggerProvider.forceFlush());
5324
+ const proxied = trace.getTracerProvider();
5325
+ const tracerProvider = proxied.getDelegate?.() ?? proxied;
5326
+ if (typeof tracerProvider.forceFlush === "function") {
5327
+ flushes.push(tracerProvider.forceFlush());
5138
5328
  }
5329
+ await Promise.allSettled(flushes);
5139
5330
  }
5140
- function withCreds(opts, creds2) {
5141
- if (creds2?.apiKey) opts["apiKey"] = creds2.apiKey;
5142
- if (creds2?.baseURL) opts["baseURL"] = creds2.baseURL;
5143
- return opts;
5331
+ function getLogger(name) {
5332
+ return logs.getLogger(name);
5144
5333
  }
5145
- var deepgram, openai, anthropic, cartesia, elevenlabs, assemblyai, google, silero, livekitTurn, connector;
5146
- var init_providers2 = __esm({
5147
- "src/providers/index.ts"() {
5148
- init_ssrf();
5149
- init_transport_callback();
5150
- init_connector_llm();
5151
- init_wrap();
5152
- deepgram = {
5153
- stt(options = {}) {
5154
- return async () => {
5155
- const dg = await importOptional("@voicelayer/agents-plugin-deepgram", "deepgram.stt");
5156
- return new dg.STT(
5157
- withCreds(
5158
- { model: options.model ?? "nova-3", language: options.language ?? "en-US" },
5159
- options
5160
- )
5161
- );
5162
- };
5163
- },
5164
- tts(options = {}) {
5165
- return async () => {
5166
- const dg = await importOptional("@voicelayer/agents-plugin-deepgram", "deepgram.tts");
5167
- return new dg.TTS(
5168
- withCreds({ model: options.model ?? "aura-2-harmonia-en" }, options)
5169
- );
5170
- };
5171
- }
5172
- };
5173
- openai = {
5174
- llm(options = {}) {
5175
- return async () => {
5176
- const oa = await importOptional("@voicelayer/agents-plugin-openai", "openai.llm");
5177
- return new oa.LLM(
5178
- withCreds({ model: options.model ?? "gpt-4o-mini" }, options)
5179
- );
5180
- };
5181
- },
5182
- tts(options = {}) {
5183
- return async () => {
5184
- const oa = await importOptional("@voicelayer/agents-plugin-openai", "openai.tts");
5185
- const opts = { model: options.model ?? "tts-1" };
5186
- if (options.voice) opts["voice"] = options.voice;
5187
- if (options.instructions) opts["instructions"] = options.instructions;
5188
- return new oa.TTS(withCreds(opts, options));
5189
- };
5190
- },
5191
- // Speech-to-speech via the OpenAI Realtime API. Returns a RealtimeProvider
5192
- // that the SDK uses in place of the stt/llm/tts pipeline.
5193
- realtime(options = {}) {
5194
- return async () => {
5195
- const oa = await importOptional("@voicelayer/agents-plugin-openai", "openai.realtime");
5196
- const opts = {};
5197
- if (options.model) opts["model"] = options.model;
5198
- if (options.voice) opts["voice"] = options.voice;
5199
- return new oa.realtime.RealtimeModel(withCreds(opts, options));
5200
- };
5201
- }
5202
- };
5203
- anthropic = {
5204
- llm(options = {}) {
5205
- return async () => {
5206
- const an = await importOptional("@livekit/agents-plugin-anthropic", "anthropic.llm");
5207
- return new an.LLM(
5208
- withCreds({ model: options.model ?? "claude-sonnet-5-5" }, options)
5209
- );
5210
- };
5211
- }
5212
- };
5213
- cartesia = {
5214
- tts(options = {}) {
5215
- return async () => {
5216
- const ct = await importOptional("@voicelayer/agents-plugin-cartesia", "cartesia.tts");
5217
- return new ct.TTS(options);
5218
- };
5219
- },
5220
- stt(options = {}) {
5221
- return async () => {
5222
- const ct = await importOptional("@voicelayer/agents-plugin-cartesia", "cartesia.stt");
5223
- const opts = {};
5224
- if (options.model) opts["model"] = options.model;
5225
- if (options.language) opts["language"] = options.language;
5226
- return new ct.STT(withCreds(opts, options));
5227
- };
5228
- }
5229
- };
5230
- elevenlabs = {
5231
- tts(options = {}) {
5232
- return async () => {
5233
- const el = await importOptional("@voicelayer/agents-plugin-elevenlabs", "elevenlabs.tts");
5234
- const opts = {};
5235
- if (options.voice) opts["voiceId"] = options.voice;
5236
- if (options.model) opts["model"] = options.model;
5237
- return new el.TTS(withCreds(opts, options));
5238
- };
5239
- }
5240
- };
5241
- assemblyai = {
5242
- stt(options = {}) {
5243
- return async () => {
5244
- const aai = await importOptional("@voicelayer/agents-plugin-assemblyai", "assemblyai.stt");
5245
- const opts = {};
5246
- if (options.language) opts["language"] = options.language;
5247
- return new aai.STT(withCreds(opts, options));
5248
- };
5249
- }
5250
- };
5251
- google = {
5252
- llm(options = {}) {
5253
- return async () => {
5254
- const g = await importOptional("@voicelayer/agents-plugin-google", "google.llm");
5255
- return new g.LLM(
5256
- withCreds({ model: options.model ?? "gemini-3.8-flash" }, options)
5257
- );
5258
- };
5259
- },
5260
- realtime(options = {}) {
5261
- return async () => {
5262
- const g = await importOptional("@voicelayer/agents-plugin-google", "google.realtime");
5263
- const opts = {};
5264
- if (options.model) opts["model"] = options.model;
5265
- if (options.voice) opts["voice"] = options.voice;
5266
- return new g.beta.realtime.RealtimeModel(withCreds(opts, options));
5267
- };
5268
- }
5269
- };
5270
- silero = {
5271
- vad() {
5272
- return async () => {
5273
- const sil = await importOptional("@voicelayer/agents-plugin-silero", "silero.vad");
5274
- return await sil.VAD.load();
5275
- };
5276
- }
5277
- };
5278
- livekitTurn = {
5279
- english() {
5280
- return async () => {
5281
- const lk = await importOptional("@voicelayer/agents-plugin-livekit", "livekitTurn.english");
5282
- return new lk.turnDetector.EnglishModel();
5283
- };
5284
- },
5285
- multilingual() {
5286
- return async () => {
5287
- const lk = await importOptional("@voicelayer/agents-plugin-livekit", "livekitTurn.multilingual");
5288
- return new lk.turnDetector.MultilingualModel();
5289
- };
5290
- }
5291
- };
5292
- connector = {
5293
- llm(config = {}) {
5294
- return async (call) => {
5295
- if (config.url) {
5296
- await assertPublicHttpsUrl(
5297
- config.url,
5298
- config.allowHosts ? { allowHosts: config.allowHosts } : {}
5299
- );
5300
- return openai.llm({
5301
- ...config.model ? { model: config.model } : {},
5302
- ...config.apiKey ? { apiKey: config.apiKey } : {},
5303
- baseURL: config.url
5304
- })(call);
5305
- }
5306
- const transport = config.transport ?? (config.onQuery ? callbackTransport(config.onQuery) : void 0);
5307
- if (!transport) {
5308
- throw new Error("connector.llm requires one of: url, onQuery, or transport");
5309
- }
5310
- const projectId = typeof call.metadata["projectId"] === "string" ? call.metadata["projectId"] : void 0;
5311
- const llmOptions = {
5312
- ...config.model ? { model: config.model } : {},
5313
- ...config.temperature !== void 0 ? { temperature: config.temperature } : {},
5314
- ...config.fallbackText ? { fallbackText: config.fallbackText } : {},
5315
- ...call.callId ? { callId: call.callId } : {},
5316
- ...projectId ? { projectId } : {}
5317
- };
5318
- return await createConnectorLLM(transport, llmOptions);
5319
- };
5320
- }
5334
+ var DEFAULT_OTLP_ENDPOINT, DEFAULT_METRIC_EXPORT_INTERVAL_MS, started, sdk, loggerProvider, metricReader;
5335
+ var init_start = __esm({
5336
+ "../observability/dist/start.js"() {
5337
+ DEFAULT_OTLP_ENDPOINT = "https://otel.vlayers.ai/v1/otlp";
5338
+ DEFAULT_METRIC_EXPORT_INTERVAL_MS = 15e3;
5339
+ started = false;
5340
+ sdk = null;
5341
+ loggerProvider = null;
5342
+ metricReader = null;
5343
+ }
5344
+ });
5345
+
5346
+ // ../observability/dist/attributes.js
5347
+ function callContextToAttributes(ctx) {
5348
+ const out = {};
5349
+ if (ctx.projectId)
5350
+ out[ATTR.projectId] = ctx.projectId;
5351
+ if (ctx.callId)
5352
+ out[ATTR.callId] = ctx.callId;
5353
+ if (ctx.campaignId)
5354
+ out[ATTR.campaignId] = ctx.campaignId;
5355
+ if (ctx.room)
5356
+ out[ATTR.room] = ctx.room;
5357
+ if (ctx.agentId)
5358
+ out[ATTR.agentId] = ctx.agentId;
5359
+ if (ctx.phoneNumberId)
5360
+ out[ATTR.phoneNumberId] = ctx.phoneNumberId;
5361
+ if (ctx.bindingId)
5362
+ out[ATTR.bindingId] = ctx.bindingId;
5363
+ return out;
5364
+ }
5365
+ var ATTR;
5366
+ var init_attributes = __esm({
5367
+ "../observability/dist/attributes.js"() {
5368
+ ATTR = {
5369
+ projectId: "vl.project_id",
5370
+ callId: "vl.call_id",
5371
+ campaignId: "vl.campaign_id",
5372
+ room: "vl.room",
5373
+ agentId: "vl.agent_id",
5374
+ phoneNumberId: "vl.phone_number_id",
5375
+ bindingId: "vl.binding_id",
5376
+ source: "vl.source",
5377
+ kind: "vl.kind"
5321
5378
  };
5322
5379
  }
5323
5380
  });
5324
-
5325
- // src/providers/registry.ts
5326
- function entry(id, bits) {
5327
- const identity = providerIdentity(id);
5328
- if (!identity) throw new Error(`PROVIDER_REGISTRY: no provider identity in contracts for '${id}'`);
5329
- return { name: identity.id, capabilities: identity.capabilities, ...bits };
5381
+ async function withCallContext(opts, fn) {
5382
+ const tracer = trace.getTracer(TRACER_NAME);
5383
+ const ctx = {
5384
+ ...opts.projectId ? { projectId: opts.projectId } : {},
5385
+ ...opts.callId ? { callId: opts.callId } : {},
5386
+ ...opts.room ? { room: opts.room } : {},
5387
+ ...opts.agentId ? { agentId: opts.agentId } : {},
5388
+ ...opts.phoneNumberId ? { phoneNumberId: opts.phoneNumberId } : {},
5389
+ ...opts.bindingId ? { bindingId: opts.bindingId } : {}
5390
+ };
5391
+ return callContextStore.run(ctx, () => tracer.startActiveSpan(opts.kind, async (span) => {
5392
+ span.setAttributes(callContextToAttributes(opts));
5393
+ span.setAttribute(ATTR.kind, opts.kind);
5394
+ if (opts.source)
5395
+ span.setAttribute(ATTR.source, opts.source);
5396
+ if (opts.attributes)
5397
+ span.setAttributes(opts.attributes);
5398
+ try {
5399
+ return await fn(span);
5400
+ } catch (err) {
5401
+ span.recordException(err);
5402
+ span.setStatus({ code: SpanStatusCode.ERROR, message: err.message });
5403
+ throw err;
5404
+ } finally {
5405
+ span.end();
5406
+ }
5407
+ }));
5330
5408
  }
5331
- function providerEntry(name) {
5332
- if (!name) return void 0;
5333
- return PROVIDER_REGISTRY[PROVIDER_ALIASES[name] ?? name];
5409
+ function getCurrentCallContext() {
5410
+ return callContextStore.getStore();
5334
5411
  }
5335
- function resolveStt(name, o) {
5336
- return (providerEntry(name)?.stt ?? PROVIDER_REGISTRY["deepgram"].stt)(o);
5412
+ var TRACER_NAME, callContextStore;
5413
+ var init_call_context = __esm({
5414
+ "../observability/dist/call-context.js"() {
5415
+ init_attributes();
5416
+ TRACER_NAME = "@voicelayer/observability";
5417
+ callContextStore = new AsyncLocalStorage();
5418
+ }
5419
+ });
5420
+ var init_trace_propagation = __esm({
5421
+ "../observability/dist/trace-propagation.js"() {
5422
+ }
5423
+ });
5424
+ function meter() {
5425
+ return metrics.getMeter(METER_NAME, METER_VERSION);
5337
5426
  }
5338
- function resolveTts(name, o) {
5339
- return (providerEntry(name)?.tts ?? PROVIDER_REGISTRY["deepgram"].tts)(o);
5427
+ function attrs(a) {
5428
+ const out = {
5429
+ [ATTR.projectId]: a.projectId,
5430
+ [ATTR.callId]: a.callId,
5431
+ "vl.provider": a.provider
5432
+ };
5433
+ if (a.campaignId)
5434
+ out[ATTR.campaignId] = a.campaignId;
5435
+ if (a.model)
5436
+ out["vl.model"] = a.model;
5437
+ return out;
5340
5438
  }
5341
- function resolveLlm(name, o) {
5342
- return (providerEntry(name)?.llm ?? PROVIDER_REGISTRY["openai"].llm)(withCurrentModel(name, o));
5439
+ function llmInputTokens() {
5440
+ return _llmInputTokens ??= meter().createCounter(USAGE_METRIC_NAMES.llmInputTokens, {
5441
+ description: "Input tokens consumed by an LLM call",
5442
+ unit: "{tokens}"
5443
+ });
5343
5444
  }
5344
- function resolveRealtime(name, o) {
5345
- return (providerEntry(name)?.realtime ?? PROVIDER_REGISTRY["openai"].realtime)(withCurrentModel(name, o));
5445
+ function llmOutputTokens() {
5446
+ return _llmOutputTokens ??= meter().createCounter(USAGE_METRIC_NAMES.llmOutputTokens, {
5447
+ description: "Output tokens produced by an LLM call",
5448
+ unit: "{tokens}"
5449
+ });
5346
5450
  }
5347
- function withCurrentModel(name, o) {
5348
- if (!name || !o.model) return o;
5349
- const current = currentModelFor(name, o.model);
5350
- if (!current.retired) return o;
5351
- console.warn("[agent] the configured model is retired; running its replacement", {
5352
- provider: name,
5353
- model: current.retired,
5354
- replacement: current.model
5355
- });
5356
- return { ...o, model: current.model };
5357
- }
5358
- var model, lang, voice, creds, PROVIDER_REGISTRY;
5359
- var init_registry = __esm({
5360
- "src/providers/registry.ts"() {
5361
- init_providers2();
5362
- init_src();
5363
- model = (o) => o.model ? { model: o.model } : {};
5364
- lang = (o) => o.language ? { language: o.language } : {};
5365
- voice = (o) => o.voice ? { voice: o.voice } : {};
5366
- creds = (o) => o.creds ?? {};
5367
- PROVIDER_REGISTRY = {
5368
- deepgram: entry("deepgram", {
5369
- failureClass: "vendor-api",
5370
- engines: ["livekit"],
5371
- stt: (o) => deepgram.stt({ ...model(o), ...lang(o), ...creds(o) }),
5372
- // Deepgram folds the voice into the model id (aura-2-<name>-en): wire config
5373
- // passes a voice, code config passes a model — both land as `model`.
5374
- tts: (o) => {
5375
- const m = o.voice ?? o.model;
5376
- return deepgram.tts({ ...m ? { model: m } : {}, ...creds(o) });
5377
- }
5378
- }),
5379
- openai: entry("openai", {
5380
- failureClass: "vendor-api",
5381
- engines: ["livekit", "text"],
5382
- llm: (o) => openai.llm({ ...model(o), ...creds(o) }),
5383
- tts: (o) => openai.tts({ ...voice(o), ...model(o), ...creds(o) }),
5384
- realtime: (o) => openai.realtime({ ...model(o), ...voice(o), ...creds(o) })
5385
- }),
5386
- anthropic: entry("anthropic", {
5387
- failureClass: "vendor-api",
5388
- engines: ["livekit", "text"],
5389
- llm: (o) => anthropic.llm({ ...model(o), ...creds(o) })
5390
- }),
5391
- google: entry("google", {
5392
- failureClass: "vendor-api",
5393
- engines: ["livekit", "text"],
5394
- llm: (o) => google.llm({ ...model(o), ...creds(o) }),
5395
- realtime: (o) => google.realtime({ ...model(o), ...voice(o), ...creds(o) })
5396
- }),
5397
- cartesia: entry("cartesia", {
5398
- failureClass: "vendor-api",
5399
- engines: ["livekit"],
5400
- stt: (o) => cartesia.stt({ ...model(o), ...lang(o), ...creds(o) }),
5401
- tts: (o) => cartesia.tts({ ...voice(o), ...model(o), ...creds(o) })
5402
- }),
5403
- elevenlabs: entry("elevenlabs", {
5404
- failureClass: "vendor-api",
5405
- engines: ["livekit"],
5406
- tts: (o) => elevenlabs.tts({ ...voice(o), ...model(o), ...creds(o) })
5407
- }),
5408
- assemblyai: entry("assemblyai", {
5409
- failureClass: "vendor-api",
5410
- engines: ["livekit"],
5411
- stt: (o) => assemblyai.stt({ ...lang(o), ...creds(o) })
5412
- // no model knob
5413
- }),
5414
- silero: entry("silero", {
5415
- failureClass: "local",
5416
- engines: ["livekit"],
5417
- vad: () => silero.vad()
5418
- })
5419
- };
5420
- }
5421
- });
5422
-
5423
- // src/providers/llm.ts
5424
- var init_llm = __esm({
5425
- "src/providers/llm.ts"() {
5426
- init_openai_default();
5427
- init_registry();
5428
- }
5429
- });
5430
- function metricExportIntervalMs() {
5431
- const raw = Number(process.env.OTEL_METRIC_EXPORT_INTERVAL);
5432
- return Number.isFinite(raw) && raw > 0 ? raw : DEFAULT_METRIC_EXPORT_INTERVAL_MS;
5433
- }
5434
- function start(opts) {
5435
- if (started)
5436
- return;
5437
- started = true;
5438
- if (opts.debug) {
5439
- diag.setLogger(new DiagConsoleLogger(), DiagLogLevel.INFO);
5440
- }
5441
- const authKey = process.env.VOICELAYER_TELEMETRY_KEY || process.env.VOICELAYER_API_KEY;
5442
- if (!authKey) {
5443
- return;
5444
- }
5445
- if (!process.env.OTEL_EXPORTER_OTLP_ENDPOINT) {
5446
- process.env.OTEL_EXPORTER_OTLP_ENDPOINT = DEFAULT_OTLP_ENDPOINT;
5447
- }
5448
- process.env.OTEL_EXPORTER_OTLP_HEADERS = `Authorization=Bearer ${authKey}`;
5449
- const resource = resourceFromAttributes({
5450
- [ATTR_SERVICE_NAME]: opts.service,
5451
- ...opts.version ? { [ATTR_SERVICE_VERSION]: opts.version } : {}
5452
- });
5453
- loggerProvider = new LoggerProvider({
5454
- resource,
5455
- processors: [new BatchLogRecordProcessor(new OTLPLogExporter())]
5456
- });
5457
- logs.setGlobalLoggerProvider(loggerProvider);
5458
- metricReader = new PeriodicExportingMetricReader({
5459
- exportIntervalMillis: metricExportIntervalMs(),
5460
- // DELTA temporality: each export carries only the increment since the
5461
- // previous one. The read side (apps/api openobserve-repository) totals a
5462
- // metric with SUM(value) over the query window, which is correct only for
5463
- // deltas. The OTLP exporter defaults to CUMULATIVE — it re-reports the
5464
- // running total every export, so SUM double-counts every call that lives
5465
- // longer than one export interval (inflating LLM/TTS/STT usage + cost).
5466
- // Setting DELTA makes the existing SUM reads correct without touching them.
5467
- exporter: new OTLPMetricExporter({
5468
- temporalityPreference: AggregationTemporalityPreference.DELTA
5469
- })
5470
- });
5471
- sdk = new NodeSDK({
5472
- resource,
5473
- traceExporter: new OTLPTraceExporter(),
5474
- metricReader,
5475
- instrumentations: [
5476
- getNodeAutoInstrumentations({
5477
- // fs instrumentation is extremely noisy and rarely useful for app traces.
5478
- "@opentelemetry/instrumentation-fs": { enabled: false }
5479
- })
5480
- ]
5481
- });
5482
- sdk.start();
5483
- const shutdown = async () => {
5484
- try {
5485
- await sdk?.shutdown();
5486
- await loggerProvider?.shutdown();
5487
- } catch {
5488
- }
5489
- };
5490
- process.once("SIGTERM", () => void shutdown().then(() => process.exit(0)));
5491
- process.once("SIGINT", () => void shutdown().then(() => process.exit(0)));
5492
- }
5493
- async function flushTelemetry() {
5494
- const flushes = [];
5495
- if (metricReader)
5496
- flushes.push(metricReader.forceFlush());
5497
- if (loggerProvider)
5498
- flushes.push(loggerProvider.forceFlush());
5499
- const proxied = trace.getTracerProvider();
5500
- const tracerProvider = proxied.getDelegate?.() ?? proxied;
5501
- if (typeof tracerProvider.forceFlush === "function") {
5502
- flushes.push(tracerProvider.forceFlush());
5503
- }
5504
- await Promise.allSettled(flushes);
5505
- }
5506
- function getLogger(name) {
5507
- return logs.getLogger(name);
5508
- }
5509
- var DEFAULT_OTLP_ENDPOINT, DEFAULT_METRIC_EXPORT_INTERVAL_MS, started, sdk, loggerProvider, metricReader;
5510
- var init_start = __esm({
5511
- "../observability/dist/start.js"() {
5512
- DEFAULT_OTLP_ENDPOINT = "https://otel.vlayers.ai/v1/otlp";
5513
- DEFAULT_METRIC_EXPORT_INTERVAL_MS = 15e3;
5514
- started = false;
5515
- sdk = null;
5516
- loggerProvider = null;
5517
- metricReader = null;
5518
- }
5519
- });
5520
-
5521
- // ../observability/dist/attributes.js
5522
- function callContextToAttributes(ctx) {
5523
- const out = {};
5524
- if (ctx.projectId)
5525
- out[ATTR.projectId] = ctx.projectId;
5526
- if (ctx.callId)
5527
- out[ATTR.callId] = ctx.callId;
5528
- if (ctx.campaignId)
5529
- out[ATTR.campaignId] = ctx.campaignId;
5530
- if (ctx.room)
5531
- out[ATTR.room] = ctx.room;
5532
- if (ctx.agentId)
5533
- out[ATTR.agentId] = ctx.agentId;
5534
- if (ctx.phoneNumberId)
5535
- out[ATTR.phoneNumberId] = ctx.phoneNumberId;
5536
- if (ctx.bindingId)
5537
- out[ATTR.bindingId] = ctx.bindingId;
5538
- return out;
5539
- }
5540
- var ATTR;
5541
- var init_attributes = __esm({
5542
- "../observability/dist/attributes.js"() {
5543
- ATTR = {
5544
- projectId: "vl.project_id",
5545
- callId: "vl.call_id",
5546
- campaignId: "vl.campaign_id",
5547
- room: "vl.room",
5548
- agentId: "vl.agent_id",
5549
- phoneNumberId: "vl.phone_number_id",
5550
- bindingId: "vl.binding_id",
5551
- source: "vl.source",
5552
- kind: "vl.kind"
5553
- };
5554
- }
5555
- });
5556
- async function withCallContext(opts, fn) {
5557
- const tracer = trace.getTracer(TRACER_NAME);
5558
- const ctx = {
5559
- ...opts.projectId ? { projectId: opts.projectId } : {},
5560
- ...opts.callId ? { callId: opts.callId } : {},
5561
- ...opts.room ? { room: opts.room } : {},
5562
- ...opts.agentId ? { agentId: opts.agentId } : {},
5563
- ...opts.phoneNumberId ? { phoneNumberId: opts.phoneNumberId } : {},
5564
- ...opts.bindingId ? { bindingId: opts.bindingId } : {}
5565
- };
5566
- return callContextStore.run(ctx, () => tracer.startActiveSpan(opts.kind, async (span) => {
5567
- span.setAttributes(callContextToAttributes(opts));
5568
- span.setAttribute(ATTR.kind, opts.kind);
5569
- if (opts.source)
5570
- span.setAttribute(ATTR.source, opts.source);
5571
- if (opts.attributes)
5572
- span.setAttributes(opts.attributes);
5573
- try {
5574
- return await fn(span);
5575
- } catch (err) {
5576
- span.recordException(err);
5577
- span.setStatus({ code: SpanStatusCode.ERROR, message: err.message });
5578
- throw err;
5579
- } finally {
5580
- span.end();
5581
- }
5582
- }));
5583
- }
5584
- function getCurrentCallContext() {
5585
- return callContextStore.getStore();
5586
- }
5587
- var TRACER_NAME, callContextStore;
5588
- var init_call_context = __esm({
5589
- "../observability/dist/call-context.js"() {
5590
- init_attributes();
5591
- TRACER_NAME = "@voicelayer/observability";
5592
- callContextStore = new AsyncLocalStorage();
5593
- }
5594
- });
5595
- var init_trace_propagation = __esm({
5596
- "../observability/dist/trace-propagation.js"() {
5597
- }
5598
- });
5599
- function meter() {
5600
- return metrics.getMeter(METER_NAME, METER_VERSION);
5601
- }
5602
- function attrs(a) {
5603
- const out = {
5604
- [ATTR.projectId]: a.projectId,
5605
- [ATTR.callId]: a.callId,
5606
- "vl.provider": a.provider
5607
- };
5608
- if (a.campaignId)
5609
- out[ATTR.campaignId] = a.campaignId;
5610
- if (a.model)
5611
- out["vl.model"] = a.model;
5612
- return out;
5613
- }
5614
- function llmInputTokens() {
5615
- return _llmInputTokens ??= meter().createCounter(USAGE_METRIC_NAMES.llmInputTokens, {
5616
- description: "Input tokens consumed by an LLM call",
5617
- unit: "{tokens}"
5618
- });
5619
- }
5620
- function llmOutputTokens() {
5621
- return _llmOutputTokens ??= meter().createCounter(USAGE_METRIC_NAMES.llmOutputTokens, {
5622
- description: "Output tokens produced by an LLM call",
5623
- unit: "{tokens}"
5624
- });
5625
- }
5626
- function ttsChars() {
5627
- return _ttsChars ??= meter().createCounter(USAGE_METRIC_NAMES.ttsChars, {
5628
- description: "Characters synthesized by TTS",
5629
- unit: "{chars}"
5451
+ function ttsChars() {
5452
+ return _ttsChars ??= meter().createCounter(USAGE_METRIC_NAMES.ttsChars, {
5453
+ description: "Characters synthesized by TTS",
5454
+ unit: "{chars}"
5630
5455
  });
5631
5456
  }
5632
5457
  function sttSeconds() {
@@ -5850,15 +5675,568 @@ var init_latency_span = __esm({
5850
5675
  }
5851
5676
  });
5852
5677
 
5853
- // ../observability/dist/index.js
5854
- var init_dist2 = __esm({
5855
- "../observability/dist/index.js"() {
5856
- init_start();
5857
- init_call_context();
5858
- init_trace_propagation();
5859
- init_attributes();
5860
- init_metrics();
5861
- init_latency_span();
5678
+ // ../observability/dist/index.js
5679
+ var init_dist2 = __esm({
5680
+ "../observability/dist/index.js"() {
5681
+ init_start();
5682
+ init_call_context();
5683
+ init_trace_propagation();
5684
+ init_attributes();
5685
+ init_metrics();
5686
+ init_latency_span();
5687
+ }
5688
+ });
5689
+
5690
+ // src/providers/resilient-llm.ts
5691
+ var resilient_llm_exports = {};
5692
+ __export(resilient_llm_exports, {
5693
+ LLM_APOLOGY: () => LLM_APOLOGY,
5694
+ ResilientLLM: () => ResilientLLM,
5695
+ isResilientLLM: () => isResilientLLM
5696
+ });
5697
+ function isResilientLLM(value) {
5698
+ return typeof value === "object" && value !== null && value[RESILIENT] === true;
5699
+ }
5700
+ function transient(error) {
5701
+ if (error instanceof APIStatusError) {
5702
+ const s = error.statusCode;
5703
+ return s === 408 || s === 429 || s < 0 || s >= 500;
5704
+ }
5705
+ return error instanceof APITimeoutError || error instanceof APIConnectionError;
5706
+ }
5707
+ var LLM_APOLOGY, RESILIENT, ResilientLLM, sameModel, ServedLLM, ResilientLLMStream;
5708
+ var init_resilient_llm = __esm({
5709
+ "src/providers/resilient-llm.ts"() {
5710
+ init_dist();
5711
+ init_dist2();
5712
+ LLM_APOLOGY = "Sorry, I'm having trouble right now. Could you give me a moment and say that again?";
5713
+ RESILIENT = /* @__PURE__ */ Symbol.for("voicelayer.resilientLLM");
5714
+ ResilientLLM = class extends llm.LLM {
5715
+ [RESILIENT] = true;
5716
+ #opts;
5717
+ #fallback = null;
5718
+ #listeners = /* @__PURE__ */ new Set();
5719
+ /** Models the provider rejected on this call (a call's pipeline is built per call): later turns go straight to the fallback. */
5720
+ rejections = new ModelRejectionCache();
5721
+ constructor(opts) {
5722
+ super();
5723
+ this.#opts = opts;
5724
+ }
5725
+ label() {
5726
+ return this.#opts.primary.label();
5727
+ }
5728
+ get model() {
5729
+ return this.#opts.primary.model;
5730
+ }
5731
+ get provider() {
5732
+ return this.#opts.primary.provider;
5733
+ }
5734
+ get apology() {
5735
+ return this.#opts.apology ?? LLM_APOLOGY;
5736
+ }
5737
+ get reasoningParams() {
5738
+ return this.#opts.reasoningParams === true;
5739
+ }
5740
+ /** The fallback LLM, built on first need; null when there is none or it is the agent's own model. */
5741
+ fallbackLLM() {
5742
+ if (!this.#opts.fallback) return null;
5743
+ this.#fallback ??= this.#opts.fallback();
5744
+ return sameModel(this.#fallback.model, this.model) ? null : this.#fallback;
5745
+ }
5746
+ /** Hear about every recovery (and every apology). Returns the unsubscribe. */
5747
+ onIncident(listener) {
5748
+ this.#listeners.add(listener);
5749
+ return () => this.#listeners.delete(listener);
5750
+ }
5751
+ /** @internal */
5752
+ report(incident) {
5753
+ const log = incident.kind === "apology" ? console.error : console.warn;
5754
+ log(`[voicelayer] llm ${incident.kind.replace(/_/g, " ")}`, incident);
5755
+ if (incident.kind === "model_fallback") {
5756
+ const call = getCurrentCallContext();
5757
+ recordModelFallback({
5758
+ model: incident.model,
5759
+ fallbackModel: incident.fallbackModel,
5760
+ code: incident.code,
5761
+ surface: "voice",
5762
+ ...call?.projectId ? { projectId: call.projectId } : {},
5763
+ ...call?.agentId ? { agentId: call.agentId } : {}
5764
+ });
5765
+ }
5766
+ for (const listener of this.#listeners) {
5767
+ try {
5768
+ listener(incident);
5769
+ } catch {
5770
+ }
5771
+ }
5772
+ }
5773
+ chat(args) {
5774
+ return new ResilientLLMStream(this, args);
5775
+ }
5776
+ prewarm() {
5777
+ this.#opts.primary.prewarm();
5778
+ }
5779
+ async aclose() {
5780
+ await Promise.all([this.#opts.primary.aclose(), this.#fallback?.aclose()]);
5781
+ }
5782
+ /** @internal */
5783
+ get primary() {
5784
+ return this.#opts.primary;
5785
+ }
5786
+ };
5787
+ sameModel = (a, b) => a.trim().toLowerCase() === b.trim().toLowerCase();
5788
+ ServedLLM = class extends llm.LLM {
5789
+ constructor(owner, served) {
5790
+ super();
5791
+ this.owner = owner;
5792
+ this.served = served;
5793
+ }
5794
+ owner;
5795
+ served;
5796
+ label() {
5797
+ return this.owner.label();
5798
+ }
5799
+ get model() {
5800
+ return this.served();
5801
+ }
5802
+ get provider() {
5803
+ return this.owner.provider;
5804
+ }
5805
+ chat(args) {
5806
+ return this.owner.chat(args);
5807
+ }
5808
+ emit(event, ...args) {
5809
+ return this.owner.emit(event, ...args);
5810
+ }
5811
+ };
5812
+ ResilientLLMStream = class extends llm.LLMStream {
5813
+ #owner;
5814
+ #args;
5815
+ #conn;
5816
+ /** The model answering this stream — what its metrics report. */
5817
+ #served;
5818
+ #current = null;
5819
+ constructor(owner, args) {
5820
+ const conn = args.connOptions ?? DEFAULT_API_CONNECT_OPTIONS;
5821
+ const served = { model: owner.model };
5822
+ super(new ServedLLM(owner, () => served.model), {
5823
+ chatCtx: args.chatCtx,
5824
+ ...args.toolCtx ? { toolCtx: args.toolCtx } : {},
5825
+ connOptions: { ...conn, maxRetry: 0 }
5826
+ });
5827
+ this.#owner = owner;
5828
+ this.#args = args;
5829
+ this.#conn = conn;
5830
+ this.#served = served;
5831
+ this.abortController.signal.addEventListener("abort", () => this.#current?.close());
5832
+ }
5833
+ get #hasTools() {
5834
+ return this.#args.toolCtx !== void 0 && Object.keys(this.#args.toolCtx).length > 0;
5835
+ }
5836
+ #extraKwargs(model2) {
5837
+ const base = this.#args.extraKwargs;
5838
+ if (!this.#owner.reasoningParams) return base;
5839
+ const effort = chatReasoningEffort(model2, { tools: this.#hasTools });
5840
+ if (effort === void 0) {
5841
+ if (!base || !("reasoning_effort" in base)) return base;
5842
+ const { reasoning_effort: _dropped, ...rest } = base;
5843
+ return rest;
5844
+ }
5845
+ return { ...base, reasoning_effort: effort };
5846
+ }
5847
+ /** One request on `target`, forwarding its chunks. Its failure is the inner LLM's 'error' event. */
5848
+ async #attempt(target) {
5849
+ let failure2 = null;
5850
+ const onError = (ev) => {
5851
+ failure2 ??= ev.error;
5852
+ };
5853
+ target.on("error", onError);
5854
+ let started2 = false;
5855
+ try {
5856
+ const extraKwargs = this.#extraKwargs(target.model);
5857
+ const stream = target.chat({
5858
+ ...this.#args,
5859
+ connOptions: { ...this.#conn, maxRetry: 0 },
5860
+ ...extraKwargs !== void 0 ? { extraKwargs } : {}
5861
+ });
5862
+ this.#current = stream;
5863
+ for await (const chunk of stream) {
5864
+ if (this.abortController.signal.aborted) break;
5865
+ started2 = true;
5866
+ this.queue.put(chunk);
5867
+ }
5868
+ } catch (err) {
5869
+ failure2 ??= err instanceof Error ? err : new Error(String(err));
5870
+ } finally {
5871
+ target.off("error", onError);
5872
+ this.#current = null;
5873
+ }
5874
+ return failure2 ? { ok: false, error: failure2, started: started2 } : { ok: true };
5875
+ }
5876
+ /** Run `target` until it answers, or a failure that retrying it won't fix. */
5877
+ async #run(target) {
5878
+ let effortRetried = false;
5879
+ for (let retries = 0; ; ) {
5880
+ const sent = this.#owner.reasoningParams ? chatReasoningEffort(target.model, { tools: this.#hasTools }) : void 0;
5881
+ const result = await this.#attempt(target);
5882
+ if (result.ok || result.started || this.abortController.signal.aborted) return result;
5883
+ if (this.#owner.reasoningParams && !effortRetried && learnReasoningEffort(target.model, { tools: this.#hasTools, sent }, result.error)) {
5884
+ effortRetried = true;
5885
+ this.#owner.report({
5886
+ kind: "reasoning_effort_adapted",
5887
+ model: target.model,
5888
+ message: providerErrorOf(result.error)?.message ?? result.error.message
5889
+ });
5890
+ continue;
5891
+ }
5892
+ if (!transient(result.error) || retries >= this.#conn.maxRetry) return result;
5893
+ const wait = intervalForRetry(this.#conn, retries);
5894
+ retries += 1;
5895
+ if (wait > 0) await new Promise((r) => setTimeout(r, wait));
5896
+ if (this.abortController.signal.aborted) return result;
5897
+ }
5898
+ }
5899
+ async run() {
5900
+ const owner = this.#owner;
5901
+ const model2 = owner.model;
5902
+ const fallback = owner.fallbackLLM();
5903
+ const known = fallback ? owner.rejections.get(model2) : null;
5904
+ let failure2;
5905
+ if (known && fallback) {
5906
+ owner.report({ kind: "model_fallback", model: model2, fallbackModel: fallback.model, code: known.code, message: known.message, cached: true });
5907
+ failure2 = new Error(known.message);
5908
+ } else {
5909
+ const first = await this.#run(owner.primary);
5910
+ if (first.ok || first.started || this.abortController.signal.aborted) return;
5911
+ failure2 = first.error;
5912
+ const rejection = modelRejectionOf(first.error);
5913
+ if (rejection) owner.rejections.set(model2, rejection);
5914
+ if (fallback) {
5915
+ const f = providerErrorOf(first.error);
5916
+ owner.report({
5917
+ kind: "model_fallback",
5918
+ model: model2,
5919
+ fallbackModel: fallback.model,
5920
+ code: rejection?.code ?? (f?.status ? String(f.status) : "error"),
5921
+ message: f?.message || first.error.message,
5922
+ cached: false
5923
+ });
5924
+ }
5925
+ }
5926
+ if (fallback) {
5927
+ this.#served.model = fallback.model;
5928
+ const second = await this.#run(fallback);
5929
+ if (second.ok || second.started || this.abortController.signal.aborted) return;
5930
+ failure2 = second.error;
5931
+ }
5932
+ owner.report({ kind: "apology", model: this.#served.model, message: providerErrorOf(failure2)?.message || failure2.message });
5933
+ this.queue.put({ id: `vl-apology-${Date.now()}`, delta: { role: "assistant", content: owner.apology } });
5934
+ }
5935
+ };
5936
+ }
5937
+ });
5938
+
5939
+ // src/providers/index.ts
5940
+ async function importOptional(spec, hint) {
5941
+ try {
5942
+ return await import(spec);
5943
+ } catch {
5944
+ throw new Error(`${hint}: "${spec}" is not installed. Add it with: pnpm add ${spec}`);
5945
+ }
5946
+ }
5947
+ function withCreds(opts, creds2) {
5948
+ if (creds2?.apiKey) opts["apiKey"] = creds2.apiKey;
5949
+ if (creds2?.baseURL) opts["baseURL"] = creds2.baseURL;
5950
+ return opts;
5951
+ }
5952
+ var deepgram, openai, anthropic, cartesia, elevenlabs, assemblyai, google, silero, livekitTurn, connector;
5953
+ var init_providers2 = __esm({
5954
+ "src/providers/index.ts"() {
5955
+ init_ssrf();
5956
+ init_transport_callback();
5957
+ init_connector_llm();
5958
+ init_wrap();
5959
+ deepgram = {
5960
+ stt(options = {}) {
5961
+ return async () => {
5962
+ const dg = await importOptional("@voicelayer/agents-plugin-deepgram", "deepgram.stt");
5963
+ return new dg.STT(
5964
+ withCreds(
5965
+ { model: options.model ?? "nova-3", language: options.language ?? "en-US" },
5966
+ options
5967
+ )
5968
+ );
5969
+ };
5970
+ },
5971
+ tts(options = {}) {
5972
+ return async () => {
5973
+ const dg = await importOptional("@voicelayer/agents-plugin-deepgram", "deepgram.tts");
5974
+ return new dg.TTS(
5975
+ withCreds({ model: options.model ?? "aura-2-harmonia-en" }, options)
5976
+ );
5977
+ };
5978
+ }
5979
+ };
5980
+ openai = {
5981
+ llm(options = {}) {
5982
+ return async () => {
5983
+ const oa = await importOptional("@voicelayer/agents-plugin-openai", "openai.llm");
5984
+ const { ResilientLLM: ResilientLLM2 } = await Promise.resolve().then(() => (init_resilient_llm(), resilient_llm_exports));
5985
+ const { PLATFORM_DEFAULT_CHAT_MODEL: PLATFORM_DEFAULT_CHAT_MODEL2 } = await Promise.resolve().then(() => (init_dist(), dist_exports));
5986
+ return new ResilientLLM2({
5987
+ primary: new oa.LLM(withCreds({ model: options.model ?? "gpt-4o-mini" }, options)),
5988
+ fallback: () => new oa.LLM(withCreds({ model: PLATFORM_DEFAULT_CHAT_MODEL2 }, options)),
5989
+ reasoningParams: true
5990
+ });
5991
+ };
5992
+ },
5993
+ tts(options = {}) {
5994
+ return async () => {
5995
+ const oa = await importOptional("@voicelayer/agents-plugin-openai", "openai.tts");
5996
+ const opts = { model: options.model ?? "tts-1" };
5997
+ if (options.voice) opts["voice"] = options.voice;
5998
+ if (options.instructions) opts["instructions"] = options.instructions;
5999
+ return new oa.TTS(withCreds(opts, options));
6000
+ };
6001
+ },
6002
+ // Speech-to-speech via the OpenAI Realtime API. Returns a RealtimeProvider
6003
+ // that the SDK uses in place of the stt/llm/tts pipeline.
6004
+ realtime(options = {}) {
6005
+ return async () => {
6006
+ const oa = await importOptional("@voicelayer/agents-plugin-openai", "openai.realtime");
6007
+ const opts = {};
6008
+ if (options.model) opts["model"] = options.model;
6009
+ if (options.voice) opts["voice"] = options.voice;
6010
+ return new oa.realtime.RealtimeModel(withCreds(opts, options));
6011
+ };
6012
+ }
6013
+ };
6014
+ anthropic = {
6015
+ llm(options = {}) {
6016
+ return async () => {
6017
+ const an = await importOptional("@livekit/agents-plugin-anthropic", "anthropic.llm");
6018
+ return new an.LLM(
6019
+ withCreds({ model: options.model ?? "claude-sonnet-5-5" }, options)
6020
+ );
6021
+ };
6022
+ }
6023
+ };
6024
+ cartesia = {
6025
+ tts(options = {}) {
6026
+ return async () => {
6027
+ const ct = await importOptional("@voicelayer/agents-plugin-cartesia", "cartesia.tts");
6028
+ return new ct.TTS(options);
6029
+ };
6030
+ },
6031
+ stt(options = {}) {
6032
+ return async () => {
6033
+ const ct = await importOptional("@voicelayer/agents-plugin-cartesia", "cartesia.stt");
6034
+ const opts = {};
6035
+ if (options.model) opts["model"] = options.model;
6036
+ if (options.language) opts["language"] = options.language;
6037
+ return new ct.STT(withCreds(opts, options));
6038
+ };
6039
+ }
6040
+ };
6041
+ elevenlabs = {
6042
+ tts(options = {}) {
6043
+ return async () => {
6044
+ const el = await importOptional("@voicelayer/agents-plugin-elevenlabs", "elevenlabs.tts");
6045
+ const opts = {};
6046
+ if (options.voice) opts["voiceId"] = options.voice;
6047
+ if (options.model) opts["model"] = options.model;
6048
+ return new el.TTS(withCreds(opts, options));
6049
+ };
6050
+ }
6051
+ };
6052
+ assemblyai = {
6053
+ stt(options = {}) {
6054
+ return async () => {
6055
+ const aai = await importOptional("@voicelayer/agents-plugin-assemblyai", "assemblyai.stt");
6056
+ const opts = {};
6057
+ if (options.language) opts["language"] = options.language;
6058
+ return new aai.STT(withCreds(opts, options));
6059
+ };
6060
+ }
6061
+ };
6062
+ google = {
6063
+ llm(options = {}) {
6064
+ return async () => {
6065
+ const g = await importOptional("@voicelayer/agents-plugin-google", "google.llm");
6066
+ const { ResilientLLM: ResilientLLM2 } = await Promise.resolve().then(() => (init_resilient_llm(), resilient_llm_exports));
6067
+ return new ResilientLLM2({
6068
+ primary: new g.LLM(withCreds({ model: options.model ?? "gemini-3.8-flash" }, options))
6069
+ });
6070
+ };
6071
+ },
6072
+ realtime(options = {}) {
6073
+ return async () => {
6074
+ const g = await importOptional("@voicelayer/agents-plugin-google", "google.realtime");
6075
+ const opts = {};
6076
+ if (options.model) opts["model"] = options.model;
6077
+ if (options.voice) opts["voice"] = options.voice;
6078
+ return new g.beta.realtime.RealtimeModel(withCreds(opts, options));
6079
+ };
6080
+ }
6081
+ };
6082
+ silero = {
6083
+ vad() {
6084
+ return async () => {
6085
+ const sil = await importOptional("@voicelayer/agents-plugin-silero", "silero.vad");
6086
+ return await sil.VAD.load();
6087
+ };
6088
+ }
6089
+ };
6090
+ livekitTurn = {
6091
+ english() {
6092
+ return async () => {
6093
+ const lk = await importOptional("@voicelayer/agents-plugin-livekit", "livekitTurn.english");
6094
+ return new lk.turnDetector.EnglishModel();
6095
+ };
6096
+ },
6097
+ multilingual() {
6098
+ return async () => {
6099
+ const lk = await importOptional("@voicelayer/agents-plugin-livekit", "livekitTurn.multilingual");
6100
+ return new lk.turnDetector.MultilingualModel();
6101
+ };
6102
+ }
6103
+ };
6104
+ connector = {
6105
+ llm(config = {}) {
6106
+ return async (call) => {
6107
+ if (config.url) {
6108
+ await assertPublicHttpsUrl(
6109
+ config.url,
6110
+ config.allowHosts ? { allowHosts: config.allowHosts } : {}
6111
+ );
6112
+ return openai.llm({
6113
+ ...config.model ? { model: config.model } : {},
6114
+ ...config.apiKey ? { apiKey: config.apiKey } : {},
6115
+ baseURL: config.url
6116
+ })(call);
6117
+ }
6118
+ const transport = config.transport ?? (config.onQuery ? callbackTransport(config.onQuery) : void 0);
6119
+ if (!transport) {
6120
+ throw new Error("connector.llm requires one of: url, onQuery, or transport");
6121
+ }
6122
+ const projectId = typeof call.metadata["projectId"] === "string" ? call.metadata["projectId"] : void 0;
6123
+ const llmOptions = {
6124
+ ...config.model ? { model: config.model } : {},
6125
+ ...config.temperature !== void 0 ? { temperature: config.temperature } : {},
6126
+ ...config.fallbackText ? { fallbackText: config.fallbackText } : {},
6127
+ ...call.callId ? { callId: call.callId } : {},
6128
+ ...projectId ? { projectId } : {}
6129
+ };
6130
+ return await createConnectorLLM(transport, llmOptions);
6131
+ };
6132
+ }
6133
+ };
6134
+ }
6135
+ });
6136
+
6137
+ // src/providers/registry.ts
6138
+ function entry(id, bits) {
6139
+ const identity = providerIdentity(id);
6140
+ if (!identity) throw new Error(`PROVIDER_REGISTRY: no provider identity in contracts for '${id}'`);
6141
+ return { name: identity.id, capabilities: identity.capabilities, ...bits };
6142
+ }
6143
+ function providerEntry(name) {
6144
+ if (!name) return void 0;
6145
+ return PROVIDER_REGISTRY[PROVIDER_ALIASES[name] ?? name];
6146
+ }
6147
+ function resolveStt(name, o) {
6148
+ return (providerEntry(name)?.stt ?? PROVIDER_REGISTRY["deepgram"].stt)(o);
6149
+ }
6150
+ function resolveTts(name, o) {
6151
+ return (providerEntry(name)?.tts ?? PROVIDER_REGISTRY["deepgram"].tts)(o);
6152
+ }
6153
+ function resolveLlm(name, o) {
6154
+ return (providerEntry(name)?.llm ?? PROVIDER_REGISTRY["openai"].llm)(withCurrentModel(name, o));
6155
+ }
6156
+ function resolveRealtime(name, o) {
6157
+ return (providerEntry(name)?.realtime ?? PROVIDER_REGISTRY["openai"].realtime)(withCurrentModel(name, o));
6158
+ }
6159
+ function withCurrentModel(name, o) {
6160
+ if (!name || !o.model) return o;
6161
+ const current = currentModelFor(name, o.model);
6162
+ if (!current.retired) return o;
6163
+ console.warn("[agent] the configured model is retired; running its replacement", {
6164
+ provider: name,
6165
+ model: current.retired,
6166
+ replacement: current.model
6167
+ });
6168
+ return { ...o, model: current.model };
6169
+ }
6170
+ var model, lang, voice, creds, PROVIDER_REGISTRY;
6171
+ var init_registry = __esm({
6172
+ "src/providers/registry.ts"() {
6173
+ init_providers2();
6174
+ init_src();
6175
+ model = (o) => o.model ? { model: o.model } : {};
6176
+ lang = (o) => o.language ? { language: o.language } : {};
6177
+ voice = (o) => o.voice ? { voice: o.voice } : {};
6178
+ creds = (o) => o.creds ?? {};
6179
+ PROVIDER_REGISTRY = {
6180
+ deepgram: entry("deepgram", {
6181
+ failureClass: "vendor-api",
6182
+ engines: ["livekit"],
6183
+ stt: (o) => deepgram.stt({ ...model(o), ...lang(o), ...creds(o) }),
6184
+ // Deepgram folds the voice into the model id (aura-2-<name>-en): wire config
6185
+ // passes a voice, code config passes a model — both land as `model`.
6186
+ tts: (o) => {
6187
+ const m = o.voice ?? o.model;
6188
+ return deepgram.tts({ ...m ? { model: m } : {}, ...creds(o) });
6189
+ }
6190
+ }),
6191
+ openai: entry("openai", {
6192
+ failureClass: "vendor-api",
6193
+ engines: ["livekit", "text"],
6194
+ llm: (o) => openai.llm({ ...model(o), ...creds(o) }),
6195
+ tts: (o) => openai.tts({ ...voice(o), ...model(o), ...creds(o) }),
6196
+ realtime: (o) => openai.realtime({ ...model(o), ...voice(o), ...creds(o) })
6197
+ }),
6198
+ anthropic: entry("anthropic", {
6199
+ failureClass: "vendor-api",
6200
+ engines: ["livekit", "text"],
6201
+ llm: (o) => anthropic.llm({ ...model(o), ...creds(o) })
6202
+ }),
6203
+ google: entry("google", {
6204
+ failureClass: "vendor-api",
6205
+ engines: ["livekit", "text"],
6206
+ llm: (o) => google.llm({ ...model(o), ...creds(o) }),
6207
+ realtime: (o) => google.realtime({ ...model(o), ...voice(o), ...creds(o) })
6208
+ }),
6209
+ cartesia: entry("cartesia", {
6210
+ failureClass: "vendor-api",
6211
+ engines: ["livekit"],
6212
+ stt: (o) => cartesia.stt({ ...model(o), ...lang(o), ...creds(o) }),
6213
+ tts: (o) => cartesia.tts({ ...voice(o), ...model(o), ...creds(o) })
6214
+ }),
6215
+ elevenlabs: entry("elevenlabs", {
6216
+ failureClass: "vendor-api",
6217
+ engines: ["livekit"],
6218
+ tts: (o) => elevenlabs.tts({ ...voice(o), ...model(o), ...creds(o) })
6219
+ }),
6220
+ assemblyai: entry("assemblyai", {
6221
+ failureClass: "vendor-api",
6222
+ engines: ["livekit"],
6223
+ stt: (o) => assemblyai.stt({ ...lang(o), ...creds(o) })
6224
+ // no model knob
6225
+ }),
6226
+ silero: entry("silero", {
6227
+ failureClass: "local",
6228
+ engines: ["livekit"],
6229
+ vad: () => silero.vad()
6230
+ })
6231
+ };
6232
+ }
6233
+ });
6234
+
6235
+ // src/providers/llm.ts
6236
+ var init_llm = __esm({
6237
+ "src/providers/llm.ts"() {
6238
+ init_openai_default();
6239
+ init_registry();
5862
6240
  }
5863
6241
  });
5864
6242
 
@@ -9226,14 +9604,23 @@ function defaultComplete() {
9226
9604
  model: model2,
9227
9605
  fallbackModel: RUNNER_MODEL,
9228
9606
  helper: "agent runner",
9229
- call: (m) => client.chat.completions.create(
9230
- {
9231
- model: m,
9232
- ...chatCompletionParams(m, { ...temperature !== void 0 ? { temperature } : {}, reasoningEffort: "low" }),
9233
- messages,
9234
- ...tools.length ? { tools } : {}
9235
- },
9236
- { timeout: RUNNER_TIMEOUT_MS }
9607
+ // With tools, a model that takes effort 'none' gets 'none' (OpenAI refuses tools at any other effort on chat
9608
+ // completions for GPT-5.2 and later), and a reasoning_effort 400 is learned and retried once (burn-down G-45).
9609
+ call: (m) => withReasoningEffortRetry(
9610
+ { model: m, tools: tools.length > 0, requested: "low" },
9611
+ () => client.chat.completions.create(
9612
+ {
9613
+ model: m,
9614
+ ...chatCompletionParams(m, {
9615
+ ...temperature !== void 0 ? { temperature } : {},
9616
+ reasoningEffort: "low",
9617
+ tools: tools.length > 0
9618
+ }),
9619
+ messages,
9620
+ ...tools.length ? { tools } : {}
9621
+ },
9622
+ { timeout: RUNNER_TIMEOUT_MS }
9623
+ )
9237
9624
  ),
9238
9625
  warn: graphWarn
9239
9626
  });
@@ -11984,6 +12371,9 @@ var MemoryClient = class {
11984
12371
  );
11985
12372
  }
11986
12373
  };
12374
+
12375
+ // src/calls.ts
12376
+ init_src();
11987
12377
  var CallSchema = z.object({
11988
12378
  id: z.string(),
11989
12379
  projectId: z.string(),
@@ -12042,6 +12432,7 @@ z.object({
12042
12432
  });
12043
12433
  var CallStateSyncResponse = z.object({ ok: z.literal(true) }).or(z.object({}).passthrough());
12044
12434
  var DtmfEventResponse = z.object({ accepted: z.literal(true) }).or(z.object({}).passthrough());
12435
+ var CallFinalizeResponse = z.object({ ok: z.boolean() }).or(z.object({}).passthrough());
12045
12436
  var UsageReportResponse = z.object({ recorded: z.boolean() }).or(z.object({}).passthrough());
12046
12437
  var CallEventEnvelopeResponse = z.object({
12047
12438
  event: z.object({
@@ -12080,13 +12471,28 @@ var CallsClient = class {
12080
12471
  ...input.attributes !== void 0 ? { attributes: input.attributes } : {},
12081
12472
  ...input.participants !== void 0 ? { participants: input.participants } : {},
12082
12473
  ...input.legs !== void 0 ? { legs: input.legs } : {},
12083
- ...input.startedAt !== void 0 ? { startedAt: input.startedAt } : {}
12474
+ ...input.startedAt !== void 0 ? { startedAt: input.startedAt } : {},
12475
+ ...input.pipelineKeys !== void 0 ? { pipelineKeys: input.pipelineKeys } : {},
12476
+ ...input.finalizes !== void 0 ? { finalizes: input.finalizes } : {}
12084
12477
  }
12085
12478
  },
12086
12479
  CallByRoomResponse
12087
12480
  );
12088
12481
  return call;
12089
12482
  }
12483
+ /**
12484
+ * The call's admission (burn-down G-41/G-42/G-43), asked once right after the caller joins and before the agent
12485
+ * speaks: admitted with its duration limits, or refused with the end reason the platform recorded and the line to
12486
+ * speak before hanging up. Throws on a transport / server error — the runtime retries, then refuses (fail closed).
12487
+ */
12488
+ async admit(input) {
12489
+ return this.transport.request(
12490
+ // The runtime owns the retries (bounded, then a polite refusal): a server error comes straight back, and one
12491
+ // attempt never holds a caller longer than 2.5 s.
12492
+ { method: "POST", path: "/v1/calls/admission", body: input, noRetryStatuses: [500, 502, 503, 504], signal: AbortSignal.timeout(2500) },
12493
+ CallAdmissionResponse
12494
+ );
12495
+ }
12090
12496
  async syncState(callId, input) {
12091
12497
  await this.transport.request(
12092
12498
  {
@@ -12120,11 +12526,12 @@ var CallsClient = class {
12120
12526
  }
12121
12527
  // One-shot cascade usage report at session close — the platform prices it
12122
12528
  // into the call's bill (voice_components). Idempotent server-side.
12123
- async reportUsage(callId, report) {
12529
+ async reportUsage(callId, report, opts = {}) {
12124
12530
  await this.transport.request(
12125
12531
  {
12126
12532
  method: "POST",
12127
12533
  path: `/v1/calls/${encodeURIComponent(callId)}/usage-report`,
12534
+ ...opts.signal ? { signal: opts.signal } : {},
12128
12535
  body: {
12129
12536
  durationMs: report.durationMs,
12130
12537
  llm: report.llm,
@@ -12135,6 +12542,22 @@ var CallsClient = class {
12135
12542
  UsageReportResponse
12136
12543
  );
12137
12544
  }
12545
+ /**
12546
+ * The worker's last word on a call (burn-down G-20): what the call captured — every declared field — sent at the
12547
+ * session's close after the worker's other writes for the call have landed. The API keeps the fields on the call
12548
+ * and releases the call's held `call.ended` webhook. Idempotent.
12549
+ */
12550
+ async finalize(callId, input, opts = {}) {
12551
+ await this.transport.request(
12552
+ {
12553
+ method: "POST",
12554
+ path: `/v1/calls/${encodeURIComponent(callId)}/finalize`,
12555
+ ...opts.signal ? { signal: opts.signal } : {},
12556
+ body: { fields: input.fields }
12557
+ },
12558
+ CallFinalizeResponse
12559
+ );
12560
+ }
12138
12561
  async appendDtmfEvent(callId, input) {
12139
12562
  await this.transport.request(
12140
12563
  {
@@ -12850,7 +13273,7 @@ async function resolvePipeline(cfg, call, preloadedVad) {
12850
13273
  () => language.startsWith("en") ? livekitTurn.english() : livekitTurn.multilingual(),
12851
13274
  () => language.startsWith("en") ? livekitTurn.english() : livekitTurn.multilingual()
12852
13275
  ) : null;
12853
- const [stt, llm3, tts, vad] = await Promise.all([
13276
+ const [stt, llm4, tts, vad] = await Promise.all([
12854
13277
  sttFactory2(call),
12855
13278
  llmFactory2(call),
12856
13279
  ttsFactory2(call),
@@ -12866,7 +13289,7 @@ async function resolvePipeline(cfg, call, preloadedVad) {
12866
13289
  });
12867
13290
  }
12868
13291
  }
12869
- return { stt, llm: llm3, tts, vad, turnDetector };
13292
+ return { stt, llm: llm4, tts, vad, turnDetector };
12870
13293
  }
12871
13294
  var DEFAULT_REQUIRED_ENV = [
12872
13295
  "LIVEKIT_URL",
@@ -13304,6 +13727,38 @@ function buildRedisBrainPubSub(redisUrl) {
13304
13727
  // src/runtime/pipeline-from-config.ts
13305
13728
  init_providers2();
13306
13729
  init_registry();
13730
+
13731
+ // src/runtime/pipeline-key-owner.ts
13732
+ var vaultKeyed = /* @__PURE__ */ new WeakSet();
13733
+ var isObject = (v) => v !== null && (typeof v === "object" || typeof v === "function");
13734
+ function vaultKeyedFactory(factory) {
13735
+ return async (call) => {
13736
+ const provider = await factory(call);
13737
+ if (isObject(provider)) vaultKeyed.add(provider);
13738
+ return provider;
13739
+ };
13740
+ }
13741
+ function builtWithVaultKey(provider) {
13742
+ return isObject(provider) && vaultKeyed.has(provider);
13743
+ }
13744
+ function providerLabel(provider) {
13745
+ const label = isObject(provider) ? provider.provider : void 0;
13746
+ return typeof label === "string" && label.length > 0 ? label : "unknown";
13747
+ }
13748
+ function pipelineKeys(pipeline, envKeyIsCustomers) {
13749
+ const slot = (provider) => ({
13750
+ provider: providerLabel(provider),
13751
+ byok: builtWithVaultKey(provider) || envKeyIsCustomers
13752
+ });
13753
+ return {
13754
+ ...pipeline.llm != null ? { llm: slot(pipeline.llm) } : {},
13755
+ ...pipeline.stt != null ? { stt: slot(pipeline.stt) } : {},
13756
+ ...pipeline.tts != null ? { tts: slot(pipeline.tts) } : {},
13757
+ ...pipeline.realtime != null ? { realtime: slot(pipeline.realtime) } : {}
13758
+ };
13759
+ }
13760
+
13761
+ // src/runtime/pipeline-from-config.ts
13307
13762
  function modelsFromWireConfig(p, creds2) {
13308
13763
  const language = p.language ?? "en-US";
13309
13764
  if (p.mode === "realtime" && p.realtime) {
@@ -13317,17 +13772,20 @@ function modelsFromWireConfig(p, creds2) {
13317
13772
  out.tts = ttsFactory(p.voice.provider, p.voice.voiceId, p.voice.model, creds2?.[p.voice.provider]);
13318
13773
  return out;
13319
13774
  }
13775
+ function owned(factory, creds2) {
13776
+ return creds2 ? vaultKeyedFactory(factory) : factory;
13777
+ }
13320
13778
  function realtimeFactory(r, creds2) {
13321
- return resolveRealtime(r.provider, { model: r.model, ...r.voice ? { voice: r.voice } : {}, ...creds2 ? { creds: creds2 } : {} });
13779
+ return owned(resolveRealtime(r.provider, { model: r.model, ...r.voice ? { voice: r.voice } : {}, ...creds2 ? { creds: creds2 } : {} }), creds2);
13322
13780
  }
13323
13781
  function llmFactory(provider, model2, creds2) {
13324
- return resolveLlm(provider, { model: model2, ...creds2 ? { creds: creds2 } : {} });
13782
+ return owned(resolveLlm(provider, { model: model2, ...creds2 ? { creds: creds2 } : {} }), creds2);
13325
13783
  }
13326
13784
  function sttFactory(provider, model2, language, creds2) {
13327
- return resolveStt(provider, { ...model2 ? { model: model2 } : {}, language, ...creds2 ? { creds: creds2 } : {} });
13785
+ return owned(resolveStt(provider, { ...model2 ? { model: model2 } : {}, language, ...creds2 ? { creds: creds2 } : {} }), creds2);
13328
13786
  }
13329
13787
  function ttsFactory(provider, voiceId, model2, creds2) {
13330
- return resolveTts(provider, { voice: voiceId, ...model2 ? { model: model2 } : {}, ...creds2 ? { creds: creds2 } : {} });
13788
+ return owned(resolveTts(provider, { voice: voiceId, ...model2 ? { model: model2 } : {}, ...creds2 ? { creds: creds2 } : {} }), creds2);
13331
13789
  }
13332
13790
 
13333
13791
  // src/agent.ts
@@ -15513,6 +15971,9 @@ function extractText2(msg) {
15513
15971
  const trimmed = txt.trim();
15514
15972
  return trimmed.length === 0 ? null : trimmed;
15515
15973
  }
15974
+
15975
+ // src/runtime/handoff.ts
15976
+ init_src();
15516
15977
  function createHandoffRuntime(config, deps) {
15517
15978
  return {
15518
15979
  async handoff(to, opts) {
@@ -15661,7 +16122,9 @@ function buildDefaultSipOps(trunkOverride, fromNumber) {
15661
16122
  participantIdentity: identity,
15662
16123
  playDialtone: false,
15663
16124
  waitUntilAnswered: true,
15664
- ...fromNumber ? { fromNumber } : {}
16125
+ ...fromNumber ? { fromNumber } : {},
16126
+ // LiveKit's own ceiling on the transfer leg (burn-down G-43): it can't outlive the platform's longest call.
16127
+ maxCallDuration: SIP_LEG_MAX_CALL_DURATION_SECONDS
15665
16128
  });
15666
16129
  return { identity };
15667
16130
  }
@@ -16185,16 +16648,16 @@ async function exchangeWorkerToken(input) {
16185
16648
  const internalToken = process.env["INTERNAL_SERVICE_TOKEN"];
16186
16649
  if (!apiUrl || !internalToken) return null;
16187
16650
  const delays = input.retryDelaysMs ?? EXCHANGE_RETRY_DELAYS_MS;
16188
- const deadline = Date.now() + (input.budgetMs ?? EXCHANGE_BUDGET_MS);
16651
+ const deadline2 = Date.now() + (input.budgetMs ?? EXCHANGE_BUDGET_MS);
16189
16652
  for (let attempt = 0; ; attempt++) {
16190
- const left = deadline - Date.now();
16653
+ const left = deadline2 - Date.now();
16191
16654
  if (left <= 0) return null;
16192
16655
  const r = await exchangeOnce(input, apiUrl, internalToken, Math.min(input.attemptTimeoutMs ?? EXCHANGE_ATTEMPT_TIMEOUT_MS, left));
16193
16656
  if ("token" in r) return r.token;
16194
16657
  if ("stale" in r) throw new StaleRouteError(r.stale);
16195
16658
  const base = delays[attempt];
16196
16659
  const wait = base === void 0 ? void 0 : Math.max(base, "retryAfterMs" in r && r.retryAfterMs !== void 0 ? r.retryAfterMs : 0);
16197
- if (r.final || wait === void 0 || Date.now() + wait >= deadline) return null;
16660
+ if (r.final || wait === void 0 || Date.now() + wait >= deadline2) return null;
16198
16661
  await new Promise((resolve) => setTimeout(resolve, wait));
16199
16662
  }
16200
16663
  }
@@ -16330,7 +16793,7 @@ async function answerTextTurn(deps) {
16330
16793
  const { turn, config } = deps;
16331
16794
  const agentId = turn.agentId;
16332
16795
  const now = deps.now ?? Date.now;
16333
- const { voice: voice3, llm: llm3, initializeLogger, loggerOptions } = await import('@livekit/agents');
16796
+ const { voice: voice3, llm: llm4, initializeLogger, loggerOptions } = await import('@livekit/agents');
16334
16797
  if (loggerOptions() === void 0) initializeLogger({ pretty: false, level: "warn" });
16335
16798
  let effective = config;
16336
16799
  let pipeline = null;
@@ -16372,7 +16835,7 @@ async function answerTextTurn(deps) {
16372
16835
  getCtx
16373
16836
  });
16374
16837
  const instructions = processRt.augmentPrompt(composeSystemPrompt(effective.prompt, effective.routingInstructions));
16375
- const chatCtx = llm3.ChatContext.empty();
16838
+ const chatCtx = llm4.ChatContext.empty();
16376
16839
  let skipReply = false;
16377
16840
  for (const h of turn.history) {
16378
16841
  if (h.role === "user" && security && security.inputGuard.check(h.content).action === "block") {
@@ -16393,7 +16856,7 @@ async function answerTextTurn(deps) {
16393
16856
  );
16394
16857
  const brain = pipeline?.brain === "bound";
16395
16858
  const label = pipeline?.llm ?? { provider: "openai", model: model2.model };
16396
- const byok = pipeline ? pipeline.byokProviders.includes(label.provider) : false;
16859
+ const byok = builtWithVaultKey(model2);
16397
16860
  const usage = brain ? [{ provider: "brain", model: model2.model || "brain", inputTokens: 0, outputTokens: 0, byok: true }] : [];
16398
16861
  const onMetrics = (m) => {
16399
16862
  if (brain) return;
@@ -17040,8 +17503,18 @@ function guardIsActive(opts) {
17040
17503
  return false;
17041
17504
  }
17042
17505
 
17506
+ // src/runtime/close-deadline.ts
17507
+ var CLOSE_REQUEST_TIMEOUT_MS = 5e3;
17508
+ function deadline(ms) {
17509
+ const controller = new AbortController();
17510
+ const timer = setTimeout(() => controller.abort(new Error(`timed out after ${ms} ms`)), ms);
17511
+ timer.unref?.();
17512
+ return { signal: controller.signal, clear: () => clearTimeout(timer) };
17513
+ }
17514
+
17043
17515
  // src/runtime/call-sync.ts
17044
17516
  var END_REASON_WAIT_MS = 2e3;
17517
+ var FINALIZE_WRITES_WAIT_MS = 3e3;
17045
17518
  function makeNoopHandle(callId) {
17046
17519
  return {
17047
17520
  callId,
@@ -17053,6 +17526,13 @@ function makeNoopHandle(callId) {
17053
17526
  publishDtmf() {
17054
17527
  },
17055
17528
  async recordEndReason() {
17529
+ },
17530
+ publishTurnLatency() {
17531
+ },
17532
+ track(write) {
17533
+ return write;
17534
+ },
17535
+ async finalize() {
17056
17536
  }
17057
17537
  };
17058
17538
  }
@@ -17065,6 +17545,16 @@ async function createCallSyncHandle(init) {
17065
17545
  });
17066
17546
  return makeNoopHandle(init.call.callId);
17067
17547
  }
17548
+ const inFlight = /* @__PURE__ */ new Set();
17549
+ const track = (write) => {
17550
+ const settled = write.then(
17551
+ () => void 0,
17552
+ () => void 0
17553
+ );
17554
+ inFlight.add(settled);
17555
+ void settled.then(() => inFlight.delete(settled));
17556
+ return write;
17557
+ };
17068
17558
  return {
17069
17559
  callId,
17070
17560
  resolved: true,
@@ -17074,13 +17564,13 @@ async function createCallSyncHandle(init) {
17074
17564
  startMs,
17075
17565
  (input.endAt ?? input.at).getTime() - init.call.startedAt.getTime()
17076
17566
  );
17077
- void init.client.calls.appendTranscriptSegment(callId, {
17567
+ void track(init.client.calls.appendTranscriptSegment(callId, {
17078
17568
  speaker: input.speaker,
17079
17569
  text: input.text,
17080
17570
  startMs,
17081
17571
  endMs,
17082
17572
  final: input.final
17083
- }).catch((err) => {
17573
+ })).catch((err) => {
17084
17574
  console.warn("[sdk.call-sync] segment publish failed", {
17085
17575
  speaker: input.speaker,
17086
17576
  err: err instanceof Error ? err.message : err
@@ -17088,12 +17578,43 @@ async function createCallSyncHandle(init) {
17088
17578
  });
17089
17579
  },
17090
17580
  syncState(input) {
17091
- void init.client.calls.syncState(callId, input).catch((err) => {
17581
+ void track(init.client.calls.syncState(callId, input)).catch((err) => {
17092
17582
  console.warn("[sdk.call-sync] state sync failed", {
17093
17583
  err: err instanceof Error ? err.message : err
17094
17584
  });
17095
17585
  });
17096
17586
  },
17587
+ publishTurnLatency(payload) {
17588
+ void track(init.client.calls.appendEvent(callId, { kind: "turn.latency", payload: { ...payload } })).catch((err) => {
17589
+ console.warn("[sdk.call-sync] turn latency publish failed", {
17590
+ turn: payload.turn,
17591
+ err: err instanceof Error ? err.message : err
17592
+ });
17593
+ });
17594
+ },
17595
+ track,
17596
+ async finalize(fields) {
17597
+ let timer;
17598
+ try {
17599
+ await Promise.race([
17600
+ Promise.all([...inFlight]),
17601
+ new Promise((resolve) => {
17602
+ timer = setTimeout(resolve, FINALIZE_WRITES_WAIT_MS);
17603
+ })
17604
+ ]);
17605
+ clearTimeout(timer);
17606
+ const limit = deadline(CLOSE_REQUEST_TIMEOUT_MS);
17607
+ try {
17608
+ await init.client.calls.finalize(callId, { fields: [...fields] }, { signal: limit.signal });
17609
+ } finally {
17610
+ limit.clear();
17611
+ }
17612
+ } catch (err) {
17613
+ console.warn("[sdk.call-sync] finalize failed", { err: err instanceof Error ? err.message : err });
17614
+ } finally {
17615
+ clearTimeout(timer);
17616
+ }
17617
+ },
17097
17618
  async recordEndReason(reason) {
17098
17619
  let timer;
17099
17620
  try {
@@ -17113,7 +17634,7 @@ async function createCallSyncHandle(init) {
17113
17634
  }
17114
17635
  },
17115
17636
  publishDtmf(input) {
17116
- void init.client.calls.appendDtmfEvent(callId, input).catch((err) => {
17637
+ void track(init.client.calls.appendDtmfEvent(callId, input)).catch((err) => {
17117
17638
  console.warn("[sdk.call-sync] dtmf publish failed", {
17118
17639
  direction: input.direction,
17119
17640
  digit: input.digit,
@@ -17138,7 +17659,10 @@ async function resolveCallIdWithRetry(init) {
17138
17659
  attributes: initialAttributes(init.call),
17139
17660
  participants: initialParticipants(init.call),
17140
17661
  legs: initialLegs(init.call, init.roomName),
17141
- startedAt: init.call.startedAt.toISOString()
17662
+ startedAt: init.call.startedAt.toISOString(),
17663
+ ...init.pipelineKeys ? { pipelineKeys: init.pipelineKeys } : {},
17664
+ // this worker sends POST /v1/calls/:id/finalize at the close: the API holds call.ended for it (G-20)
17665
+ finalizes: true
17142
17666
  });
17143
17667
  return summary.id;
17144
17668
  } catch {
@@ -17208,7 +17732,7 @@ function initialLegs(call, roomName) {
17208
17732
  startedAt
17209
17733
  }
17210
17734
  ];
17211
- if (destination) {
17735
+ if (destination && call.metadata["direction"] !== "outbound") {
17212
17736
  legs.unshift({
17213
17737
  legKey: "primary-inbound",
17214
17738
  kind: "sip-inbound",
@@ -17228,28 +17752,29 @@ function looksLikeSipParticipant(value) {
17228
17752
 
17229
17753
  // src/runtime/metrics-bridge.ts
17230
17754
  init_dist2();
17755
+ var payerKey = (byok) => byok === void 0 ? "" : byok ? "byok" : "platform";
17231
17756
  var UsageAccumulator = class {
17232
17757
  llm = /* @__PURE__ */ new Map();
17233
17758
  tts = /* @__PURE__ */ new Map();
17234
17759
  stt = /* @__PURE__ */ new Map();
17235
- // `byok` keeps a line whose payer is known (the SDK's helper calls) apart from LiveKit's own line on the same model,
17236
- // which the API prices by the workspace's vault keys.
17237
17760
  addLlm(provider, model2, inputTokens, outputTokens, byok) {
17238
- const key = `${provider}:${model2}:${byok === void 0 ? "" : byok ? "byok" : "platform"}`;
17761
+ const key = `${provider}:${model2}:${payerKey(byok)}`;
17239
17762
  const entry2 = this.llm.get(key) ?? { provider, model: model2, inputTokens: 0, outputTokens: 0, ...byok !== void 0 ? { byok } : {} };
17240
17763
  entry2.inputTokens += Math.max(0, inputTokens || 0);
17241
17764
  entry2.outputTokens += Math.max(0, outputTokens || 0);
17242
17765
  this.llm.set(key, entry2);
17243
17766
  }
17244
- addTts(provider, chars) {
17245
- const entry2 = this.tts.get(provider) ?? { provider, chars: 0 };
17767
+ addTts(provider, chars, byok) {
17768
+ const key = `${provider}:${payerKey(byok)}`;
17769
+ const entry2 = this.tts.get(key) ?? { provider, chars: 0, ...byok !== void 0 ? { byok } : {} };
17246
17770
  entry2.chars += Math.max(0, chars || 0);
17247
- this.tts.set(provider, entry2);
17771
+ this.tts.set(key, entry2);
17248
17772
  }
17249
- addStt(provider, seconds) {
17250
- const entry2 = this.stt.get(provider) ?? { provider, seconds: 0 };
17773
+ addStt(provider, seconds, byok) {
17774
+ const key = `${provider}:${payerKey(byok)}`;
17775
+ const entry2 = this.stt.get(key) ?? { provider, seconds: 0, ...byok !== void 0 ? { byok } : {} };
17251
17776
  entry2.seconds += Math.max(0, seconds || 0);
17252
- this.stt.set(provider, entry2);
17777
+ this.stt.set(key, entry2);
17253
17778
  }
17254
17779
  isEmpty() {
17255
17780
  return this.llm.size === 0 && this.tts.size === 0 && this.stt.size === 0;
@@ -17302,10 +17827,13 @@ function attachEndpointingProbe(session, events, ctx, deps) {
17302
17827
  });
17303
17828
  recordTurnLatencySpan({ ...base, stage: "endpointing", latencyMs: delayMs });
17304
17829
  recordTurnLatencySpan({ ...base, stage: "stt", latencyMs: delayMs });
17830
+ ctx.turns?.note({ stage: "endpointing", latencyMs: delayMs });
17831
+ ctx.turns?.note({ stage: "stt", latencyMs: delayMs });
17305
17832
  };
17306
17833
  deps.vad.onSpeechEnd((sample) => {
17307
17834
  if (!armed) return;
17308
17835
  speechEndAt = sample.speechEndAt;
17836
+ ctx.turns?.noteSpeechEnd(sample.speechEndAt);
17309
17837
  tryEmit();
17310
17838
  });
17311
17839
  session.on(events.userStateChanged, (event) => {
@@ -17347,7 +17875,8 @@ function dispatch(metrics2, ctx) {
17347
17875
  provider,
17348
17876
  ...model2 ? { model: model2 } : {},
17349
17877
  inputTokens: m.promptTokens,
17350
- outputTokens: m.completionTokens
17878
+ outputTokens: m.completionTokens,
17879
+ ...ctx.keyOwner?.llm ? { byok: ctx.keyOwner.llm.byok } : {}
17351
17880
  });
17352
17881
  recordLlmUsage({
17353
17882
  ...base,
@@ -17363,6 +17892,7 @@ function dispatch(metrics2, ctx) {
17363
17892
  } else if (typeof m.ttftMs === "number" && m.ttftMs >= 0) {
17364
17893
  recordTurnLatency({ ...base, firstTokenMs: m.ttftMs });
17365
17894
  recordTurnLatencySpan({ ...base, stage: "llm", latencyMs: m.ttftMs });
17895
+ ctx.turns?.note({ stage: "llm", latencyMs: m.ttftMs, startedAt: m.timestamp - Math.max(0, m.durationMs) });
17366
17896
  }
17367
17897
  break;
17368
17898
  }
@@ -17370,7 +17900,7 @@ function dispatch(metrics2, ctx) {
17370
17900
  const m = metrics2;
17371
17901
  const provider = m.metadata?.modelProvider ?? "unknown";
17372
17902
  const model2 = m.metadata?.modelName;
17373
- ctx.usage?.record({ kind: "tts", provider, chars: m.charactersCount });
17903
+ ctx.usage?.record({ kind: "tts", provider, chars: m.charactersCount, ...ctx.keyOwner?.tts ? { byok: ctx.keyOwner.tts.byok } : {} });
17374
17904
  recordTtsUsage({
17375
17905
  ...base,
17376
17906
  provider,
@@ -17380,6 +17910,7 @@ function dispatch(metrics2, ctx) {
17380
17910
  if (typeof m.ttfbMs === "number" && m.ttfbMs >= 0) {
17381
17911
  recordTurnLatency({ ...base, ttsStartMs: m.ttfbMs });
17382
17912
  recordTurnLatencySpan({ ...base, stage: "tts", latencyMs: m.ttfbMs });
17913
+ ctx.turns?.note({ stage: "tts", latencyMs: m.ttfbMs, startedAt: m.timestamp - Math.max(0, m.durationMs) });
17383
17914
  }
17384
17915
  break;
17385
17916
  }
@@ -17387,7 +17918,12 @@ function dispatch(metrics2, ctx) {
17387
17918
  const m = metrics2;
17388
17919
  const provider = m.metadata?.modelProvider ?? "unknown";
17389
17920
  const model2 = m.metadata?.modelName;
17390
- ctx.usage?.record({ kind: "stt", provider, seconds: (m.audioDurationMs ?? 0) / 1e3 });
17921
+ ctx.usage?.record({
17922
+ kind: "stt",
17923
+ provider,
17924
+ seconds: (m.audioDurationMs ?? 0) / 1e3,
17925
+ ...ctx.keyOwner?.stt ? { byok: ctx.keyOwner.stt.byok } : {}
17926
+ });
17391
17927
  recordSttUsage({
17392
17928
  ...base,
17393
17929
  provider,
@@ -17410,6 +17946,7 @@ function dispatch(metrics2, ctx) {
17410
17946
  if (typeof m.ttftMs === "number" && m.ttftMs >= 0) {
17411
17947
  recordTurnLatency({ ...base, firstTokenMs: m.ttftMs });
17412
17948
  recordTurnLatencySpan({ ...base, stage: "llm", latencyMs: m.ttftMs, realtime: true });
17949
+ ctx.turns?.note({ stage: "llm", latencyMs: m.ttftMs, startedAt: m.timestamp, realtime: true });
17413
17950
  }
17414
17951
  if (typeof m.sessionDurationMs === "number" && m.sessionDurationMs > 0) {
17415
17952
  recordRealtimeSessionDuration({
@@ -17431,6 +17968,11 @@ function dispatch(metrics2, ctx) {
17431
17968
  });
17432
17969
  recordTurnLatencySpan({ ...base, stage: "endpointing", latencyMs: m.endOfUtteranceDelayMs });
17433
17970
  recordTurnLatencySpan({ ...base, stage: "stt", latencyMs: m.transcriptionDelayMs });
17971
+ if (ctx.turns) {
17972
+ if (m.endOfUtteranceDelayMs > 0) ctx.turns.note({ stage: "endpointing", latencyMs: m.endOfUtteranceDelayMs });
17973
+ if (m.transcriptionDelayMs > 0) ctx.turns.note({ stage: "stt", latencyMs: m.transcriptionDelayMs });
17974
+ if (m.lastSpeakingTimeMs > 0) ctx.turns.noteSpeechEnd(m.lastSpeakingTimeMs);
17975
+ }
17434
17976
  break;
17435
17977
  }
17436
17978
  case "interruption_metrics": {
@@ -17468,10 +18010,10 @@ function createTelemetrySink(accumulator = new UsageAccumulator()) {
17468
18010
  );
17469
18011
  break;
17470
18012
  case "tts":
17471
- accumulator.addTts(sample.provider, sample.chars ?? 0);
18013
+ accumulator.addTts(sample.provider, sample.chars ?? 0, sample.byok);
17472
18014
  break;
17473
18015
  case "stt":
17474
- accumulator.addStt(sample.provider, sample.seconds ?? 0);
18016
+ accumulator.addStt(sample.provider, sample.seconds ?? 0, sample.byok);
17475
18017
  break;
17476
18018
  }
17477
18019
  },
@@ -17554,7 +18096,12 @@ async function reportUsageAtClose(opts) {
17554
18096
  try {
17555
18097
  await opts.meter.drain(opts.drainMs ?? HANGUP_DRAIN_MS);
17556
18098
  if (opts.usage.isEmpty()) return;
17557
- await opts.send(opts.usage.report(opts.durationMs));
18099
+ const limit = deadline(CLOSE_REQUEST_TIMEOUT_MS);
18100
+ try {
18101
+ await opts.send(opts.usage.report(opts.durationMs), limit.signal);
18102
+ } finally {
18103
+ limit.clear();
18104
+ }
17558
18105
  } catch (err) {
17559
18106
  opts.warn?.("[agent] usage report failed", { err: err instanceof Error ? err.message : String(err) });
17560
18107
  }
@@ -17603,6 +18150,7 @@ function attachTurnModelClock(session, events, ctx, deps = {}) {
17603
18150
  safely2(() => {
17604
18151
  recordTurnLatencySpan({ ...base, stage: "llm", latencyMs, allowZero: true });
17605
18152
  if (latencyMs > 0) recordTurnLatency({ ...base, firstTokenMs: latencyMs });
18153
+ deps.turns?.note({ stage: "llm", latencyMs, replyAt: t.windowEnd });
17606
18154
  });
17607
18155
  };
17608
18156
  const releaseParked = () => {
@@ -17719,6 +18267,145 @@ function attachTurnModelClock(session, events, ctx, deps = {}) {
17719
18267
  };
17720
18268
  }
17721
18269
 
18270
+ // src/runtime/turn-latency.ts
18271
+ init_src();
18272
+ var MAX_TURN_LATENCY_EVENTS = TURN_LATENCY_MAX_PER_CALL;
18273
+ var MAX_STAGE_MS = 6e5;
18274
+ var noStages = () => ({ endpointingMs: null, sttMs: null, llmMs: null, ttsMs: null, realtime: false });
18275
+ function attachTurnLatencyRecorder(session, events, deps) {
18276
+ const now = deps.now ?? Date.now;
18277
+ const maxTurns = deps.maxTurns ?? MAX_TURN_LATENCY_EVENTS;
18278
+ let waiting = null;
18279
+ let open = null;
18280
+ let turns = 0;
18281
+ let lastReplyAt = -Infinity;
18282
+ let lastSpeechEndAt = null;
18283
+ let closed = false;
18284
+ const ms = (v) => v === null || !Number.isFinite(v) || v < 0 || v > MAX_STAGE_MS ? null : Math.round(v);
18285
+ const emit = (t) => {
18286
+ const endpointingMs = ms(t.endpointingMs);
18287
+ const llmMs = ms(t.llmMs);
18288
+ const ttsMs = ms(t.ttsMs);
18289
+ const measured = t.speechEndAt !== null ? ms(t.replyAt - t.speechEndAt) : null;
18290
+ const summed = endpointingMs !== null && llmMs !== null && ttsMs !== null ? ms(endpointingMs + llmMs + ttsMs) : null;
18291
+ try {
18292
+ deps.emit({
18293
+ turn: t.index,
18294
+ atMs: Math.max(0, Math.round(t.replyAt - deps.callStartedAt)),
18295
+ endpointingMs,
18296
+ sttMs: ms(t.sttMs),
18297
+ llmMs,
18298
+ ttsMs,
18299
+ endToEndMs: t.realtime ? llmMs : measured ?? summed,
18300
+ ...t.realtime ? { realtime: true } : {}
18301
+ });
18302
+ } catch (err) {
18303
+ console.warn("[agent] turn latency emit failed", { err: err instanceof Error ? err.message : String(err) });
18304
+ }
18305
+ };
18306
+ const at = (event) => {
18307
+ const t = event?.createdAt;
18308
+ return typeof t === "number" ? t : now();
18309
+ };
18310
+ const targetFor = (s) => {
18311
+ if (waiting && (s.startedAt === void 0 || s.startedAt >= waiting.firstFinalAt)) return waiting;
18312
+ return open;
18313
+ };
18314
+ if (!deps.realtime) {
18315
+ session.on(events.userInputTranscribed, (event) => {
18316
+ if (closed || !event?.isFinal) return;
18317
+ waiting ??= { ...noStages(), firstFinalAt: at(event) };
18318
+ });
18319
+ session.on(events.agentStateChanged, (event) => {
18320
+ if (closed || event?.newState !== "speaking" || !waiting) return;
18321
+ const replyAt = at(event);
18322
+ if (open) emit(open);
18323
+ open = null;
18324
+ if (turns >= maxTurns) {
18325
+ waiting = null;
18326
+ return;
18327
+ }
18328
+ turns += 1;
18329
+ const speechEndAt = lastSpeechEndAt !== null && lastSpeechEndAt > lastReplyAt && lastSpeechEndAt <= replyAt ? lastSpeechEndAt : null;
18330
+ const { firstFinalAt: _first, ...stages } = waiting;
18331
+ open = { ...stages, index: turns, replyAt, speechEndAt };
18332
+ waiting = null;
18333
+ lastReplyAt = replyAt;
18334
+ });
18335
+ }
18336
+ return {
18337
+ note(s) {
18338
+ if (closed || !Number.isFinite(s.latencyMs) || s.latencyMs < 0) return;
18339
+ if (deps.realtime) {
18340
+ if (s.stage !== "llm" || turns >= maxTurns) return;
18341
+ turns += 1;
18342
+ emit({ ...noStages(), llmMs: s.latencyMs, realtime: true, index: turns, replyAt: (s.startedAt ?? now()) + s.latencyMs, speechEndAt: null });
18343
+ return;
18344
+ }
18345
+ switch (s.stage) {
18346
+ case "endpointing":
18347
+ case "stt": {
18348
+ const key = s.stage === "endpointing" ? "endpointingMs" : "sttMs";
18349
+ if (waiting) waiting[key] = s.latencyMs;
18350
+ else if (open && open[key] === null) open[key] = s.latencyMs;
18351
+ return;
18352
+ }
18353
+ case "llm": {
18354
+ if (s.replyAt !== void 0) {
18355
+ if (open && open.replyAt === s.replyAt) open.llmMs = (open.llmMs ?? 0) + s.latencyMs;
18356
+ return;
18357
+ }
18358
+ const t = targetFor(s);
18359
+ if (t) t.llmMs = (t.llmMs ?? 0) + s.latencyMs;
18360
+ return;
18361
+ }
18362
+ case "tts": {
18363
+ const t = targetFor(s);
18364
+ if (t && t.ttsMs === null) t.ttsMs = s.latencyMs;
18365
+ return;
18366
+ }
18367
+ }
18368
+ },
18369
+ noteSpeechEnd(at2) {
18370
+ if (Number.isFinite(at2) && at2 > 0) lastSpeechEndAt = at2;
18371
+ },
18372
+ close() {
18373
+ if (closed) return;
18374
+ closed = true;
18375
+ if (open) emit(open);
18376
+ open = null;
18377
+ waiting = null;
18378
+ }
18379
+ };
18380
+ }
18381
+
18382
+ // src/runtime/call-outcome.ts
18383
+ init_src();
18384
+ var cut = (s) => s.length > CAPTURED_STRING_MAX_CHARS ? s.slice(0, CAPTURED_STRING_MAX_CHARS) : s;
18385
+ function wireValue(value) {
18386
+ if (value === void 0 || value === null) return null;
18387
+ if (typeof value === "string") return value.trim().length === 0 ? null : cut(value);
18388
+ if (typeof value === "number") return Number.isFinite(value) ? value : null;
18389
+ if (typeof value === "boolean") return value;
18390
+ if (value instanceof Date) return Number.isNaN(value.getTime()) ? null : value.toISOString();
18391
+ if (Array.isArray(value)) {
18392
+ const items = value.filter((v) => v !== void 0 && v !== null).map((v) => cut(typeof v === "string" ? v : JSON.stringify(v))).slice(0, CAPTURED_ARRAY_MAX_ITEMS);
18393
+ return items.length === 0 ? null : items;
18394
+ }
18395
+ try {
18396
+ return cut(JSON.stringify(value));
18397
+ } catch {
18398
+ return null;
18399
+ }
18400
+ }
18401
+ function capturedFields(def, data) {
18402
+ if (!def) return [];
18403
+ return Object.entries(def.collect).filter(([name]) => name.length > 0 && name.length <= 64).slice(0, CAPTURED_FIELDS_MAX).map(([name, field]) => {
18404
+ const value = wireValue(data[name]);
18405
+ return { name, type: field.type, required: field.required === true, filled: value !== null, value };
18406
+ });
18407
+ }
18408
+
17722
18409
  // src/agent.ts
17723
18410
  init_model_meter();
17724
18411
 
@@ -17884,42 +18571,266 @@ function fieldSpecToField(spec) {
17884
18571
  if (spec.required !== void 0) field.required = spec.required;
17885
18572
  if (spec.pattern !== void 0) {
17886
18573
  try {
17887
- field.pattern = new RegExp(spec.pattern);
17888
- } catch {
18574
+ field.pattern = new RegExp(spec.pattern);
18575
+ } catch {
18576
+ }
18577
+ }
18578
+ if (spec.min !== void 0) field.min = spec.min;
18579
+ if (spec.max !== void 0) field.max = spec.max;
18580
+ if (spec.enum !== void 0) field.enum = spec.enum;
18581
+ if (spec.ask !== void 0) field.ask = spec.ask;
18582
+ if (spec.extract !== void 0) field.extract = spec.extract;
18583
+ return field;
18584
+ }
18585
+ function triggerSpecToDefinition(spec) {
18586
+ const condition = conditionSpecToCondition(spec.on);
18587
+ const action = actionSpecToAction(spec.then);
18588
+ if (!condition || !action) return void 0;
18589
+ return {
18590
+ on: condition,
18591
+ then: action,
18592
+ ...spec.onlyWhile ? { onlyWhile: spec.onlyWhile } : {}
18593
+ };
18594
+ }
18595
+ function conditionSpecToCondition(spec) {
18596
+ if (spec.kind === "regex") {
18597
+ try {
18598
+ return new RegExp(spec.pattern, spec.flags);
18599
+ } catch {
18600
+ return void 0;
18601
+ }
18602
+ }
18603
+ return `signal:${spec.name}`;
18604
+ }
18605
+ function actionSpecToAction(spec) {
18606
+ if (spec.kind === "handoff") return "handoff";
18607
+ if (spec.kind === "endCall") return "endCall";
18608
+ if (spec.kind === "say") return { say: spec.text };
18609
+ return void 0;
18610
+ }
18611
+
18612
+ // src/runtime/call-admission.ts
18613
+ init_src();
18614
+
18615
+ // src/end-call.ts
18616
+ var defaultClock = {
18617
+ now: () => Date.now(),
18618
+ setTimeout: (fn, ms) => setTimeout(fn, ms),
18619
+ clearTimeout: (h) => clearTimeout(h)
18620
+ };
18621
+ var HOST_HANGUP_HOOK_LIMIT_MS = 2e3;
18622
+ var GUARANTEE_RETRY_MS = 1500;
18623
+ var GOODBYE_LIMIT_MS = 1e4;
18624
+ var DEFAULT_MAX_CALL_DURATION_MS = 36e5;
18625
+ function createEndCallController(deps) {
18626
+ const clock = deps.clock ?? defaultClock;
18627
+ const log = deps.log ?? (() => {
18628
+ });
18629
+ let ended = false;
18630
+ let idleTimer = null;
18631
+ let durationTimer = null;
18632
+ let warnTimer = null;
18633
+ const clearIdle = () => {
18634
+ if (idleTimer !== null) {
18635
+ clock.clearTimeout(idleTimer);
18636
+ idleTimer = null;
18637
+ }
18638
+ };
18639
+ const clearDuration = () => {
18640
+ if (durationTimer !== null) {
18641
+ clock.clearTimeout(durationTimer);
18642
+ durationTimer = null;
18643
+ }
18644
+ if (warnTimer !== null) {
18645
+ clock.clearTimeout(warnTimer);
18646
+ warnTimer = null;
18647
+ }
18648
+ };
18649
+ const sleep2 = (ms) => new Promise((resolve) => {
18650
+ clock.setTimeout(resolve, ms);
18651
+ });
18652
+ const guarantee = async (teardown = deps.runTeardown) => {
18653
+ if (deps.guaranteeMs === false) return;
18654
+ const deadline2 = clock.now() + deps.guaranteeMs;
18655
+ while (!deps.isEnded() && clock.now() < deadline2) {
18656
+ await sleep2(Math.min(GUARANTEE_RETRY_MS, Math.max(0, deadline2 - clock.now())));
18657
+ if (deps.isEnded()) return;
18658
+ log("[agent] end-call teardown not confirmed; retrying", { teardown: deps.teardown });
18659
+ try {
18660
+ await teardown();
18661
+ } catch (err) {
18662
+ log("[agent] end-call retry teardown threw", { err: String(err) });
18663
+ }
18664
+ }
18665
+ };
18666
+ let hostHangup = null;
18667
+ const hostHangUp = (reason) => {
18668
+ if (deps.isEnded()) return Promise.resolve();
18669
+ hostHangup ??= (async () => {
18670
+ ended = true;
18671
+ clearIdle();
18672
+ clearDuration();
18673
+ const info = { ...reason !== void 0 ? { reason } : {}, trigger: "host", teardown: "room" };
18674
+ deps.emit?.(info);
18675
+ log("[agent] hang up (host)", { reason });
18676
+ if (deps.onEnd) {
18677
+ let limit;
18678
+ await Promise.race([
18679
+ Promise.resolve().then(() => deps.onEnd(info)).catch((err) => log("[agent] endCall.onEnd threw on a host hangup", { err: String(err) })),
18680
+ new Promise((resolve) => {
18681
+ limit = clock.setTimeout(resolve, HOST_HANGUP_HOOK_LIMIT_MS);
18682
+ })
18683
+ ]).finally(() => clock.clearTimeout(limit));
18684
+ }
18685
+ const teardown = deps.runRoomTeardown ?? deps.runTeardown;
18686
+ try {
18687
+ await teardown();
18688
+ } catch (err) {
18689
+ log("[agent] host hangup teardown threw", { err: String(err) });
18690
+ }
18691
+ await guarantee(teardown);
18692
+ if (!deps.isEnded()) hostHangup = null;
18693
+ })();
18694
+ return hostHangup;
18695
+ };
18696
+ const hangUp = async (opts) => {
18697
+ if (opts?.trigger === "host") return hostHangUp(opts.reason);
18698
+ if (ended) return;
18699
+ ended = true;
18700
+ clearIdle();
18701
+ clearDuration();
18702
+ const info = {
18703
+ ...opts?.reason !== void 0 ? { reason: opts.reason } : {},
18704
+ trigger: opts?.trigger ?? "user",
18705
+ teardown: deps.teardown
18706
+ };
18707
+ deps.emit?.(info);
18708
+ log("[agent] hang up", { trigger: info.trigger, teardown: info.teardown });
18709
+ let suppress = false;
18710
+ if (deps.onEnd) {
18711
+ try {
18712
+ suppress = await deps.onEnd(info) === false;
18713
+ } catch (err) {
18714
+ log("[agent] endCall.onEnd threw; running default teardown", { err: String(err) });
18715
+ }
18716
+ }
18717
+ if (suppress) {
18718
+ log("[agent] endCall.onEnd suppressed default teardown", { trigger: info.trigger });
18719
+ return;
18720
+ }
18721
+ try {
18722
+ await deps.runTeardown();
18723
+ } catch (err) {
18724
+ log("[agent] end-call teardown threw", { err: String(err) });
17889
18725
  }
17890
- }
17891
- if (spec.min !== void 0) field.min = spec.min;
17892
- if (spec.max !== void 0) field.max = spec.max;
17893
- if (spec.enum !== void 0) field.enum = spec.enum;
17894
- if (spec.ask !== void 0) field.ask = spec.ask;
17895
- if (spec.extract !== void 0) field.extract = spec.extract;
17896
- return field;
17897
- }
17898
- function triggerSpecToDefinition(spec) {
17899
- const condition = conditionSpecToCondition(spec.on);
17900
- const action = actionSpecToAction(spec.then);
17901
- if (!condition || !action) return void 0;
18726
+ await guarantee();
18727
+ };
17902
18728
  return {
17903
- on: condition,
17904
- then: action,
17905
- ...spec.onlyWhile ? { onlyWhile: spec.onlyWhile } : {}
18729
+ hangUp,
18730
+ armIdle() {
18731
+ if (ended || deps.idleHangupMs === false) return;
18732
+ clearIdle();
18733
+ idleTimer = clock.setTimeout(() => {
18734
+ idleTimer = null;
18735
+ void hangUp({ trigger: "idle", reason: "no activity on the line" });
18736
+ }, deps.idleHangupMs);
18737
+ },
18738
+ cancelIdle() {
18739
+ clearIdle();
18740
+ },
18741
+ armDuration() {
18742
+ if (ended || deps.maxCallDurationMs === false) return;
18743
+ clearDuration();
18744
+ const durationMs = deps.maxCallDurationMs ?? DEFAULT_MAX_CALL_DURATION_MS;
18745
+ const wrapUp = deps.wrapUp;
18746
+ if (wrapUp && wrapUp.warnBeforeMs > 0 && durationMs > wrapUp.warnBeforeMs) {
18747
+ warnTimer = clock.setTimeout(() => {
18748
+ warnTimer = null;
18749
+ if (ended) return;
18750
+ try {
18751
+ void Promise.resolve(wrapUp.warn()).catch((err) => log("[agent] wrap-up line failed", { err: String(err) }));
18752
+ } catch (err) {
18753
+ log("[agent] wrap-up line failed", { err: String(err) });
18754
+ }
18755
+ }, durationMs - wrapUp.warnBeforeMs);
18756
+ }
18757
+ durationTimer = clock.setTimeout(() => {
18758
+ durationTimer = null;
18759
+ void (async () => {
18760
+ if (wrapUp && !ended) {
18761
+ let limit;
18762
+ await Promise.race([
18763
+ wrapUp.goodbye().catch((err) => log("[agent] goodbye line failed", { err: String(err) })),
18764
+ new Promise((resolve) => {
18765
+ limit = clock.setTimeout(resolve, GOODBYE_LIMIT_MS);
18766
+ })
18767
+ ]).finally(() => clock.clearTimeout(limit));
18768
+ }
18769
+ await hangUp({
18770
+ trigger: "max_duration",
18771
+ reason: `call exceeded the ${durationMs}ms duration ceiling`
18772
+ });
18773
+ })();
18774
+ }, durationMs);
18775
+ },
18776
+ noteEngineClosed() {
18777
+ if (ended) return;
18778
+ log("[agent] realtime engine closed; guaranteeing teardown");
18779
+ void hangUp({ trigger: "engine-closed", reason: "realtime engine closed" });
18780
+ },
18781
+ dispose() {
18782
+ clearIdle();
18783
+ clearDuration();
18784
+ }
17906
18785
  };
17907
18786
  }
17908
- function conditionSpecToCondition(spec) {
17909
- if (spec.kind === "regex") {
18787
+
18788
+ // src/runtime/call-admission.ts
18789
+ var CALL_ADMISSION_RETRY_DELAYS_MS = [300, 1e3];
18790
+ var ADMISSION_UNAVAILABLE_END_REASON = `${INBOUND_REFUSED_END_REASON_PREFIX}gate_unavailable`;
18791
+ function notAVerdict(err) {
18792
+ if (err instanceof VoiceLayerAuthError || err instanceof VoiceLayerValidationError) return true;
18793
+ return err instanceof VoiceLayerHttpError && err.status >= 400 && err.status < 500 && err.status !== 408 && err.status !== 429;
18794
+ }
18795
+ async function requestCallAdmission(admit, req, opts = {}) {
18796
+ if (!admit) return { kind: "admitted", limits: null };
18797
+ const delays = opts.delaysMs ?? CALL_ADMISSION_RETRY_DELAYS_MS;
18798
+ const sleep2 = opts.sleep ?? ((ms) => new Promise((r) => setTimeout(r, ms)));
18799
+ const log = opts.log ?? ((msg, attrs2) => console.warn(msg, attrs2 ?? {}));
18800
+ for (let attempt = 1; ; attempt++) {
17910
18801
  try {
17911
- return new RegExp(spec.pattern, spec.flags);
17912
- } catch {
17913
- return void 0;
18802
+ const res = await admit(req);
18803
+ return res.admitted ? { kind: "admitted", limits: res.limits } : { kind: "refused", endReason: res.endReason, message: res.message };
18804
+ } catch (err) {
18805
+ if (notAVerdict(err)) {
18806
+ if (opts.platformWorker) {
18807
+ log("[agent] call admission refused for this room on a platform worker; refusing the call (fail closed)", { err: String(err) });
18808
+ return { kind: "refused", endReason: ADMISSION_UNAVAILABLE_END_REASON, message: INBOUND_REFUSED_MESSAGE };
18809
+ }
18810
+ log("[agent] call admission unavailable for this room; running on the agent settings", { err: String(err) });
18811
+ return { kind: "admitted", limits: null };
18812
+ }
18813
+ if (attempt > delays.length) {
18814
+ log("[agent] call admission failed after retries; refusing the call (fail closed)", { err: String(err), attempts: attempt });
18815
+ return { kind: "refused", endReason: ADMISSION_UNAVAILABLE_END_REASON, message: INBOUND_REFUSED_MESSAGE };
18816
+ }
18817
+ await sleep2(delays[attempt - 1]);
17914
18818
  }
17915
18819
  }
17916
- return `signal:${spec.name}`;
17917
18820
  }
17918
- function actionSpecToAction(spec) {
17919
- if (spec.kind === "handoff") return "handoff";
17920
- if (spec.kind === "endCall") return "endCall";
17921
- if (spec.kind === "say") return { say: spec.text };
17922
- return void 0;
18821
+ function effectiveMaxCallDurationMs(code, limits) {
18822
+ if (!limits) return code === void 0 ? DEFAULT_MAX_CALL_DURATION_MS : code;
18823
+ const ceiling = limits.ceilingSeconds * 1e3;
18824
+ const base = limits.agentMaxSeconds !== null ? limits.agentMaxSeconds * 1e3 : code === void 0 ? DEFAULT_MAX_CALL_DURATION_MS : code;
18825
+ return base === false ? ceiling : Math.min(base, ceiling);
18826
+ }
18827
+ async function playAdmissionRefusal(verdict, deps) {
18828
+ try {
18829
+ await deps.say(verdict.message, { allowInterruptions: false });
18830
+ } catch (err) {
18831
+ (deps.log ?? ((m, a) => console.warn(m, a ?? {})))("[agent] admission refusal line failed to play", { err: String(err) });
18832
+ }
18833
+ await deps.hangUp(verdict.endReason);
17923
18834
  }
17924
18835
  init_src();
17925
18836
  var AskHostInput = z.object({
@@ -18471,7 +19382,7 @@ function asText(value) {
18471
19382
  function masked(value) {
18472
19383
  return maskPii(asText(value).slice(0, READ_CAP));
18473
19384
  }
18474
- var cut = (text, cap) => text.length > cap ? `${text.slice(0, cap)}\u2026` : text;
19385
+ var cut2 = (text, cap) => text.length > cap ? `${text.slice(0, cap)}\u2026` : text;
18475
19386
  function toolAuditPayload(r) {
18476
19387
  const args = masked(r.params);
18477
19388
  const result = r.ok ? masked(r.result) : void 0;
@@ -18480,11 +19391,11 @@ function toolAuditPayload(r) {
18480
19391
  const payload = {
18481
19392
  invocationId: r.invocationId,
18482
19393
  toolName: r.toolName,
18483
- args: cut(args, cap),
18484
- ...result !== void 0 ? { result: cut(result, cap) } : {},
19394
+ args: cut2(args, cap),
19395
+ ...result !== void 0 ? { result: cut2(result, cap) } : {},
18485
19396
  ok: r.ok,
18486
19397
  blocked: r.blocked,
18487
- ...error !== void 0 ? { errorMessage: cut(error, Math.min(cap, 300)) } : {},
19398
+ ...error !== void 0 ? { errorMessage: cut2(error, Math.min(cap, 300)) } : {},
18488
19399
  startedAt: r.startedAt.toISOString(),
18489
19400
  durationMs: r.durationMs,
18490
19401
  // the tool got the caller's values in place of the call's PII tokens (G-33): how many and where — never a value
@@ -18504,151 +19415,8 @@ function toolAuditObserver(emit) {
18504
19415
  };
18505
19416
  }
18506
19417
 
18507
- // src/end-call.ts
18508
- var defaultClock = {
18509
- now: () => Date.now(),
18510
- setTimeout: (fn, ms) => setTimeout(fn, ms),
18511
- clearTimeout: (h) => clearTimeout(h)
18512
- };
18513
- var HOST_HANGUP_HOOK_LIMIT_MS = 2e3;
18514
- var GUARANTEE_RETRY_MS = 1500;
18515
- var DEFAULT_MAX_CALL_DURATION_MS = 36e5;
18516
- function createEndCallController(deps) {
18517
- const clock = deps.clock ?? defaultClock;
18518
- const log = deps.log ?? (() => {
18519
- });
18520
- let ended = false;
18521
- let idleTimer = null;
18522
- let durationTimer = null;
18523
- const clearIdle = () => {
18524
- if (idleTimer !== null) {
18525
- clock.clearTimeout(idleTimer);
18526
- idleTimer = null;
18527
- }
18528
- };
18529
- const clearDuration = () => {
18530
- if (durationTimer !== null) {
18531
- clock.clearTimeout(durationTimer);
18532
- durationTimer = null;
18533
- }
18534
- };
18535
- const sleep2 = (ms) => new Promise((resolve) => {
18536
- clock.setTimeout(resolve, ms);
18537
- });
18538
- const guarantee = async (teardown = deps.runTeardown) => {
18539
- if (deps.guaranteeMs === false) return;
18540
- const deadline = clock.now() + deps.guaranteeMs;
18541
- while (!deps.isEnded() && clock.now() < deadline) {
18542
- await sleep2(Math.min(GUARANTEE_RETRY_MS, Math.max(0, deadline - clock.now())));
18543
- if (deps.isEnded()) return;
18544
- log("[agent] end-call teardown not confirmed; retrying", { teardown: deps.teardown });
18545
- try {
18546
- await teardown();
18547
- } catch (err) {
18548
- log("[agent] end-call retry teardown threw", { err: String(err) });
18549
- }
18550
- }
18551
- };
18552
- let hostHangup = null;
18553
- const hostHangUp = (reason) => {
18554
- if (deps.isEnded()) return Promise.resolve();
18555
- hostHangup ??= (async () => {
18556
- ended = true;
18557
- clearIdle();
18558
- clearDuration();
18559
- const info = { ...reason !== void 0 ? { reason } : {}, trigger: "host", teardown: "room" };
18560
- deps.emit?.(info);
18561
- log("[agent] hang up (host)", { reason });
18562
- if (deps.onEnd) {
18563
- let limit;
18564
- await Promise.race([
18565
- Promise.resolve().then(() => deps.onEnd(info)).catch((err) => log("[agent] endCall.onEnd threw on a host hangup", { err: String(err) })),
18566
- new Promise((resolve) => {
18567
- limit = clock.setTimeout(resolve, HOST_HANGUP_HOOK_LIMIT_MS);
18568
- })
18569
- ]).finally(() => clock.clearTimeout(limit));
18570
- }
18571
- const teardown = deps.runRoomTeardown ?? deps.runTeardown;
18572
- try {
18573
- await teardown();
18574
- } catch (err) {
18575
- log("[agent] host hangup teardown threw", { err: String(err) });
18576
- }
18577
- await guarantee(teardown);
18578
- if (!deps.isEnded()) hostHangup = null;
18579
- })();
18580
- return hostHangup;
18581
- };
18582
- const hangUp = async (opts) => {
18583
- if (opts?.trigger === "host") return hostHangUp(opts.reason);
18584
- if (ended) return;
18585
- ended = true;
18586
- clearIdle();
18587
- clearDuration();
18588
- const info = {
18589
- ...opts?.reason !== void 0 ? { reason: opts.reason } : {},
18590
- trigger: opts?.trigger ?? "user",
18591
- teardown: deps.teardown
18592
- };
18593
- deps.emit?.(info);
18594
- log("[agent] hang up", { trigger: info.trigger, teardown: info.teardown });
18595
- let suppress = false;
18596
- if (deps.onEnd) {
18597
- try {
18598
- suppress = await deps.onEnd(info) === false;
18599
- } catch (err) {
18600
- log("[agent] endCall.onEnd threw; running default teardown", { err: String(err) });
18601
- }
18602
- }
18603
- if (suppress) {
18604
- log("[agent] endCall.onEnd suppressed default teardown", { trigger: info.trigger });
18605
- return;
18606
- }
18607
- try {
18608
- await deps.runTeardown();
18609
- } catch (err) {
18610
- log("[agent] end-call teardown threw", { err: String(err) });
18611
- }
18612
- await guarantee();
18613
- };
18614
- return {
18615
- hangUp,
18616
- armIdle() {
18617
- if (ended || deps.idleHangupMs === false) return;
18618
- clearIdle();
18619
- idleTimer = clock.setTimeout(() => {
18620
- idleTimer = null;
18621
- void hangUp({ trigger: "idle", reason: "no activity on the line" });
18622
- }, deps.idleHangupMs);
18623
- },
18624
- cancelIdle() {
18625
- clearIdle();
18626
- },
18627
- armDuration() {
18628
- if (ended || deps.maxCallDurationMs === false) return;
18629
- clearDuration();
18630
- const durationMs = deps.maxCallDurationMs ?? DEFAULT_MAX_CALL_DURATION_MS;
18631
- durationTimer = clock.setTimeout(() => {
18632
- durationTimer = null;
18633
- void hangUp({
18634
- trigger: "max_duration",
18635
- reason: `call exceeded the ${durationMs}ms duration ceiling`
18636
- });
18637
- }, durationMs);
18638
- },
18639
- noteEngineClosed() {
18640
- if (ended) return;
18641
- log("[agent] realtime engine closed; guaranteeing teardown");
18642
- void hangUp({ trigger: "engine-closed", reason: "realtime engine closed" });
18643
- },
18644
- dispose() {
18645
- clearIdle();
18646
- clearDuration();
18647
- }
18648
- };
18649
- }
18650
-
18651
19418
  // src/agent.ts
19419
+ init_resilient_llm();
18652
19420
  init_helper_models();
18653
19421
  var FLOW_HANGUP_GRACE_MS = Number(process.env["VL_FLOW_HANGUP_GRACE_MS"] ?? "1200");
18654
19422
  function isLiveKitChildProcess() {
@@ -18962,6 +19730,7 @@ var Agent = class {
18962
19730
  const built = await resolveCallPipeline(effectiveConfig.models ?? {}, callInfo, job.proc.userData.vad);
18963
19731
  const pipeline = built.pipeline;
18964
19732
  if (built.failure && !flowBootFailure) flowBootFailure = built.failure;
19733
+ const keyOwner = pipelineKeys(pipeline, callCredential?.source !== "worker-token");
18965
19734
  const vadTap = tapVad(pipeline.vad, { endOfSpeechType: VADEventType.END_OF_SPEECH });
18966
19735
  const usingRealtime = pipeline.realtime != null;
18967
19736
  let runtimeLlm = usingRealtime ? pipeline.realtime : pipeline.llm;
@@ -18974,11 +19743,15 @@ var Agent = class {
18974
19743
  seedCallPrefill(processRt, dispatchMdEarly);
18975
19744
  await seedProjectVariables(processRt, callCredential);
18976
19745
  let opEventCallId = null;
19746
+ let trackWrite = (write) => write;
18977
19747
  const emitOpEvent = (kind, payload) => {
18978
19748
  if (!sdkClient || !opEventCallId) return;
18979
- void sdkClient.calls.appendEvent(opEventCallId, { kind, payload }).catch(() => {
19749
+ void trackWrite(sdkClient.calls.appendEvent(opEventCallId, { kind, payload })).catch(() => {
18980
19750
  });
18981
19751
  };
19752
+ if (isResilientLLM(pipeline.llm)) {
19753
+ pipeline.llm.onIncident((incident) => emitOpEvent(`engine.llm.${incident.kind}`, { ...incident }));
19754
+ }
18982
19755
  const graphEvents = graphMode ? withOpEventFanout(createGraphEvents(processRt), emitOpEvent) : null;
18983
19756
  const isOutbound = stringValue(dispatchMdEarly["direction"]) === "outbound";
18984
19757
  const perCallAmd = stringValue(dispatchMdEarly["amd"]);
@@ -19003,8 +19776,23 @@ var Agent = class {
19003
19776
  const callSync = await createCallSyncHandle({
19004
19777
  client: sdkClient,
19005
19778
  roomName: job.room.name ?? `room-${job.job?.id ?? "unknown"}`,
19006
- call: callInfo
19779
+ call: callInfo,
19780
+ pipelineKeys: keyOwner
19007
19781
  });
19782
+ trackWrite = callSync.track;
19783
+ const callerIsPhone = isSipParticipant2(participant);
19784
+ const callAdmission = await requestCallAdmission(
19785
+ sdkClient ? (req) => sdkClient.calls.admit(req) : null,
19786
+ {
19787
+ roomName: job.room.name ?? `room-${job.job?.id ?? "unknown"}`,
19788
+ ...stringValue(callInfo.metadata["agentId"]) ? { agentId: stringValue(callInfo.metadata["agentId"]) } : {},
19789
+ ...stringValue(callInfo.metadata["phoneNumberId"]) ? { phoneNumberId: stringValue(callInfo.metadata["phoneNumberId"]) } : {},
19790
+ caller: { channel: callerIsPhone ? "phone" : "web", number: callerIsPhone ? callInfo.callerId : null }
19791
+ },
19792
+ // our pool's per-call worker token: no "no verdict" — a missing row is a refusal, never an unbilled call
19793
+ { platformWorker: callCredential?.source === "worker-token" }
19794
+ );
19795
+ const admissionLimits = callAdmission.kind === "admitted" ? callAdmission.limits : null;
19008
19796
  const transcriptLines = [];
19009
19797
  const transcript = createTranscriptHandle(() => transcriptLines);
19010
19798
  const ctxRef = { current: null };
@@ -19136,7 +19924,8 @@ ${callIntent}`
19136
19924
  const endCallTeardown = endCallPolicy.teardown ?? "room";
19137
19925
  const endCallGuaranteeMs = endCallPolicy.guaranteeMs === void 0 ? 4e3 : endCallPolicy.guaranteeMs;
19138
19926
  const endCallIdleMs = endCallPolicy.idleHangupMs === void 0 ? 2e4 : endCallPolicy.idleHangupMs;
19139
- const endCallMaxDurationMs = endCallPolicy.maxCallDurationMs === void 0 ? DEFAULT_MAX_CALL_DURATION_MS : endCallPolicy.maxCallDurationMs;
19927
+ const endCallMaxDurationMs = effectiveMaxCallDurationMs(endCallPolicy.maxCallDurationMs, admissionLimits);
19928
+ const lineRef = { say: null };
19140
19929
  const endCallRoomName = job.room.name ?? `room-${job.job?.id ?? "unknown"}`;
19141
19930
  const localDisconnect = async () => {
19142
19931
  try {
@@ -19149,6 +19938,17 @@ ${callIntent}`
19149
19938
  guaranteeMs: endCallGuaranteeMs,
19150
19939
  idleHangupMs: endCallIdleMs,
19151
19940
  maxCallDurationMs: endCallMaxDurationMs,
19941
+ ...admissionLimits ? {
19942
+ wrapUp: {
19943
+ warnBeforeMs: admissionLimits.wrapUpWarningSeconds * 1e3,
19944
+ warn: () => lineRef.say?.(admissionLimits.wrapUpMessage),
19945
+ goodbye: async () => {
19946
+ const reported = callSync.recordEndReason("max_duration");
19947
+ await lineRef.say?.(admissionLimits.goodbyeMessage);
19948
+ await reported;
19949
+ }
19950
+ }
19951
+ } : {},
19152
19952
  // 'room' deletes the whole LK room (drops the SIP/PSTN leg → carrier BYE);
19153
19953
  // 'agent' only disconnects the agent participant.
19154
19954
  runTeardown: endCallTeardown === "room" ? () => endRoomBestEffort(endCallRoomName, localDisconnect) : localDisconnect,
@@ -19187,6 +19987,7 @@ ${callIntent}`
19187
19987
  }
19188
19988
  });
19189
19989
  const session = adapter.raw.session;
19990
+ lineRef.say = (text) => adapter.say(text, { allowInterruptions: false });
19190
19991
  sessionRef.current = session;
19191
19992
  const handoffRt = createHandoffRuntime(effectiveConfig.handoff, {
19192
19993
  session,
@@ -19423,6 +20224,7 @@ ${callIntent}`
19423
20224
  })
19424
20225
  });
19425
20226
  let usageReported = Promise.resolve();
20227
+ let finalized = Promise.resolve();
19426
20228
  session.once(Events.Close, () => {
19427
20229
  sessionClosed = true;
19428
20230
  endCallController.dispose();
@@ -19440,7 +20242,7 @@ ${callIntent}`
19440
20242
  meter: modelMeter,
19441
20243
  usage: usageSink,
19442
20244
  durationMs: outcome.durationMs,
19443
- send: (report) => calls.reportUsage(callSync.callId, report),
20245
+ send: (report, signal) => calls.reportUsage(callSync.callId, report, { signal }),
19444
20246
  warn: (message, meta) => console.warn(message, { callId: callSync.callId, ...meta })
19445
20247
  });
19446
20248
  }
@@ -19580,6 +20382,11 @@ ${callIntent}`
19580
20382
  callId: callSync.callId,
19581
20383
  ...callCampaignId ? { campaignId: callCampaignId } : {}
19582
20384
  };
20385
+ const turnLatency = attachTurnLatencyRecorder(
20386
+ session,
20387
+ { userInputTranscribed: Events.UserInputTranscribed, agentStateChanged: Events.AgentStateChanged },
20388
+ { callStartedAt: callInfo.startedAt.getTime(), emit: (payload) => callSync.publishTurnLatency(payload), realtime: usingRealtime }
20389
+ );
19583
20390
  const modelClock = graphMode && !usingRealtime ? attachTurnModelClock(
19584
20391
  session,
19585
20392
  {
@@ -19588,13 +20395,26 @@ ${callIntent}`
19588
20395
  agentStateChanged: Events.AgentStateChanged,
19589
20396
  speechCreated: Events.SpeechCreated
19590
20397
  },
19591
- metricsCtx
20398
+ metricsCtx,
20399
+ { turns: turnLatency }
19592
20400
  ) : void 0;
19593
20401
  if (modelClock) session.once(Events.Close, () => modelClock.close());
20402
+ session.once(Events.Close, () => {
20403
+ turnLatency.close();
20404
+ finalized = (async () => {
20405
+ await modelMeter.drain(HANGUP_DRAIN_MS);
20406
+ await new Promise((resolve) => setImmediate(resolve));
20407
+ await callSync.finalize(capturedFields(effectiveConfig.process, processRt.getData()));
20408
+ })().catch((err) => {
20409
+ console.warn("[agent] finalize failed", { callId: callSync.callId, err: err instanceof Error ? err.message : String(err) });
20410
+ });
20411
+ });
19594
20412
  modelMeter.bind(metricsCtx, modelClock);
19595
20413
  attachMetricsBridge(session, Events.MetricsCollected, {
19596
20414
  ...metricsCtx,
19597
20415
  usage: usageSink,
20416
+ keyOwner,
20417
+ turns: turnLatency,
19598
20418
  ...modelClock ? { modelClock } : {}
19599
20419
  });
19600
20420
  if (graphMode) {
@@ -19604,7 +20424,7 @@ ${callIntent}`
19604
20424
  userStateChanged: Events.UserStateChanged,
19605
20425
  userInputTranscribed: Events.UserInputTranscribed
19606
20426
  },
19607
- metricsCtx,
20427
+ { ...metricsCtx, turns: turnLatency },
19608
20428
  { vad: vadTap }
19609
20429
  );
19610
20430
  }
@@ -19667,11 +20487,11 @@ ${callIntent}`
19667
20487
  room: job.room.name ?? null,
19668
20488
  participant: participant.identity
19669
20489
  });
19670
- endCallController.armDuration();
20490
+ if (callAdmission.kind !== "refused") endCallController.armDuration();
19671
20491
  {
19672
20492
  const announceMd = parseJobMetadata(job.job?.metadata);
19673
20493
  const announcement = stringValue(announceMd["recordingAnnouncement"]);
19674
- if (announcement) {
20494
+ if (announcement && callAdmission.kind !== "refused") {
19675
20495
  try {
19676
20496
  await adapter.say(announcement, { allowInterruptions: false });
19677
20497
  } catch (err) {
@@ -19679,7 +20499,13 @@ ${callIntent}`
19679
20499
  }
19680
20500
  }
19681
20501
  }
19682
- if (puppetOwnsOpening(this.config.puppetMode, flowBootFailure)) {
20502
+ if (callAdmission.kind === "refused") {
20503
+ console.warn("[agent] call refused at admission", { room: job.room.name ?? null, endReason: callAdmission.endReason });
20504
+ await playAdmissionRefusal(callAdmission, {
20505
+ say: (text, opts) => adapter.say(text, opts),
20506
+ hangUp: (reason) => endCallController.hangUp({ trigger: "admission", reason })
20507
+ });
20508
+ } else if (puppetOwnsOpening(this.config.puppetMode, flowBootFailure)) {
19683
20509
  const dispatchMd = parseJobMetadata(job.job?.metadata);
19684
20510
  const initialDirective = stringValue(dispatchMd["initialDirective"]);
19685
20511
  if (initialDirective) {
@@ -19771,6 +20597,7 @@ ${callIntent}`
19771
20597
  }
19772
20598
  await waitForSessionClose;
19773
20599
  await usageReported;
20600
+ await finalized;
19774
20601
  } catch (err) {
19775
20602
  console.error("[agent] runJob failed", {
19776
20603
  room: job.room.name ?? null,
@@ -19880,7 +20707,7 @@ function sharedBrainPubSub() {
19880
20707
  async function applyStoredPipeline(client, agentId, base, codeMode) {
19881
20708
  try {
19882
20709
  const stored = await client.agents.getConfig(agentId);
19883
- if (!stored) return { config: base, brain: "none", byokProviders: [] };
20710
+ if (!stored) return { config: base, brain: "none" };
19884
20711
  const mode = codeMode ?? stored.mode;
19885
20712
  const providers = /* @__PURE__ */ new Set();
19886
20713
  if (mode === "realtime" && stored.realtime) {
@@ -19955,7 +20782,6 @@ async function applyStoredPipeline(client, agentId, base, codeMode) {
19955
20782
  models: { ...base.models ?? {}, ...mapped, ...brainLlm ? { llm: brainLlm } : {} }
19956
20783
  },
19957
20784
  brain: brainLlm ? "bound" : brainConnectorId && mode !== "realtime" ? "unreachable" : "none",
19958
- byokProviders: Object.keys(creds2),
19959
20785
  ...mode !== "realtime" && stored.model ? { llm: { provider: stored.model.provider, model: stored.model.model } } : {}
19960
20786
  };
19961
20787
  } catch (err) {
@@ -19963,7 +20789,7 @@ async function applyStoredPipeline(client, agentId, base, codeMode) {
19963
20789
  agentId,
19964
20790
  err: err instanceof Error ? err.message : String(err)
19965
20791
  });
19966
- return { config: base, brain: "unknown", byokProviders: [] };
20792
+ return { config: base, brain: "unknown" };
19967
20793
  }
19968
20794
  }
19969
20795
  function firstString(...values) {