@voicelayer/sdk 0.6.1 → 0.6.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -14,6 +14,7 @@ import '@opentelemetry/sdk-logs';
14
14
  import '@opentelemetry/sdk-metrics';
15
15
  import '@opentelemetry/sdk-node';
16
16
  import '@opentelemetry/semantic-conventions';
17
+ import { llm, DEFAULT_API_CONNECT_OPTIONS, intervalForRetry, APIStatusError, APITimeoutError, APIConnectionError } from '@livekit/agents';
17
18
  import http from 'http';
18
19
  import https from 'https';
19
20
  import { Readable } from 'stream';
@@ -693,7 +694,7 @@ var init_consultation = __esm({
693
694
  });
694
695
  }
695
696
  });
696
- var CallEventKind, OP_CALL_EVENT_KIND, OpCallEventKind, commandId, SayCommand, HangupCommand, DtmfCommand, InstructCommand, InjectContextCommand;
697
+ var CallEventKind, OP_CALL_EVENT_KIND, OpCallEventKind, LatencyMs, commandId, SayCommand, HangupCommand, DtmfCommand, InstructCommand, InjectContextCommand;
697
698
  var init_call_events = __esm({
698
699
  "../contracts/src/call-events.ts"() {
699
700
  init_consultation();
@@ -719,7 +720,9 @@ var init_call_events = __esm({
719
720
  "dtmf.sent",
720
721
  // A host's runtime intervention (burn-down G-6): call_say / call_send_guidance / call_inject_context / call_instruct,
721
722
  // with masked args and the API key — written by the API only, never through the worker's /events endpoint.
722
- "mcp.interaction"
723
+ "mcp.interaction",
724
+ // One per caller turn the agent answered (burn-down G-44): where that turn's wait went — see TurnLatencyPayload.
725
+ "turn.latency"
723
726
  ]);
724
727
  OP_CALL_EVENT_KIND = /^(tool|handoff|lookup|record|notify|engine)\.[a-z0-9_]{1,48}(\.[a-z0-9_]{1,48})?$/;
725
728
  OpCallEventKind = z.string().regex(OP_CALL_EVENT_KIND);
@@ -746,6 +749,17 @@ var init_call_events = __esm({
746
749
  code: z.number().int().min(0).max(15),
747
750
  participantId: z.string().optional()
748
751
  });
752
+ LatencyMs = z.number().int().min(0).max(6e5).nullable();
753
+ z.object({
754
+ turn: z.number().int().min(1),
755
+ atMs: z.number().int().min(0),
756
+ endpointingMs: LatencyMs,
757
+ sttMs: LatencyMs,
758
+ llmMs: LatencyMs,
759
+ ttsMs: LatencyMs,
760
+ endToEndMs: LatencyMs,
761
+ realtime: z.boolean().optional()
762
+ });
749
763
  z.object({
750
764
  totalUsd: z.string(),
751
765
  breakdown: z.object({
@@ -862,9 +876,65 @@ var init_text_turns = __esm({
862
876
  ]);
863
877
  }
864
878
  });
879
+ var PLATFORM_MAX_CALL_MINUTES, Message, CallerRateLimit, CallLimits, CallerChannel, CallDurationLimits;
880
+ var init_call_limits = __esm({
881
+ "../contracts/src/call-limits.ts"() {
882
+ PLATFORM_MAX_CALL_MINUTES = 240;
883
+ Message = z.string().trim().min(1).max(300);
884
+ CallerRateLimit = z.object({
885
+ enabled: z.boolean().optional(),
886
+ perTenMinutes: z.number().int().min(1).max(100).optional(),
887
+ perDay: z.number().int().min(1).max(1e3).optional(),
888
+ withheldPerTenMinutes: z.number().int().min(1).max(100).optional(),
889
+ withheldPerDay: z.number().int().min(1).max(1e3).optional()
890
+ });
891
+ CallLimits = z.object({
892
+ /** The agent's own ceiling in minutes; clamped to the workspace's plan ceiling. Absent ⇒ DEFAULT_MAX_CALL_MINUTES. */
893
+ maxDurationMinutes: z.number().int().min(1).max(PLATFORM_MAX_CALL_MINUTES).optional(),
894
+ /** Seconds before the limit the wrap-up line is spoken; 0 ⇒ no warning. */
895
+ wrapUpWarningSeconds: z.number().int().min(0).max(300).optional(),
896
+ wrapUpMessage: Message.optional(),
897
+ goodbyeMessage: Message.optional(),
898
+ callerRateLimit: CallerRateLimit.optional(),
899
+ rateLimitedMessage: Message.optional()
900
+ });
901
+ CallerChannel = z.enum(["phone", "web"]);
902
+ z.object({
903
+ roomName: z.string().min(1).max(256),
904
+ /** From the dispatch metadata: lets the API admit (gate) a room whose call row doesn't exist yet. */
905
+ agentId: z.string().max(128).optional(),
906
+ phoneNumberId: z.string().max(128).optional(),
907
+ caller: z.object({
908
+ channel: CallerChannel,
909
+ /** The caller's number as the carrier presented it; null / absent / not E.164 ⇒ the withheld bucket. */
910
+ number: z.string().max(64).nullable().optional()
911
+ })
912
+ });
913
+ CallDurationLimits = z.object({
914
+ /** The agent's own ceiling (already clamped to the plan), or null when it set none. */
915
+ agentMaxSeconds: z.number().int().min(1).nullable(),
916
+ /** The plan ceiling — the platform bound no call outlives. */
917
+ ceilingSeconds: z.number().int().min(1),
918
+ wrapUpWarningSeconds: z.number().int().min(0),
919
+ wrapUpMessage: z.string(),
920
+ goodbyeMessage: z.string()
921
+ });
922
+ z.discriminatedUnion("admitted", [
923
+ z.object({ admitted: z.literal(true), limits: CallDurationLimits }),
924
+ z.object({
925
+ admitted: z.literal(false),
926
+ /** 'rate_limited:caller' | 'inbound_refused:<code>' — already recorded as the call's end reason. */
927
+ endReason: z.string(),
928
+ /** The fixed line to speak before hanging up. */
929
+ message: z.string()
930
+ })
931
+ ]);
932
+ }
933
+ });
865
934
  var VoiceConfig, ModelConfig, SttConfig, RealtimeConfig, PipelineMode, ToolBinding, ProcessFieldType, ProcessFieldSpec, ProcessCompleteWhen, ProcessBackendAckSpec, ProcessSpec, TriggerConditionSpec, TriggerActionSpec, TriggerSpec, RequiredInfoField, AgentConfig;
866
935
  var init_agent_config = __esm({
867
936
  "../contracts/src/agent-config.ts"() {
937
+ init_call_limits();
868
938
  init_consultation();
869
939
  VoiceConfig = z.object({
870
940
  // Finite, vetted list — platform-level enum.
@@ -1028,6 +1098,9 @@ var init_agent_config = __esm({
1028
1098
  // LLM slot through the connector instead of a first-party model. Null/absent →
1029
1099
  // first-party pipeline (unchanged default). See docs/specs/voice-brain-connector.
1030
1100
  brainConnectorId: z.string().uuid().nullable().optional(),
1101
+ // Per-agent call bounds: caller rate limit, max duration and their spoken lines (burn-down G-41, G-43). Absent ⇒
1102
+ // every default (call-limits.ts) — the rate limit is ON by default.
1103
+ callLimits: CallLimits.optional(),
1031
1104
  // Provenance: 'sdk' (hand-coded defineAgent worker), 'flow' (canvas deploy),
1032
1105
  // 'playbook' (agent-mode deploy), or 'connector' (brain-connector agent).
1033
1106
  // DERIVED from the linked agents row's deploy tags (flowId / playbookId /
@@ -1341,6 +1414,7 @@ var init_agent_versions = __esm({
1341
1414
  "../contracts/src/agent-versions.ts"() {
1342
1415
  init_agent_config();
1343
1416
  init_agents();
1417
+ init_call_limits();
1344
1418
  init_consultation();
1345
1419
  z.object({
1346
1420
  systemPrompt: z.string().max(64e3),
@@ -1355,6 +1429,8 @@ var init_agent_versions = __esm({
1355
1429
  consultation: ConsultationPolicy,
1356
1430
  tags: z.array(z.string().max(64)).max(32).nullable(),
1357
1431
  brainConnectorId: z.string().uuid().nullable(),
1432
+ // Versions published before call limits existed carry none: read as null (every default).
1433
+ callLimits: CallLimits.nullable().default(null),
1358
1434
  flowVersion: z.number().int().nullable(),
1359
1435
  processSchema: ProcessSchemaDTO.nullable()
1360
1436
  });
@@ -1390,6 +1466,7 @@ var init_agent_versions = __esm({
1390
1466
  "language",
1391
1467
  "tools",
1392
1468
  "brainConnectorId",
1469
+ "callLimits",
1393
1470
  "consultation",
1394
1471
  "tags",
1395
1472
  "flowVersion",
@@ -1531,6 +1608,9 @@ var init_model_catalog = __esm({
1531
1608
  CostEstimateBilledPerMinute = z.object({
1532
1609
  platformFee: z.number().nonnegative(),
1533
1610
  providerPassthrough: z.number().nonnegative(),
1611
+ // The flat own-key fee per minute (burn-down B-26); 0 when every provider runs on the platform's keys. Optional
1612
+ // only for responses from an API older than the fee.
1613
+ ownKeyFee: z.number().nonnegative().optional(),
1534
1614
  telephony: z.number().nonnegative(),
1535
1615
  allIn: z.number().nonnegative()
1536
1616
  });
@@ -1760,9 +1840,46 @@ var init_call_review = __esm({
1760
1840
  });
1761
1841
  }
1762
1842
  });
1843
+ var CAPTURED_FIELDS_MAX, CAPTURED_STRING_MAX_CHARS, CAPTURED_ARRAY_MAX_ITEMS, CapturedFieldValue, CapturedField, StructuredOutputs;
1844
+ var init_call_outcome = __esm({
1845
+ "../contracts/src/call-outcome.ts"() {
1846
+ init_agent_config();
1847
+ CAPTURED_FIELDS_MAX = 200;
1848
+ CAPTURED_STRING_MAX_CHARS = 8e3;
1849
+ CAPTURED_ARRAY_MAX_ITEMS = 100;
1850
+ CapturedFieldValue = z.union([
1851
+ z.string().max(CAPTURED_STRING_MAX_CHARS),
1852
+ z.number().finite(),
1853
+ z.boolean(),
1854
+ z.array(z.string().max(CAPTURED_STRING_MAX_CHARS)).max(CAPTURED_ARRAY_MAX_ITEMS),
1855
+ z.null()
1856
+ ]);
1857
+ CapturedField = z.object({
1858
+ name: z.string().min(1).max(64),
1859
+ type: ProcessFieldType,
1860
+ required: z.boolean(),
1861
+ // false ⇒ the call never captured it; `value` is then null
1862
+ filled: z.boolean(),
1863
+ value: CapturedFieldValue
1864
+ });
1865
+ StructuredOutputs = z.record(CapturedFieldValue);
1866
+ z.object({
1867
+ // Every declared field, in declaration order; empty when the agent declares none.
1868
+ fields: z.array(CapturedField).max(CAPTURED_FIELDS_MAX)
1869
+ }).superRefine((body, ctx) => {
1870
+ const seen = /* @__PURE__ */ new Set();
1871
+ for (const f of body.fields) {
1872
+ if (seen.has(f.name)) ctx.addIssue({ code: z.ZodIssueCode.custom, message: `duplicate field ${f.name}` });
1873
+ seen.add(f.name);
1874
+ if (!f.filled && f.value !== null) ctx.addIssue({ code: z.ZodIssueCode.custom, message: `${f.name}: an unfilled field has no value` });
1875
+ }
1876
+ });
1877
+ }
1878
+ });
1763
1879
  var TranscriptTurn, ConsultationAudit, ToolCallAudit, McpInteractionAudit;
1764
1880
  var init_call_result = __esm({
1765
1881
  "../contracts/src/call-result.ts"() {
1882
+ init_call_outcome();
1766
1883
  TranscriptTurn = z.object({
1767
1884
  speaker: z.enum(["caller", "agent"]),
1768
1885
  text: z.string(),
@@ -1811,9 +1928,13 @@ var init_call_result = __esm({
1811
1928
  startedAt: z.string().datetime(),
1812
1929
  endedAt: z.string().datetime(),
1813
1930
  durationMs: z.number().int().min(0),
1931
+ // Final segments only, in call order.
1814
1932
  transcript: z.array(TranscriptTurn),
1815
- // Whatever the agent's process-schema captured. No prescribed shape.
1816
- structuredOutputs: z.record(z.unknown()).optional(),
1933
+ // The filled fields of the agent's process / flow, keyed by name (burn-down G-20). Absent when the agent declares no
1934
+ // fields, or its worker never reported them.
1935
+ structuredOutputs: StructuredOutputs.optional(),
1936
+ // Every declared field with what the call captured and whether it was filled — the same list `call.ended` carries.
1937
+ fields: z.array(CapturedField).optional(),
1817
1938
  recordingUrl: z.string().url().optional(),
1818
1939
  cost: z.object({
1819
1940
  totalUsd: z.string(),
@@ -1825,9 +1946,11 @@ var init_call_result = __esm({
1825
1946
  });
1826
1947
  }
1827
1948
  });
1828
- var WEBHOOK_EVENT_TYPES, CallWebhookDirection, CallWebhookOutcome, CallWebhookAttributes, Party;
1949
+ var WEBHOOK_EVENT_TYPES, CallWebhookDirection, CallWebhookOutcome, CallWebhookAttributes, Party, CALL_ENDED_PAYLOAD_VERSION, CallEndedTranscript;
1829
1950
  var init_webhook_events = __esm({
1830
1951
  "../contracts/src/webhook-events.ts"() {
1952
+ init_call_result();
1953
+ init_call_outcome();
1831
1954
  WEBHOOK_EVENT_TYPES = ["call.started", "call.ended", "recording.ready"];
1832
1955
  z.enum(WEBHOOK_EVENT_TYPES);
1833
1956
  CallWebhookDirection = z.enum(["inbound", "outbound", "web"]);
@@ -1845,7 +1968,18 @@ var init_webhook_events = __esm({
1845
1968
  // set, it is the same value `call.ended` carries; `call.ended` always has one.
1846
1969
  direction: CallWebhookDirection.nullable()
1847
1970
  });
1971
+ CALL_ENDED_PAYLOAD_VERSION = 2;
1972
+ CallEndedTranscript = z.object({
1973
+ // Final segments only (what was said, never an interim STT guess), in call order. Empty when `truncated`.
1974
+ turns: z.array(TranscriptTurn),
1975
+ // How many final turns the call has — the length of `turns` unless truncated.
1976
+ turnCount: z.number().int().min(0),
1977
+ truncated: z.boolean(),
1978
+ // Where the whole transcript is: GET /v1/calls/:callId/result (its `transcript`). Always set.
1979
+ fetchPath: z.string().regex(/^\/v1\/calls\/[0-9a-f-]{36}\/result$/)
1980
+ });
1848
1981
  z.object({
1982
+ version: z.literal(CALL_ENDED_PAYLOAD_VERSION),
1849
1983
  callId: z.string().uuid(),
1850
1984
  durationMs: z.number().int().min(0),
1851
1985
  endedAt: z.string().datetime(),
@@ -1856,7 +1990,20 @@ var init_webhook_events = __esm({
1856
1990
  toE164: Party,
1857
1991
  direction: CallWebhookDirection,
1858
1992
  outcome: CallWebhookOutcome,
1859
- attributes: CallWebhookAttributes
1993
+ attributes: CallWebhookAttributes,
1994
+ // Every field the agent declared (its process, or a flow's slots), in declaration order, with what the call captured
1995
+ // and whether it was filled. Empty when the agent declares none.
1996
+ fields: z.array(CapturedField),
1997
+ // The filled fields as one object keyed by name (as GET /v1/calls/:id/result returns them). Absent when the agent
1998
+ // declares no fields.
1999
+ structuredOutputs: StructuredOutputs.optional(),
2000
+ transcript: CallEndedTranscript,
2001
+ // true: the agent's worker confirmed its last transcript segment and captured field had landed before this payload
2002
+ // was built. false: built without that word (a worker that crashed or runs an SDK before 0.6.2, or one that didn't
2003
+ // answer within the hold) — the transcript and fields are what had arrived; GET /v1/calls/:id/result has any later.
2004
+ finalized: z.boolean(),
2005
+ // true: the payload would have passed CALL_ENDED_PAYLOAD_MAX_BYTES, so values were left out (see the cap) — fetch them.
2006
+ valuesOmitted: z.boolean()
1860
2007
  });
1861
2008
  z.object({
1862
2009
  callId: z.string().uuid(),
@@ -2432,166 +2579,6 @@ var init_telephony = __esm({
2432
2579
  });
2433
2580
  }
2434
2581
  });
2435
- var SHORTENER_HOSTS, PLACEHOLDER_HOSTS, PublicUrl, A2pBusinessType, A2pBusinessInfo, A2pOptInType, A2pUseCase, A2pMessagingProfile, A2pState;
2436
- var init_a2p = __esm({
2437
- "../contracts/src/a2p.ts"() {
2438
- SHORTENER_HOSTS = [
2439
- "bit.ly",
2440
- "tinyurl.com",
2441
- "t.co",
2442
- "goo.gl",
2443
- "ow.ly",
2444
- "is.gd",
2445
- "buff.ly",
2446
- "rebrand.ly",
2447
- "cutt.ly",
2448
- "shorturl.at",
2449
- "tiny.cc"
2450
- ];
2451
- PLACEHOLDER_HOSTS = ["acme.com", "example.com", "example.org", "test.com", "localhost"];
2452
- PublicUrl = z.string().url().refine((v) => v.startsWith("https://"), "must be https").refine((v) => {
2453
- try {
2454
- const host = new URL(v).hostname.toLowerCase().replace(/^www\./, "");
2455
- return !SHORTENER_HOSTS.includes(host);
2456
- } catch {
2457
- return false;
2458
- }
2459
- }, "public URL shorteners are rejected by carriers \u2014 use your own domain").refine((v) => {
2460
- try {
2461
- const host = new URL(v).hostname.toLowerCase().replace(/^www\./, "");
2462
- return !PLACEHOLDER_HOSTS.some((p) => host === p || host.endsWith(`.${p}`));
2463
- } catch {
2464
- return false;
2465
- }
2466
- }, "placeholder domain \u2014 reviewers will follow this link and reject the campaign");
2467
- A2pBusinessType = z.enum([
2468
- "Sole Proprietorship",
2469
- "Partnership",
2470
- "Corporation",
2471
- "Co-operative",
2472
- "Limited Liability Corporation",
2473
- "Non-profit Corporation"
2474
- ]);
2475
- A2pBusinessInfo = z.object({
2476
- legalName: z.string().trim().min(2).max(200),
2477
- /** EIN (US) or equivalent registration number. */
2478
- registrationNumber: z.string().trim().min(4).max(50),
2479
- businessType: A2pBusinessType,
2480
- /** Publicly reachable production site — not staging, not a 404. */
2481
- website: PublicUrl,
2482
- industry: z.string().trim().min(2).max(60),
2483
- address: z.object({
2484
- street: z.string().trim().min(2).max(200),
2485
- city: z.string().trim().min(1).max(100),
2486
- region: z.string().trim().min(1).max(100),
2487
- postalCode: z.string().trim().min(2).max(20),
2488
- isoCountry: z.string().trim().length(2)
2489
- }),
2490
- authorizedRep: z.object({
2491
- firstName: z.string().trim().min(1).max(100),
2492
- lastName: z.string().trim().min(1).max(100),
2493
- email: z.string().trim().email(),
2494
- phone: z.string().trim().regex(/^\+[1-9]\d{6,14}$/, "must be E.164"),
2495
- jobTitle: z.string().trim().min(2).max(100)
2496
- })
2497
- });
2498
- A2pOptInType = z.enum(["WEB_FORM", "PAPER_FORM", "VERBAL", "VIA_TEXT", "MOBILE_QR_CODE"]);
2499
- A2pUseCase = z.enum([
2500
- "MIXED",
2501
- "CUSTOMER_CARE",
2502
- "MARKETING",
2503
- "ACCOUNT_NOTIFICATION",
2504
- "2FA",
2505
- "DELIVERY_NOTIFICATION",
2506
- "HIGHER_EDUCATION",
2507
- "POLLING_VOTING",
2508
- "PUBLIC_SERVICE_ANNOUNCEMENT",
2509
- "LOW_VOLUME"
2510
- ]);
2511
- A2pMessagingProfile = z.object({
2512
- useCase: A2pUseCase,
2513
- /** Specific, not generic. "We send texts" gets rejected; describe the actual
2514
- * messages and when they are sent. */
2515
- description: z.string().trim().min(40).max(4096),
2516
- /**
2517
- * The single most-rejected field. Must describe HOW people opt in, state the
2518
- * message frequency, include the "message and data rates may apply"
2519
- * disclosure, and link to publicly reachable evidence. Twilio's API bounds it
2520
- * to 40–2049 characters.
2521
- */
2522
- messageFlow: z.string().trim().min(40).max(2049),
2523
- optInType: A2pOptInType,
2524
- /** Publicly accessible screenshots/pages showing the opt-in. Reviewers open
2525
- * these; anything behind a login fails. */
2526
- optInEvidenceUrls: z.array(PublicUrl).min(1).max(5),
2527
- /** Real messages the customer will send. Must reflect the declared use case
2528
- * and carry opt-out language. */
2529
- messageSamples: z.array(z.string().trim().min(10).max(1024)).min(2).max(5),
2530
- privacyPolicyUrl: PublicUrl,
2531
- termsAndConditionsUrl: PublicUrl,
2532
- hasEmbeddedLinks: z.boolean().default(false),
2533
- hasEmbeddedPhone: z.boolean().default(false)
2534
- });
2535
- z.object({
2536
- business: A2pBusinessInfo,
2537
- messaging: A2pMessagingProfile,
2538
- /** Register against Twilio's mock endpoints — exercises the full pipeline
2539
- * with no fees and no real carrier submission. Used in CI and staging. */
2540
- mock: z.boolean().default(false)
2541
- }).superRefine((v, ctx) => {
2542
- const flow = v.messaging.messageFlow.toLowerCase();
2543
- if (!/(msg|message)\s*(&|and)\s*data rates/.test(flow)) {
2544
- ctx.addIssue({
2545
- code: z.ZodIssueCode.custom,
2546
- path: ["messaging", "messageFlow"],
2547
- message: 'must include a "Message and data rates may apply" disclosure \u2014 carriers reject without it'
2548
- });
2549
- }
2550
- if (!/\d/.test(flow) || !/(msg|message|text)/.test(flow)) {
2551
- ctx.addIssue({
2552
- code: z.ZodIssueCode.custom,
2553
- path: ["messaging", "messageFlow"],
2554
- message: 'must state message frequency, e.g. "Up to 4 msgs/month"'
2555
- });
2556
- }
2557
- const hasOptOut = v.messaging.messageSamples.some((s) => /stop/i.test(s));
2558
- if (!hasOptOut) {
2559
- ctx.addIssue({
2560
- code: z.ZodIssueCode.custom,
2561
- path: ["messaging", "messageSamples"],
2562
- message: 'at least one sample must include opt-out language (e.g. "Reply STOP to opt out")'
2563
- });
2564
- }
2565
- });
2566
- A2pState = z.enum([
2567
- "none",
2568
- "profile_pending",
2569
- "profile_approved",
2570
- "profile_failed",
2571
- "brand_pending",
2572
- "brand_approved",
2573
- "brand_failed",
2574
- "campaign_pending",
2575
- "messaging_ready",
2576
- "campaign_failed"
2577
- ]);
2578
- z.object({
2579
- state: A2pState,
2580
- customerProfileSid: z.string().nullable(),
2581
- trustProductSid: z.string().nullable(),
2582
- brandSid: z.string().nullable(),
2583
- messagingServiceSid: z.string().nullable(),
2584
- campaignSid: z.string().nullable(),
2585
- mock: z.boolean(),
2586
- /** Carrier/Twilio rejection details, surfaced verbatim so the customer can
2587
- * fix the specific field rather than guess. */
2588
- failures: z.array(z.object({ code: z.number().nullable(), field: z.string().nullable(), message: z.string() })).default([]),
2589
- /** Plain-language next step for the dashboard. */
2590
- nextAction: z.string().nullable(),
2591
- updatedAt: z.string().nullable()
2592
- });
2593
- }
2594
- });
2595
2582
  var ConnectorMode, ConnectorStatus, ConnectorNormalize;
2596
2583
  var init_connector = __esm({
2597
2584
  "../contracts/src/connector.ts"() {
@@ -2706,7 +2693,7 @@ var init_connector_stream = __esm({
2706
2693
  ]);
2707
2694
  }
2708
2695
  });
2709
- var FLOW_BOOT_FAILURES;
2696
+ var FLOW_BOOT_FAILURES, AGENT_END_REASONS;
2710
2697
  var init_call_end_reason = __esm({
2711
2698
  "../contracts/src/call-end-reason.ts"() {
2712
2699
  FLOW_BOOT_FAILURES = [
@@ -2717,9 +2704,11 @@ var init_call_end_reason = __esm({
2717
2704
  "worker_identity_refused",
2718
2705
  "provider_unavailable"
2719
2706
  ];
2720
- z.enum(
2721
- FLOW_BOOT_FAILURES.map((cause) => `flow_boot:${cause}`)
2722
- );
2707
+ AGENT_END_REASONS = [
2708
+ "max_duration",
2709
+ ...FLOW_BOOT_FAILURES.map((cause) => `flow_boot:${cause}`)
2710
+ ];
2711
+ z.enum(AGENT_END_REASONS);
2723
2712
  }
2724
2713
  });
2725
2714
  var EnvironmentSpec;
@@ -3915,6 +3904,23 @@ var init_wallet = __esm({
3915
3904
  });
3916
3905
  }
3917
3906
  });
3907
+ var PipelineSlotKey;
3908
+ var init_pipeline_keys = __esm({
3909
+ "../contracts/src/pipeline-keys.ts"() {
3910
+ PipelineSlotKey = z.object({
3911
+ /** The provider as the slot reports it (LiveKit's label: `openai`, `api.openai.com`, `Deepgram`, …). */
3912
+ provider: z.string().min(1).max(200),
3913
+ /** true = the workspace's own key built it; false = the platform's. */
3914
+ byok: z.boolean()
3915
+ });
3916
+ z.object({
3917
+ llm: PipelineSlotKey.optional(),
3918
+ stt: PipelineSlotKey.optional(),
3919
+ tts: PipelineSlotKey.optional(),
3920
+ realtime: PipelineSlotKey.optional()
3921
+ });
3922
+ }
3923
+ });
3918
3924
 
3919
3925
  // ../contracts/src/index.ts
3920
3926
  var init_src = __esm({
@@ -3944,6 +3950,7 @@ var init_src = __esm({
3944
3950
  init_tests();
3945
3951
  init_call_review();
3946
3952
  init_call_result();
3953
+ init_call_outcome();
3947
3954
  init_webhook_events();
3948
3955
  init_numbers();
3949
3956
  init_project_limits();
@@ -3959,12 +3966,12 @@ var init_src = __esm({
3959
3966
  init_connection();
3960
3967
  init_secret_headers();
3961
3968
  init_telephony();
3962
- init_a2p();
3963
3969
  init_connector();
3964
3970
  init_connector_stream();
3965
3971
  init_flow_compile();
3966
3972
  init_flow_compile_check();
3967
3973
  init_call_end_reason();
3974
+ init_call_limits();
3968
3975
  init_flow_enrichment();
3969
3976
  init_environments();
3970
3977
  init_variables();
@@ -3980,6 +3987,7 @@ var init_src = __esm({
3980
3987
  init_wallet();
3981
3988
  init_network_address();
3982
3989
  init_dispatch_metadata();
3990
+ init_pipeline_keys();
3983
3991
  }
3984
3992
  });
3985
3993
 
@@ -4069,32 +4077,116 @@ function acceptsReasoningEffort(model2) {
4069
4077
  const id = baseModelId(model2);
4070
4078
  return isReasoningModel(model2) && !/^o1-(mini|preview)/.test(id) && !/-chat(-|$)/.test(id);
4071
4079
  }
4080
+ function acceptsNoReasoningEffort(model2) {
4081
+ if (!acceptsReasoningEffort(model2))
4082
+ return false;
4083
+ const id = baseModelId(model2);
4084
+ if (/-pro\b/.test(id))
4085
+ return false;
4086
+ const gpt = /^gpt-(\d+)(?:\.(\d+))?/.exec(id);
4087
+ if (!gpt)
4088
+ return false;
4089
+ const major = Number(gpt[1]);
4090
+ const minor = gpt[2] !== void 0 ? Number(gpt[2]) : 0;
4091
+ return major > 5 || major === 5 && minor >= 1;
4092
+ }
4093
+ function forgetLearnedReasoningEfforts() {
4094
+ learnedEfforts.clear();
4095
+ }
4096
+ function chatReasoningEffort(model2, opts) {
4097
+ const learned = learnedEfforts.get(learnedKey(model2, opts.tools === true));
4098
+ if (learned !== void 0)
4099
+ return learned === "omit" ? void 0 : learned;
4100
+ if (!acceptsReasoningEffort(model2))
4101
+ return void 0;
4102
+ if (opts.tools === true && acceptsNoReasoningEffort(model2))
4103
+ return "none";
4104
+ return opts.requested;
4105
+ }
4072
4106
  function chatCompletionParams(model2, input) {
4107
+ const effort = chatReasoningEffort(model2, {
4108
+ ...input.tools !== void 0 ? { tools: input.tools } : {},
4109
+ ...input.reasoningEffort !== void 0 ? { requested: input.reasoningEffort } : {}
4110
+ });
4073
4111
  if (isReasoningModel(model2)) {
4074
4112
  return {
4075
4113
  ...input.maxTokens !== void 0 ? { max_completion_tokens: Math.max(input.maxTokens, REASONING_MIN_COMPLETION_TOKENS) } : {},
4076
- ...input.reasoningEffort !== void 0 && acceptsReasoningEffort(model2) ? { reasoning_effort: input.reasoningEffort } : {}
4114
+ ...effort !== void 0 ? { reasoning_effort: effort } : {}
4077
4115
  };
4078
4116
  }
4079
4117
  return {
4118
+ // only when a provider taught the process that this model (one we don't know as a reasoning model) takes an effort
4119
+ ...effort !== void 0 ? { reasoning_effort: effort } : {},
4080
4120
  ...input.maxTokens !== void 0 ? { max_tokens: input.maxTokens } : {},
4081
4121
  ...input.temperature !== void 0 ? { temperature: input.temperature } : {},
4082
4122
  ...input.topP !== void 0 ? { top_p: input.topP } : {}
4083
4123
  };
4084
4124
  }
4085
- function modelRejectionOf(err) {
4125
+ function acceptsSamplingParams(model2) {
4126
+ return !isReasoningModel(model2);
4127
+ }
4128
+ function providerErrorOf(err) {
4129
+ if (err === null || typeof err !== "object")
4130
+ return null;
4086
4131
  const e = err;
4087
- const status = typeof e?.status === "number" ? e.status : null;
4088
- if (status !== 400 && status !== 403 && status !== 404)
4132
+ const body = e["body"] !== null && typeof e["body"] === "object" ? e["body"] : {};
4133
+ const pick = (k) => e[k] !== void 0 && e[k] !== null ? e[k] : body[k];
4134
+ const statusRaw = typeof e["status"] === "number" ? e["status"] : e["statusCode"];
4135
+ const status = typeof statusRaw === "number" && statusRaw > 0 ? statusRaw : null;
4136
+ const code = pick("code");
4137
+ const type = pick("type");
4138
+ const param = pick("param");
4139
+ const message = typeof body["message"] === "string" ? body["message"] : typeof e["message"] === "string" ? e["message"] : "";
4140
+ return {
4141
+ status,
4142
+ code: typeof code === "string" ? code : typeof type === "string" ? type : "",
4143
+ param: typeof param === "string" ? param : null,
4144
+ message
4145
+ };
4146
+ }
4147
+ function reasoningEffortRejectionOf(err) {
4148
+ const f = providerErrorOf(err);
4149
+ if (!f || f.status !== 400)
4089
4150
  return null;
4090
- const code = typeof e?.code === "string" ? e.code : typeof e?.type === "string" ? e.type : "";
4091
- const param = typeof e?.param === "string" ? e.param : null;
4092
- const rejected = REJECTION_CODES.has(code) || param !== null && MODEL_PARAMS.has(param);
4151
+ if (f.param !== "reasoning_effort" && !/reasoning[_ ]effort/i.test(f.message))
4152
+ return null;
4153
+ return { status: f.status, message: f.message, suggestsNone: /reasoning[_ ]effort[^.]*\bnone\b/i.test(f.message) };
4154
+ }
4155
+ function learnReasoningEffort(model2, shape, err) {
4156
+ const rejection = reasoningEffortRejectionOf(err);
4157
+ if (!rejection)
4158
+ return false;
4159
+ const next = shape.sent !== "none" && (rejection.suggestsNone || shape.sent === void 0) ? "none" : "omit";
4160
+ if ((next === "omit" ? void 0 : next) === shape.sent)
4161
+ return false;
4162
+ learnedEfforts.set(learnedKey(model2, shape.tools), next);
4163
+ return true;
4164
+ }
4165
+ async function withReasoningEffortRetry(shape, call) {
4166
+ const sent = chatReasoningEffort(shape.model, {
4167
+ tools: shape.tools,
4168
+ ...shape.requested !== void 0 ? { requested: shape.requested } : {}
4169
+ });
4170
+ try {
4171
+ return await call();
4172
+ } catch (err) {
4173
+ if (!learnReasoningEffort(shape.model, { tools: shape.tools, sent }, err))
4174
+ throw err;
4175
+ return call();
4176
+ }
4177
+ }
4178
+ function modelRejectionOf(err) {
4179
+ const f = providerErrorOf(err);
4180
+ const status = f?.status ?? null;
4181
+ if (!f || status === null || status !== 400 && status !== 403 && status !== 404)
4182
+ return null;
4183
+ const { code, param } = f;
4184
+ const rejected = REJECTION_CODES.has(code) || param !== null && MODEL_PARAMS.has(param) || reasoningEffortRejectionOf(err) !== null;
4093
4185
  if (!rejected)
4094
4186
  return null;
4095
4187
  if (status === 403 && code !== "model_not_found")
4096
4188
  return null;
4097
- return { status, code: code || "invalid_request_error", message: typeof e?.message === "string" ? e.message : "" };
4189
+ return { status, code: code || "invalid_request_error", message: f.message };
4098
4190
  }
4099
4191
  async function withModelFallback(opts) {
4100
4192
  const fallbackModel = opts.fallbackModel ?? PLATFORM_DEFAULT_CHAT_MODEL;
@@ -4115,11 +4207,13 @@ async function withModelFallback(opts) {
4115
4207
  return opts.call(fallbackModel);
4116
4208
  }
4117
4209
  }
4118
- var PLATFORM_DEFAULT_CHAT_MODEL, REASONING_MIN_COMPLETION_TOKENS, REJECTION_CODES, MODEL_PARAMS, MODEL_REJECTION_TTL_MS, ModelRejectionCache;
4210
+ var PLATFORM_DEFAULT_CHAT_MODEL, REASONING_MIN_COMPLETION_TOKENS, learnedEfforts, learnedKey, REJECTION_CODES, MODEL_PARAMS, MODEL_REJECTION_TTL_MS, ModelRejectionCache;
4119
4211
  var init_chat_params = __esm({
4120
4212
  "../llm-client/dist/chat-params.js"() {
4121
4213
  PLATFORM_DEFAULT_CHAT_MODEL = "gpt-4o-mini";
4122
4214
  REASONING_MIN_COMPLETION_TOKENS = 2048;
4215
+ learnedEfforts = /* @__PURE__ */ new Map();
4216
+ learnedKey = (model2, tools) => `${baseModelId(model2)}|${tools ? "tools" : "plain"}`;
4123
4217
  REJECTION_CODES = /* @__PURE__ */ new Set(["unsupported_parameter", "unsupported_value", "model_not_found"]);
4124
4218
  MODEL_PARAMS = /* @__PURE__ */ new Set(["model", "max_tokens", "max_completion_tokens", "temperature", "top_p", "reasoning_effort"]);
4125
4219
  MODEL_REJECTION_TTL_MS = 5 * 60 * 1e3;
@@ -4147,6 +4241,30 @@ var init_chat_params = __esm({
4147
4241
  };
4148
4242
  }
4149
4243
  });
4244
+
4245
+ // ../llm-client/dist/index.js
4246
+ var dist_exports = {};
4247
+ __export(dist_exports, {
4248
+ MODEL_REJECTION_TTL_MS: () => MODEL_REJECTION_TTL_MS,
4249
+ ModelRejectionCache: () => ModelRejectionCache,
4250
+ PLATFORM_DEFAULT_CHAT_MODEL: () => PLATFORM_DEFAULT_CHAT_MODEL,
4251
+ REASONING_MIN_COMPLETION_TOKENS: () => REASONING_MIN_COMPLETION_TOKENS,
4252
+ acceptsNoReasoningEffort: () => acceptsNoReasoningEffort,
4253
+ acceptsReasoningEffort: () => acceptsReasoningEffort,
4254
+ acceptsSamplingParams: () => acceptsSamplingParams,
4255
+ chatCompletionParams: () => chatCompletionParams,
4256
+ chatReasoningEffort: () => chatReasoningEffort,
4257
+ createChatClient: () => createChatClient,
4258
+ forgetLearnedReasoningEfforts: () => forgetLearnedReasoningEfforts,
4259
+ hasChatKey: () => hasChatKey,
4260
+ isReasoningModel: () => isReasoningModel,
4261
+ learnReasoningEffort: () => learnReasoningEffort,
4262
+ modelRejectionOf: () => modelRejectionOf,
4263
+ providerErrorOf: () => providerErrorOf,
4264
+ reasoningEffortRejectionOf: () => reasoningEffortRejectionOf,
4265
+ withModelFallback: () => withModelFallback,
4266
+ withReasoningEffortRetry: () => withReasoningEffortRetry
4267
+ });
4150
4268
  function createChatClient(config = {}) {
4151
4269
  return new OpenAI({
4152
4270
  ...config.apiKey ? { apiKey: config.apiKey } : {},
@@ -4154,6 +4272,9 @@ function createChatClient(config = {}) {
4154
4272
  ...config.timeoutMs ? { timeout: config.timeoutMs } : {}
4155
4273
  });
4156
4274
  }
4275
+ function hasChatKey(config = {}) {
4276
+ return Boolean(config.apiKey || process.env["OPENAI_API_KEY"]);
4277
+ }
4157
4278
  var init_dist = __esm({
4158
4279
  "../llm-client/dist/index.js"() {
4159
4280
  init_chat_params();
@@ -4293,6 +4414,340 @@ var init_wrap = __esm({
4293
4414
  "src/providers/wrap.ts"() {
4294
4415
  }
4295
4416
  });
4417
+ var init_start = __esm({
4418
+ "../observability/dist/start.js"() {
4419
+ }
4420
+ });
4421
+
4422
+ // ../observability/dist/attributes.js
4423
+ var ATTR;
4424
+ var init_attributes = __esm({
4425
+ "../observability/dist/attributes.js"() {
4426
+ ATTR = {
4427
+ projectId: "vl.project_id",
4428
+ callId: "vl.call_id",
4429
+ campaignId: "vl.campaign_id",
4430
+ room: "vl.room",
4431
+ agentId: "vl.agent_id",
4432
+ phoneNumberId: "vl.phone_number_id",
4433
+ bindingId: "vl.binding_id",
4434
+ source: "vl.source",
4435
+ kind: "vl.kind"
4436
+ };
4437
+ }
4438
+ });
4439
+ function getCurrentCallContext() {
4440
+ return callContextStore.getStore();
4441
+ }
4442
+ var callContextStore;
4443
+ var init_call_context = __esm({
4444
+ "../observability/dist/call-context.js"() {
4445
+ init_attributes();
4446
+ callContextStore = new AsyncLocalStorage();
4447
+ }
4448
+ });
4449
+ var init_trace_propagation = __esm({
4450
+ "../observability/dist/trace-propagation.js"() {
4451
+ }
4452
+ });
4453
+ function meter() {
4454
+ return metrics.getMeter(METER_NAME, METER_VERSION);
4455
+ }
4456
+ function modelFallbacks() {
4457
+ return _modelFallbacks ??= meter().createCounter("vl.llm.model_fallback", {
4458
+ description: "Model calls the provider rejected that fell back to the platform default model",
4459
+ unit: "{fallbacks}"
4460
+ });
4461
+ }
4462
+ function recordModelFallback(args) {
4463
+ const out = {
4464
+ "vl.model": args.model,
4465
+ "vl.fallback_model": args.fallbackModel,
4466
+ "vl.error_code": args.code
4467
+ };
4468
+ if (args.projectId)
4469
+ out[ATTR.projectId] = args.projectId;
4470
+ if (args.agentId)
4471
+ out[ATTR.agentId] = args.agentId;
4472
+ if (args.surface)
4473
+ out["vl.surface"] = args.surface;
4474
+ modelFallbacks().add(1, out);
4475
+ }
4476
+ var METER_NAME, METER_VERSION, _modelFallbacks;
4477
+ var init_metrics = __esm({
4478
+ "../observability/dist/metrics.js"() {
4479
+ init_attributes();
4480
+ METER_NAME = "voicelayer";
4481
+ METER_VERSION = "0.1.0";
4482
+ _modelFallbacks = null;
4483
+ }
4484
+ });
4485
+ var init_latency_span = __esm({
4486
+ "../observability/dist/latency-span.js"() {
4487
+ init_attributes();
4488
+ }
4489
+ });
4490
+
4491
+ // ../observability/dist/index.js
4492
+ var init_dist2 = __esm({
4493
+ "../observability/dist/index.js"() {
4494
+ init_start();
4495
+ init_call_context();
4496
+ init_trace_propagation();
4497
+ init_attributes();
4498
+ init_metrics();
4499
+ init_latency_span();
4500
+ }
4501
+ });
4502
+
4503
+ // src/providers/resilient-llm.ts
4504
+ var resilient_llm_exports = {};
4505
+ __export(resilient_llm_exports, {
4506
+ LLM_APOLOGY: () => LLM_APOLOGY,
4507
+ ResilientLLM: () => ResilientLLM,
4508
+ isResilientLLM: () => isResilientLLM
4509
+ });
4510
+ function isResilientLLM(value) {
4511
+ return typeof value === "object" && value !== null && value[RESILIENT] === true;
4512
+ }
4513
+ function transient(error) {
4514
+ if (error instanceof APIStatusError) {
4515
+ const s = error.statusCode;
4516
+ return s === 408 || s === 429 || s < 0 || s >= 500;
4517
+ }
4518
+ return error instanceof APITimeoutError || error instanceof APIConnectionError;
4519
+ }
4520
+ var LLM_APOLOGY, RESILIENT, ResilientLLM, sameModel, ServedLLM, ResilientLLMStream;
4521
+ var init_resilient_llm = __esm({
4522
+ "src/providers/resilient-llm.ts"() {
4523
+ init_dist();
4524
+ init_dist2();
4525
+ LLM_APOLOGY = "Sorry, I'm having trouble right now. Could you give me a moment and say that again?";
4526
+ RESILIENT = /* @__PURE__ */ Symbol.for("voicelayer.resilientLLM");
4527
+ ResilientLLM = class extends llm.LLM {
4528
+ [RESILIENT] = true;
4529
+ #opts;
4530
+ #fallback = null;
4531
+ #listeners = /* @__PURE__ */ new Set();
4532
+ /** Models the provider rejected on this call (a call's pipeline is built per call): later turns go straight to the fallback. */
4533
+ rejections = new ModelRejectionCache();
4534
+ constructor(opts) {
4535
+ super();
4536
+ this.#opts = opts;
4537
+ }
4538
+ label() {
4539
+ return this.#opts.primary.label();
4540
+ }
4541
+ get model() {
4542
+ return this.#opts.primary.model;
4543
+ }
4544
+ get provider() {
4545
+ return this.#opts.primary.provider;
4546
+ }
4547
+ get apology() {
4548
+ return this.#opts.apology ?? LLM_APOLOGY;
4549
+ }
4550
+ get reasoningParams() {
4551
+ return this.#opts.reasoningParams === true;
4552
+ }
4553
+ /** The fallback LLM, built on first need; null when there is none or it is the agent's own model. */
4554
+ fallbackLLM() {
4555
+ if (!this.#opts.fallback) return null;
4556
+ this.#fallback ??= this.#opts.fallback();
4557
+ return sameModel(this.#fallback.model, this.model) ? null : this.#fallback;
4558
+ }
4559
+ /** Hear about every recovery (and every apology). Returns the unsubscribe. */
4560
+ onIncident(listener) {
4561
+ this.#listeners.add(listener);
4562
+ return () => this.#listeners.delete(listener);
4563
+ }
4564
+ /** @internal */
4565
+ report(incident) {
4566
+ const log = incident.kind === "apology" ? console.error : console.warn;
4567
+ log(`[voicelayer] llm ${incident.kind.replace(/_/g, " ")}`, incident);
4568
+ if (incident.kind === "model_fallback") {
4569
+ const call = getCurrentCallContext();
4570
+ recordModelFallback({
4571
+ model: incident.model,
4572
+ fallbackModel: incident.fallbackModel,
4573
+ code: incident.code,
4574
+ surface: "voice",
4575
+ ...call?.projectId ? { projectId: call.projectId } : {},
4576
+ ...call?.agentId ? { agentId: call.agentId } : {}
4577
+ });
4578
+ }
4579
+ for (const listener of this.#listeners) {
4580
+ try {
4581
+ listener(incident);
4582
+ } catch {
4583
+ }
4584
+ }
4585
+ }
4586
+ chat(args) {
4587
+ return new ResilientLLMStream(this, args);
4588
+ }
4589
+ prewarm() {
4590
+ this.#opts.primary.prewarm();
4591
+ }
4592
+ async aclose() {
4593
+ await Promise.all([this.#opts.primary.aclose(), this.#fallback?.aclose()]);
4594
+ }
4595
+ /** @internal */
4596
+ get primary() {
4597
+ return this.#opts.primary;
4598
+ }
4599
+ };
4600
+ sameModel = (a, b) => a.trim().toLowerCase() === b.trim().toLowerCase();
4601
+ ServedLLM = class extends llm.LLM {
4602
+ constructor(owner, served) {
4603
+ super();
4604
+ this.owner = owner;
4605
+ this.served = served;
4606
+ }
4607
+ owner;
4608
+ served;
4609
+ label() {
4610
+ return this.owner.label();
4611
+ }
4612
+ get model() {
4613
+ return this.served();
4614
+ }
4615
+ get provider() {
4616
+ return this.owner.provider;
4617
+ }
4618
+ chat(args) {
4619
+ return this.owner.chat(args);
4620
+ }
4621
+ emit(event, ...args) {
4622
+ return this.owner.emit(event, ...args);
4623
+ }
4624
+ };
4625
+ ResilientLLMStream = class extends llm.LLMStream {
4626
+ #owner;
4627
+ #args;
4628
+ #conn;
4629
+ /** The model answering this stream — what its metrics report. */
4630
+ #served;
4631
+ #current = null;
4632
+ constructor(owner, args) {
4633
+ const conn = args.connOptions ?? DEFAULT_API_CONNECT_OPTIONS;
4634
+ const served = { model: owner.model };
4635
+ super(new ServedLLM(owner, () => served.model), {
4636
+ chatCtx: args.chatCtx,
4637
+ ...args.toolCtx ? { toolCtx: args.toolCtx } : {},
4638
+ connOptions: { ...conn, maxRetry: 0 }
4639
+ });
4640
+ this.#owner = owner;
4641
+ this.#args = args;
4642
+ this.#conn = conn;
4643
+ this.#served = served;
4644
+ this.abortController.signal.addEventListener("abort", () => this.#current?.close());
4645
+ }
4646
+ get #hasTools() {
4647
+ return this.#args.toolCtx !== void 0 && Object.keys(this.#args.toolCtx).length > 0;
4648
+ }
4649
+ #extraKwargs(model2) {
4650
+ const base = this.#args.extraKwargs;
4651
+ if (!this.#owner.reasoningParams) return base;
4652
+ const effort = chatReasoningEffort(model2, { tools: this.#hasTools });
4653
+ if (effort === void 0) {
4654
+ if (!base || !("reasoning_effort" in base)) return base;
4655
+ const { reasoning_effort: _dropped, ...rest } = base;
4656
+ return rest;
4657
+ }
4658
+ return { ...base, reasoning_effort: effort };
4659
+ }
4660
+ /** One request on `target`, forwarding its chunks. Its failure is the inner LLM's 'error' event. */
4661
+ async #attempt(target) {
4662
+ let failure = null;
4663
+ const onError = (ev) => {
4664
+ failure ??= ev.error;
4665
+ };
4666
+ target.on("error", onError);
4667
+ let started = false;
4668
+ try {
4669
+ const extraKwargs = this.#extraKwargs(target.model);
4670
+ const stream = target.chat({
4671
+ ...this.#args,
4672
+ connOptions: { ...this.#conn, maxRetry: 0 },
4673
+ ...extraKwargs !== void 0 ? { extraKwargs } : {}
4674
+ });
4675
+ this.#current = stream;
4676
+ for await (const chunk of stream) {
4677
+ if (this.abortController.signal.aborted) break;
4678
+ started = true;
4679
+ this.queue.put(chunk);
4680
+ }
4681
+ } catch (err) {
4682
+ failure ??= err instanceof Error ? err : new Error(String(err));
4683
+ } finally {
4684
+ target.off("error", onError);
4685
+ this.#current = null;
4686
+ }
4687
+ return failure ? { ok: false, error: failure, started } : { ok: true };
4688
+ }
4689
+ /** Run `target` until it answers, or a failure that retrying it won't fix. */
4690
+ async #run(target) {
4691
+ let effortRetried = false;
4692
+ for (let retries = 0; ; ) {
4693
+ const sent = this.#owner.reasoningParams ? chatReasoningEffort(target.model, { tools: this.#hasTools }) : void 0;
4694
+ const result = await this.#attempt(target);
4695
+ if (result.ok || result.started || this.abortController.signal.aborted) return result;
4696
+ if (this.#owner.reasoningParams && !effortRetried && learnReasoningEffort(target.model, { tools: this.#hasTools, sent }, result.error)) {
4697
+ effortRetried = true;
4698
+ this.#owner.report({
4699
+ kind: "reasoning_effort_adapted",
4700
+ model: target.model,
4701
+ message: providerErrorOf(result.error)?.message ?? result.error.message
4702
+ });
4703
+ continue;
4704
+ }
4705
+ if (!transient(result.error) || retries >= this.#conn.maxRetry) return result;
4706
+ const wait = intervalForRetry(this.#conn, retries);
4707
+ retries += 1;
4708
+ if (wait > 0) await new Promise((r) => setTimeout(r, wait));
4709
+ if (this.abortController.signal.aborted) return result;
4710
+ }
4711
+ }
4712
+ async run() {
4713
+ const owner = this.#owner;
4714
+ const model2 = owner.model;
4715
+ const fallback = owner.fallbackLLM();
4716
+ const known = fallback ? owner.rejections.get(model2) : null;
4717
+ let failure;
4718
+ if (known && fallback) {
4719
+ owner.report({ kind: "model_fallback", model: model2, fallbackModel: fallback.model, code: known.code, message: known.message, cached: true });
4720
+ failure = new Error(known.message);
4721
+ } else {
4722
+ const first = await this.#run(owner.primary);
4723
+ if (first.ok || first.started || this.abortController.signal.aborted) return;
4724
+ failure = first.error;
4725
+ const rejection = modelRejectionOf(first.error);
4726
+ if (rejection) owner.rejections.set(model2, rejection);
4727
+ if (fallback) {
4728
+ const f = providerErrorOf(first.error);
4729
+ owner.report({
4730
+ kind: "model_fallback",
4731
+ model: model2,
4732
+ fallbackModel: fallback.model,
4733
+ code: rejection?.code ?? (f?.status ? String(f.status) : "error"),
4734
+ message: f?.message || first.error.message,
4735
+ cached: false
4736
+ });
4737
+ }
4738
+ }
4739
+ if (fallback) {
4740
+ this.#served.model = fallback.model;
4741
+ const second = await this.#run(fallback);
4742
+ if (second.ok || second.started || this.abortController.signal.aborted) return;
4743
+ failure = second.error;
4744
+ }
4745
+ owner.report({ kind: "apology", model: this.#served.model, message: providerErrorOf(failure)?.message || failure.message });
4746
+ this.queue.put({ id: `vl-apology-${Date.now()}`, delta: { role: "assistant", content: owner.apology } });
4747
+ }
4748
+ };
4749
+ }
4750
+ });
4296
4751
 
4297
4752
  // src/providers/index.ts
4298
4753
  async function importOptional(spec, hint) {
@@ -4339,9 +4794,13 @@ var init_providers2 = __esm({
4339
4794
  llm(options = {}) {
4340
4795
  return async () => {
4341
4796
  const oa = await importOptional("@voicelayer/agents-plugin-openai", "openai.llm");
4342
- return new oa.LLM(
4343
- withCreds({ model: options.model ?? "gpt-4o-mini" }, options)
4344
- );
4797
+ const { ResilientLLM: ResilientLLM2 } = await Promise.resolve().then(() => (init_resilient_llm(), resilient_llm_exports));
4798
+ const { PLATFORM_DEFAULT_CHAT_MODEL: PLATFORM_DEFAULT_CHAT_MODEL2 } = await Promise.resolve().then(() => (init_dist(), dist_exports));
4799
+ return new ResilientLLM2({
4800
+ primary: new oa.LLM(withCreds({ model: options.model ?? "gpt-4o-mini" }, options)),
4801
+ fallback: () => new oa.LLM(withCreds({ model: PLATFORM_DEFAULT_CHAT_MODEL2 }, options)),
4802
+ reasoningParams: true
4803
+ });
4345
4804
  };
4346
4805
  },
4347
4806
  tts(options = {}) {
@@ -4417,9 +4876,10 @@ var init_providers2 = __esm({
4417
4876
  llm(options = {}) {
4418
4877
  return async () => {
4419
4878
  const g = await importOptional("@voicelayer/agents-plugin-google", "google.llm");
4420
- return new g.LLM(
4421
- withCreds({ model: options.model ?? "gemini-3.8-flash" }, options)
4422
- );
4879
+ const { ResilientLLM: ResilientLLM2 } = await Promise.resolve().then(() => (init_resilient_llm(), resilient_llm_exports));
4880
+ return new ResilientLLM2({
4881
+ primary: new g.LLM(withCreds({ model: options.model ?? "gemini-3.8-flash" }, options))
4882
+ });
4423
4883
  };
4424
4884
  },
4425
4885
  realtime(options = {}) {
@@ -4521,91 +4981,6 @@ var init_llm = __esm({
4521
4981
  init_registry();
4522
4982
  }
4523
4983
  });
4524
- var init_start = __esm({
4525
- "../observability/dist/start.js"() {
4526
- }
4527
- });
4528
-
4529
- // ../observability/dist/attributes.js
4530
- var ATTR;
4531
- var init_attributes = __esm({
4532
- "../observability/dist/attributes.js"() {
4533
- ATTR = {
4534
- projectId: "vl.project_id",
4535
- callId: "vl.call_id",
4536
- campaignId: "vl.campaign_id",
4537
- room: "vl.room",
4538
- agentId: "vl.agent_id",
4539
- phoneNumberId: "vl.phone_number_id",
4540
- bindingId: "vl.binding_id",
4541
- source: "vl.source",
4542
- kind: "vl.kind"
4543
- };
4544
- }
4545
- });
4546
- function getCurrentCallContext() {
4547
- return callContextStore.getStore();
4548
- }
4549
- var callContextStore;
4550
- var init_call_context = __esm({
4551
- "../observability/dist/call-context.js"() {
4552
- init_attributes();
4553
- callContextStore = new AsyncLocalStorage();
4554
- }
4555
- });
4556
- var init_trace_propagation = __esm({
4557
- "../observability/dist/trace-propagation.js"() {
4558
- }
4559
- });
4560
- function meter() {
4561
- return metrics.getMeter(METER_NAME, METER_VERSION);
4562
- }
4563
- function modelFallbacks() {
4564
- return _modelFallbacks ??= meter().createCounter("vl.llm.model_fallback", {
4565
- description: "Model calls the provider rejected that fell back to the platform default model",
4566
- unit: "{fallbacks}"
4567
- });
4568
- }
4569
- function recordModelFallback(args) {
4570
- const out = {
4571
- "vl.model": args.model,
4572
- "vl.fallback_model": args.fallbackModel,
4573
- "vl.error_code": args.code
4574
- };
4575
- if (args.projectId)
4576
- out[ATTR.projectId] = args.projectId;
4577
- if (args.agentId)
4578
- out[ATTR.agentId] = args.agentId;
4579
- if (args.surface)
4580
- out["vl.surface"] = args.surface;
4581
- modelFallbacks().add(1, out);
4582
- }
4583
- var METER_NAME, METER_VERSION, _modelFallbacks;
4584
- var init_metrics = __esm({
4585
- "../observability/dist/metrics.js"() {
4586
- init_attributes();
4587
- METER_NAME = "voicelayer";
4588
- METER_VERSION = "0.1.0";
4589
- _modelFallbacks = null;
4590
- }
4591
- });
4592
- var init_latency_span = __esm({
4593
- "../observability/dist/latency-span.js"() {
4594
- init_attributes();
4595
- }
4596
- });
4597
-
4598
- // ../observability/dist/index.js
4599
- var init_dist2 = __esm({
4600
- "../observability/dist/index.js"() {
4601
- init_start();
4602
- init_call_context();
4603
- init_trace_propagation();
4604
- init_attributes();
4605
- init_metrics();
4606
- init_latency_span();
4607
- }
4608
- });
4609
4984
 
4610
4985
  // src/runtime/helper-models.ts
4611
4986
  function rejectionsFor(client) {