@voicelayer/sdk 0.6.0 → 0.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -4675,10 +4675,86 @@ var init_dist = __esm({
4675
4675
  init_chat_params();
4676
4676
  }
4677
4677
  });
4678
+ function runWithModelMeter(listener, fn) {
4679
+ return meterScope.run(listener, fn);
4680
+ }
4681
+ function requestedModel(body) {
4682
+ const model2 = body?.model;
4683
+ return typeof model2 === "string" && model2 ? model2 : void 0;
4684
+ }
4685
+ function safely(fn) {
4686
+ try {
4687
+ fn();
4688
+ } catch (err) {
4689
+ console.warn("[voicelayer] model meter listener failed", { err: err instanceof Error ? err.message : String(err) });
4690
+ }
4691
+ }
4692
+ function metered(target, create, byok) {
4693
+ return async (...args) => {
4694
+ const listener = meterScope.getStore();
4695
+ const startedAt = Date.now();
4696
+ const model2 = requestedModel(args[0]);
4697
+ if (listener) safely(() => listener.started());
4698
+ let res;
4699
+ try {
4700
+ res = await Reflect.apply(create, target, args);
4701
+ return res;
4702
+ } finally {
4703
+ if (listener) {
4704
+ const usage = res?.usage;
4705
+ const record = {
4706
+ provider: "openai",
4707
+ model: model2 ?? res?.model ?? "unknown",
4708
+ ...usage ? { usage: { inputTokens: usage.prompt_tokens ?? 0, outputTokens: usage.completion_tokens ?? 0 } } : {},
4709
+ ...byok !== void 0 ? { byok } : {},
4710
+ startedAt,
4711
+ endedAt: Date.now()
4712
+ };
4713
+ safely(() => listener.finished(record));
4714
+ }
4715
+ }
4716
+ };
4717
+ }
4718
+ function meteredClient(client, byok) {
4719
+ const hit = wrapped.get(client);
4720
+ if (hit) return hit;
4721
+ const chat = client.chat;
4722
+ const embeddings = client.embeddings;
4723
+ const completionsCreate = chat ? metered(chat.completions, chat.completions.create, byok) : void 0;
4724
+ const embeddingsCreate = embeddings ? metered(embeddings, embeddings.create, byok) : void 0;
4725
+ const completions = chat ? new Proxy(chat.completions, { get: (t, p, r) => p === "create" ? completionsCreate : Reflect.get(t, p, r) }) : void 0;
4726
+ const chatProxy = chat ? new Proxy(chat, { get: (t, p, r) => p === "completions" ? completions : Reflect.get(t, p, r) }) : void 0;
4727
+ const embeddingsProxy = embeddings ? new Proxy(embeddings, { get: (t, p, r) => p === "create" ? embeddingsCreate : Reflect.get(t, p, r) }) : void 0;
4728
+ const proxy = new Proxy(client, {
4729
+ get(target, prop, receiver) {
4730
+ if (prop === "chat" && chatProxy) return chatProxy;
4731
+ if (prop === "embeddings" && embeddingsProxy) return embeddingsProxy;
4732
+ return Reflect.get(target, prop, receiver);
4733
+ }
4734
+ });
4735
+ wrapped.set(client, proxy);
4736
+ wrapped.set(proxy, proxy);
4737
+ return proxy;
4738
+ }
4739
+ var meterScope, wrapped;
4740
+ var init_model_meter = __esm({
4741
+ "src/runtime/model-meter.ts"() {
4742
+ meterScope = new AsyncLocalStorage();
4743
+ wrapped = /* @__PURE__ */ new WeakMap();
4744
+ }
4745
+ });
4678
4746
  function configureDefaultOpenAI(cred) {
4679
4747
  credential = cred;
4680
4748
  cached = null;
4681
4749
  }
4750
+ function configureHelperCallCredential(source) {
4751
+ callCredentialSource = source;
4752
+ cached = null;
4753
+ }
4754
+ function helperKeyIsCustomers() {
4755
+ if (credential) return true;
4756
+ return callCredentialSource !== "worker-token";
4757
+ }
4682
4758
  function defaultOpenAIOptions() {
4683
4759
  if (!credential) return {};
4684
4760
  return {
@@ -4686,27 +4762,31 @@ function defaultOpenAIOptions() {
4686
4762
  ...credential.baseURL ? { baseURL: credential.baseURL } : {}
4687
4763
  };
4688
4764
  }
4689
- function internalChatClient(opts) {
4765
+ function internalChatClient() {
4690
4766
  const cred = defaultOpenAIOptions();
4691
4767
  return createChatClient({
4692
4768
  ...cred.apiKey ? { apiKey: cred.apiKey } : {},
4693
- ...cred.baseURL ? { baseURL: cred.baseURL } : {},
4694
- ...opts?.timeoutMs ? { timeoutMs: opts.timeoutMs } : {}
4769
+ ...cred.baseURL ? { baseURL: cred.baseURL } : {}
4695
4770
  });
4696
4771
  }
4697
4772
  function defaultOpenAI() {
4698
- return scope.getStore() ?? (cached ??= internalChatClient());
4773
+ const scoped = scope.getStore();
4774
+ if (scoped) return meteredClient(scoped);
4775
+ if (!cached) cached = meteredClient(internalChatClient(), helperKeyIsCustomers());
4776
+ return cached;
4699
4777
  }
4700
4778
  function hasOpenAiKey() {
4701
4779
  return scope.getStore() !== void 0 || credential !== null || Boolean(process.env["OPENAI_API_KEY"]);
4702
4780
  }
4703
- var credential, cached, scope;
4781
+ var credential, cached, scope, callCredentialSource;
4704
4782
  var init_openai_default = __esm({
4705
4783
  "src/runtime/openai-default.ts"() {
4706
4784
  init_dist();
4785
+ init_model_meter();
4707
4786
  credential = null;
4708
4787
  cached = null;
4709
4788
  scope = new AsyncLocalStorage();
4789
+ callCredentialSource = null;
4710
4790
  }
4711
4791
  });
4712
4792
  function ipv4Parts(ip) {
@@ -5743,7 +5823,9 @@ var init_metrics = __esm({
5743
5823
  }
5744
5824
  });
5745
5825
  function recordTurnLatencySpan(args) {
5746
- if (!Number.isFinite(args.latencyMs) || args.latencyMs <= 0)
5826
+ if (!Number.isFinite(args.latencyMs) || args.latencyMs < 0)
5827
+ return;
5828
+ if (args.latencyMs === 0 && !args.allowZero)
5747
5829
  return;
5748
5830
  const attributes = {
5749
5831
  [ATTR.projectId]: args.projectId,
@@ -9137,19 +9219,22 @@ function parseArgs(raw) {
9137
9219
  }
9138
9220
  function defaultComplete() {
9139
9221
  if (!hasOpenAiKey()) return null;
9140
- const client = internalChatClient({ timeoutMs: RUNNER_TIMEOUT_MS });
9141
9222
  return async ({ model: model2, temperature, messages, tools }) => {
9223
+ const client = defaultOpenAI();
9142
9224
  const res = await withHelperModelFallback({
9143
9225
  client,
9144
9226
  model: model2,
9145
9227
  fallbackModel: RUNNER_MODEL,
9146
9228
  helper: "agent runner",
9147
- call: (m) => client.chat.completions.create({
9148
- model: m,
9149
- ...chatCompletionParams(m, { ...temperature !== void 0 ? { temperature } : {}, reasoningEffort: "low" }),
9150
- messages,
9151
- ...tools.length ? { tools } : {}
9152
- }),
9229
+ call: (m) => client.chat.completions.create(
9230
+ {
9231
+ model: m,
9232
+ ...chatCompletionParams(m, { ...temperature !== void 0 ? { temperature } : {}, reasoningEffort: "low" }),
9233
+ messages,
9234
+ ...tools.length ? { tools } : {}
9235
+ },
9236
+ { timeout: RUNNER_TIMEOUT_MS }
9237
+ ),
9153
9238
  warn: graphWarn
9154
9239
  });
9155
9240
  const msg = res.choices[0]?.message;
@@ -17147,9 +17232,11 @@ var UsageAccumulator = class {
17147
17232
  llm = /* @__PURE__ */ new Map();
17148
17233
  tts = /* @__PURE__ */ new Map();
17149
17234
  stt = /* @__PURE__ */ new Map();
17150
- addLlm(provider, model2, inputTokens, outputTokens) {
17151
- const key = `${provider}:${model2}`;
17152
- const entry2 = this.llm.get(key) ?? { provider, model: model2, inputTokens: 0, outputTokens: 0 };
17235
+ // `byok` keeps a line whose payer is known (the SDK's helper calls) apart from LiveKit's own line on the same model,
17236
+ // which the API prices by the workspace's vault keys.
17237
+ addLlm(provider, model2, inputTokens, outputTokens, byok) {
17238
+ const key = `${provider}:${model2}:${byok === void 0 ? "" : byok ? "byok" : "platform"}`;
17239
+ const entry2 = this.llm.get(key) ?? { provider, model: model2, inputTokens: 0, outputTokens: 0, ...byok !== void 0 ? { byok } : {} };
17153
17240
  entry2.inputTokens += Math.max(0, inputTokens || 0);
17154
17241
  entry2.outputTokens += Math.max(0, outputTokens || 0);
17155
17242
  this.llm.set(key, entry2);
@@ -17269,7 +17356,11 @@ function dispatch(metrics2, ctx) {
17269
17356
  inputTokens: m.promptTokens,
17270
17357
  outputTokens: m.completionTokens
17271
17358
  });
17272
- if (typeof m.ttftMs === "number" && m.ttftMs >= 0) {
17359
+ if (ctx.modelClock) {
17360
+ const startAt = m.timestamp - Math.max(0, m.durationMs);
17361
+ const waitedMs = typeof m.ttftMs === "number" && m.ttftMs >= 0 ? m.ttftMs : Math.max(0, m.durationMs);
17362
+ ctx.modelClock.noteReplyModelTime({ startAt, endAt: startAt + waitedMs });
17363
+ } else if (typeof m.ttftMs === "number" && m.ttftMs >= 0) {
17273
17364
  recordTurnLatency({ ...base, firstTokenMs: m.ttftMs });
17274
17365
  recordTurnLatencySpan({ ...base, stage: "llm", latencyMs: m.ttftMs });
17275
17366
  }
@@ -17372,7 +17463,8 @@ function createTelemetrySink(accumulator = new UsageAccumulator()) {
17372
17463
  sample.provider,
17373
17464
  sample.model ?? "unknown",
17374
17465
  sample.inputTokens ?? 0,
17375
- sample.outputTokens ?? 0
17466
+ sample.outputTokens ?? 0,
17467
+ sample.byok
17376
17468
  );
17377
17469
  break;
17378
17470
  case "tts":
@@ -17392,6 +17484,244 @@ function createTelemetrySink(accumulator = new UsageAccumulator()) {
17392
17484
  };
17393
17485
  }
17394
17486
 
17487
+ // src/runtime/call-model-usage.ts
17488
+ init_dist2();
17489
+ var MAX_UNTAGGED = 32;
17490
+ function createCallModelMeter(usage) {
17491
+ let base = null;
17492
+ let clock;
17493
+ let untagged = [];
17494
+ let inFlight = 0;
17495
+ let idle = [];
17496
+ const otel = (t) => {
17497
+ if (base) recordLlmUsage({ ...base, ...t });
17498
+ else if (untagged.length < MAX_UNTAGGED) untagged.push(t);
17499
+ };
17500
+ return {
17501
+ listener: {
17502
+ started() {
17503
+ inFlight += 1;
17504
+ },
17505
+ finished(call) {
17506
+ inFlight = Math.max(0, inFlight - 1);
17507
+ try {
17508
+ const tokens = call.usage;
17509
+ if (tokens && tokens.inputTokens + tokens.outputTokens > 0) {
17510
+ usage.record({
17511
+ kind: "llm",
17512
+ provider: call.provider,
17513
+ model: call.model,
17514
+ inputTokens: tokens.inputTokens,
17515
+ outputTokens: tokens.outputTokens,
17516
+ ...call.byok !== void 0 ? { byok: call.byok } : {}
17517
+ });
17518
+ otel({ provider: call.provider, model: call.model, inputTokens: tokens.inputTokens, outputTokens: tokens.outputTokens });
17519
+ }
17520
+ clock?.noteModelTime({ startAt: call.startedAt, endAt: call.endedAt });
17521
+ } finally {
17522
+ if (inFlight === 0 && idle.length > 0) {
17523
+ const waiting = idle;
17524
+ idle = [];
17525
+ for (const resolve of waiting) resolve();
17526
+ }
17527
+ }
17528
+ }
17529
+ },
17530
+ bind(b, c) {
17531
+ base = b;
17532
+ clock = c;
17533
+ const early = untagged;
17534
+ untagged = [];
17535
+ for (const t of early) recordLlmUsage({ ...b, ...t });
17536
+ },
17537
+ drain(timeoutMs) {
17538
+ if (inFlight === 0) return Promise.resolve();
17539
+ return new Promise((resolve) => {
17540
+ const timer = setTimeout(done, timeoutMs);
17541
+ timer.unref?.();
17542
+ function done() {
17543
+ clearTimeout(timer);
17544
+ idle = idle.filter((r) => r !== done);
17545
+ resolve();
17546
+ }
17547
+ idle.push(done);
17548
+ });
17549
+ }
17550
+ };
17551
+ }
17552
+ var HANGUP_DRAIN_MS = 3e3;
17553
+ async function reportUsageAtClose(opts) {
17554
+ try {
17555
+ await opts.meter.drain(opts.drainMs ?? HANGUP_DRAIN_MS);
17556
+ if (opts.usage.isEmpty()) return;
17557
+ await opts.send(opts.usage.report(opts.durationMs));
17558
+ } catch (err) {
17559
+ opts.warn?.("[agent] usage report failed", { err: err instanceof Error ? err.message : String(err) });
17560
+ }
17561
+ }
17562
+ var MAX_INTERVALS_PER_TURN = 64;
17563
+ var REPLY_METRICS_WAIT_MS = 1500;
17564
+ function modelTimeWithin(intervals, from, to) {
17565
+ const clipped = intervals.map((i) => ({ startAt: Math.max(i.startAt, from), endAt: Math.min(i.endAt, to) })).filter((i) => i.endAt > i.startAt).sort((a, b) => a.startAt - b.startAt);
17566
+ let total = 0;
17567
+ let runStart = -Infinity;
17568
+ let runEnd = -Infinity;
17569
+ for (const i of clipped) {
17570
+ if (i.startAt > runEnd) {
17571
+ if (runEnd > runStart) total += runEnd - runStart;
17572
+ runStart = i.startAt;
17573
+ runEnd = i.endAt;
17574
+ } else if (i.endAt > runEnd) {
17575
+ runEnd = i.endAt;
17576
+ }
17577
+ }
17578
+ if (runEnd > runStart) total += runEnd - runStart;
17579
+ return total;
17580
+ }
17581
+ var newTurn = () => ({ windowStart: 0, windowEnd: 0, intervals: [], modelReply: false, repliesPending: 0 });
17582
+ function attachTurnModelClock(session, events, ctx, deps = {}) {
17583
+ const now = deps.now ?? Date.now;
17584
+ const replyWaitMs = deps.replyWaitMs ?? REPLY_METRICS_WAIT_MS;
17585
+ const base = { projectId: ctx.projectId, callId: ctx.callId, ...ctx.campaignId ? { campaignId: ctx.campaignId } : {} };
17586
+ let phase = "idle";
17587
+ let turn = newTurn();
17588
+ let parked = null;
17589
+ let outsidePending = 0;
17590
+ let abandonedPending = 0;
17591
+ let spokeWhileDeciding = false;
17592
+ const safely2 = (fn) => {
17593
+ try {
17594
+ fn();
17595
+ } catch (err) {
17596
+ console.warn("[agent] turn model time record failed", { err: err instanceof Error ? err.message : String(err) });
17597
+ }
17598
+ };
17599
+ const record = (t) => {
17600
+ if (t.repliesPending > 0) return;
17601
+ const latencyMs = modelTimeWithin(t.intervals, t.windowStart, t.windowEnd);
17602
+ if (t.modelReply && latencyMs <= 0) return;
17603
+ safely2(() => {
17604
+ recordTurnLatencySpan({ ...base, stage: "llm", latencyMs, allowZero: true });
17605
+ if (latencyMs > 0) recordTurnLatency({ ...base, firstTokenMs: latencyMs });
17606
+ });
17607
+ };
17608
+ const releaseParked = () => {
17609
+ if (!parked) return;
17610
+ clearTimeout(parked.timer);
17611
+ const t = parked.turn;
17612
+ parked = null;
17613
+ abandonedPending += t.repliesPending;
17614
+ record(t);
17615
+ };
17616
+ const flush = () => {
17617
+ if (phase === "answered") {
17618
+ if (turn.repliesPending > 0) {
17619
+ releaseParked();
17620
+ const t = turn;
17621
+ const timer = setTimeout(releaseParked, replyWaitMs);
17622
+ timer.unref?.();
17623
+ parked = { turn: t, timer };
17624
+ } else {
17625
+ record(turn);
17626
+ }
17627
+ } else {
17628
+ abandonedPending += turn.repliesPending;
17629
+ }
17630
+ phase = "idle";
17631
+ turn = newTurn();
17632
+ spokeWhileDeciding = false;
17633
+ };
17634
+ const at = (event) => {
17635
+ const t = event?.createdAt;
17636
+ return typeof t === "number" ? t : now();
17637
+ };
17638
+ session.on(events.userStateChanged, (event) => {
17639
+ if (event?.newState !== "speaking") return;
17640
+ if (phase === "deciding") {
17641
+ spokeWhileDeciding = true;
17642
+ return;
17643
+ }
17644
+ flush();
17645
+ phase = "listening";
17646
+ });
17647
+ session.on(events.userInputTranscribed, (event) => {
17648
+ if (!event?.isFinal) return;
17649
+ if (phase === "deciding" && spokeWhileDeciding) {
17650
+ turn = { ...newTurn(), modelReply: turn.modelReply, repliesPending: turn.repliesPending, windowStart: at(event) };
17651
+ spokeWhileDeciding = false;
17652
+ return;
17653
+ }
17654
+ if (phase !== "idle" && phase !== "listening") return;
17655
+ turn.windowStart = at(event);
17656
+ phase = "deciding";
17657
+ });
17658
+ session.on(events.agentStateChanged, (event) => {
17659
+ if (event?.newState !== "speaking" || phase !== "deciding") return;
17660
+ turn.windowEnd = Math.max(turn.windowStart, at(event));
17661
+ phase = "answered";
17662
+ spokeWhileDeciding = false;
17663
+ });
17664
+ session.on(events.speechCreated, (event) => {
17665
+ if (event?.source === "say") return;
17666
+ if (phase === "deciding") {
17667
+ turn.modelReply = true;
17668
+ turn.repliesPending += 1;
17669
+ } else if (phase === "answered") {
17670
+ turn.repliesPending += 1;
17671
+ } else {
17672
+ outsidePending += 1;
17673
+ }
17674
+ });
17675
+ const push = (t, interval) => {
17676
+ if (t.intervals.length < MAX_INTERVALS_PER_TURN) t.intervals.push(interval);
17677
+ };
17678
+ return {
17679
+ noteModelTime(interval) {
17680
+ push(turn, interval);
17681
+ },
17682
+ noteReplyModelTime(interval) {
17683
+ if (parked && parked.turn.repliesPending > 0) {
17684
+ push(parked.turn, interval);
17685
+ parked.turn.repliesPending -= 1;
17686
+ if (parked.turn.repliesPending === 0) releaseParked();
17687
+ return;
17688
+ }
17689
+ const inTurn = phase === "deciding" || phase === "answered";
17690
+ if (abandonedPending > 0 && inTurn && interval.startAt < turn.windowStart) {
17691
+ abandonedPending -= 1;
17692
+ return;
17693
+ }
17694
+ if (inTurn && turn.repliesPending > 0) {
17695
+ push(turn, interval);
17696
+ turn.repliesPending -= 1;
17697
+ return;
17698
+ }
17699
+ if (abandonedPending > 0) {
17700
+ abandonedPending -= 1;
17701
+ return;
17702
+ }
17703
+ if (outsidePending > 0 || phase === "idle" || phase === "listening") {
17704
+ outsidePending = Math.max(0, outsidePending - 1);
17705
+ const ttftMs = interval.endAt - interval.startAt;
17706
+ if (ttftMs > 0) {
17707
+ safely2(() => {
17708
+ recordTurnLatency({ ...base, firstTokenMs: ttftMs });
17709
+ recordTurnLatencySpan({ ...base, stage: "llm", latencyMs: ttftMs });
17710
+ });
17711
+ }
17712
+ return;
17713
+ }
17714
+ push(turn, interval);
17715
+ },
17716
+ close() {
17717
+ flush();
17718
+ }
17719
+ };
17720
+ }
17721
+
17722
+ // src/agent.ts
17723
+ init_model_meter();
17724
+
17395
17725
  // src/runtime/vad-tap.ts
17396
17726
  function tapVad(vad, opts) {
17397
17727
  const listeners = /* @__PURE__ */ new Set();
@@ -18417,6 +18747,8 @@ var Agent = class {
18417
18747
  const agentId = typeof md["agentId"] === "string" ? md["agentId"] : void 0;
18418
18748
  const room = job.room.name ?? void 0;
18419
18749
  registerTelemetryFlush(job);
18750
+ const usageSink = createTelemetrySink();
18751
+ const modelMeter = createCallModelMeter(usageSink);
18420
18752
  try {
18421
18753
  return await withCallContext(
18422
18754
  {
@@ -18427,14 +18759,16 @@ var Agent = class {
18427
18759
  ...agentId ? { agentId } : {},
18428
18760
  ...room ? { room } : {}
18429
18761
  },
18430
- () => this.runCallSpan(job)
18762
+ // Every model call the SDK makes on its own during the call (judge, extractor, agent loop, embeddings) reports
18763
+ // into the call's usage and latency — the meter follows all the async work the call starts (burn-down G-40).
18764
+ () => runWithModelMeter(modelMeter.listener, () => this.runCallSpan(job, usageSink, modelMeter))
18431
18765
  );
18432
18766
  } finally {
18433
18767
  await flushTelemetry().catch(() => {
18434
18768
  });
18435
18769
  }
18436
18770
  }
18437
- async runCallSpan(job) {
18771
+ async runCallSpan(job, usageSink, modelMeter) {
18438
18772
  const { AutoSubscribe, voice: voice3, VADEventType } = await import('@livekit/agents');
18439
18773
  const Events = voice3.AgentSessionEventTypes;
18440
18774
  await job.connect(void 0, AutoSubscribe.SUBSCRIBE_ALL);
@@ -18460,6 +18794,7 @@ var Agent = class {
18460
18794
  const identity = admission.resolution;
18461
18795
  const sdkClient = identity.kind === "ok" ? identity.identity.client : null;
18462
18796
  const callCredential = identity.kind === "ok" ? identity.identity.credential : null;
18797
+ configureHelperCallCredential(callCredential?.source ?? null);
18463
18798
  const identityRefused = identity.kind === "refused" && (!!this.config.flowRuntime || identity.reason === "no_project_in_dispatch");
18464
18799
  if (identity.kind === "refused" && !identityRefused) {
18465
18800
  console.warn("[agent] running with no workspace credential (worker-token exchange failed)", { reason: identity.reason });
@@ -19082,12 +19417,12 @@ ${callIntent}`
19082
19417
  }
19083
19418
  }
19084
19419
  };
19085
- const usageSink = createTelemetrySink();
19086
19420
  const callEffects = new EffectStack({
19087
19421
  onDisposeError: (err) => console.warn("[agent] call-effect dispose failed", {
19088
19422
  err: err instanceof Error ? err.message : String(err)
19089
19423
  })
19090
19424
  });
19425
+ let usageReported = Promise.resolve();
19091
19426
  session.once(Events.Close, () => {
19092
19427
  sessionClosed = true;
19093
19428
  endCallController.dispose();
@@ -19099,12 +19434,14 @@ ${callIntent}`
19099
19434
  void offboardParticipant(remaining.id);
19100
19435
  }
19101
19436
  const outcome = callEndOutcome(startedAt);
19102
- if (sdkClient && !usageSink.isEmpty()) {
19103
- void sdkClient.calls.reportUsage(callSync.callId, usageSink.report(outcome.durationMs)).catch((err) => {
19104
- console.warn("[agent] usage report failed", {
19105
- callId: callSync.callId,
19106
- err: err instanceof Error ? err.message : String(err)
19107
- });
19437
+ if (sdkClient) {
19438
+ const calls = sdkClient.calls;
19439
+ usageReported = reportUsageAtClose({
19440
+ meter: modelMeter,
19441
+ usage: usageSink,
19442
+ durationMs: outcome.durationMs,
19443
+ send: (report) => calls.reportUsage(callSync.callId, report),
19444
+ warn: (message, meta) => console.warn(message, { callId: callSync.callId, ...meta })
19108
19445
  });
19109
19446
  }
19110
19447
  void audioExperience.onCallEnd(audioCtx, outcome);
@@ -19243,9 +19580,22 @@ ${callIntent}`
19243
19580
  callId: callSync.callId,
19244
19581
  ...callCampaignId ? { campaignId: callCampaignId } : {}
19245
19582
  };
19583
+ const modelClock = graphMode && !usingRealtime ? attachTurnModelClock(
19584
+ session,
19585
+ {
19586
+ userStateChanged: Events.UserStateChanged,
19587
+ userInputTranscribed: Events.UserInputTranscribed,
19588
+ agentStateChanged: Events.AgentStateChanged,
19589
+ speechCreated: Events.SpeechCreated
19590
+ },
19591
+ metricsCtx
19592
+ ) : void 0;
19593
+ if (modelClock) session.once(Events.Close, () => modelClock.close());
19594
+ modelMeter.bind(metricsCtx, modelClock);
19246
19595
  attachMetricsBridge(session, Events.MetricsCollected, {
19247
19596
  ...metricsCtx,
19248
- usage: usageSink
19597
+ usage: usageSink,
19598
+ ...modelClock ? { modelClock } : {}
19249
19599
  });
19250
19600
  if (graphMode) {
19251
19601
  attachEndpointingProbe(
@@ -19420,6 +19770,7 @@ ${callIntent}`
19420
19770
  }
19421
19771
  }
19422
19772
  await waitForSessionClose;
19773
+ await usageReported;
19423
19774
  } catch (err) {
19424
19775
  console.error("[agent] runJob failed", {
19425
19776
  room: job.room.name ?? null,
@@ -1,4 +1,4 @@
1
- export { b1 as EmailResolver, b2 as FlowResolver, b3 as GraphResolvers, b4 as GraphTraceEvent, b5 as KnowledgeResolver, b6 as LiveTextConversation, b7 as LiveTurnResult, b8 as PlaybookLibraryEntry, b9 as PlaybookResolver, ba as RegistryToolsResolver, bb as RunLiveTextOptions, ag as RunTextSessionOptions, ah as RunTextTranscriptOptions, ay as TextSession, az as TextTranscriptResult, n as ToolExecutor, bc as dedupeConversationEmails, bd as denyPrivateNetwork, be as privateNetworkOptInHonoured, bf as runLiveTextConversation, aZ as runTextSession, a_ as runTextTranscript, bg as runWithOpenAIScope, bh as toPlaybookResolution } from '../text-session-CoyEcdmg.js';
1
+ export { b1 as EmailResolver, b2 as FlowResolver, b3 as GraphResolvers, b4 as GraphTraceEvent, b5 as KnowledgeResolver, b6 as LiveTextConversation, b7 as LiveTurnResult, b8 as PlaybookLibraryEntry, b9 as PlaybookResolver, ba as RegistryToolsResolver, bb as RunLiveTextOptions, ag as RunTextSessionOptions, ah as RunTextTranscriptOptions, ay as TextSession, az as TextTranscriptResult, n as ToolExecutor, bc as dedupeConversationEmails, bd as denyPrivateNetwork, be as privateNetworkOptInHonoured, bf as runLiveTextConversation, aZ as runTextSession, a_ as runTextTranscript, bg as runWithOpenAIScope, bh as toPlaybookResolution } from '../text-session-C4Z59gG4.js';
2
2
  import '@livekit/agents';
3
3
  import 'zod';
4
4
  import '../types-KqrAfY85.js';
@@ -4159,9 +4159,78 @@ var init_dist = __esm({
4159
4159
  init_chat_params();
4160
4160
  }
4161
4161
  });
4162
+ function requestedModel(body) {
4163
+ const model2 = body?.model;
4164
+ return typeof model2 === "string" && model2 ? model2 : void 0;
4165
+ }
4166
+ function safely(fn) {
4167
+ try {
4168
+ fn();
4169
+ } catch (err) {
4170
+ console.warn("[voicelayer] model meter listener failed", { err: err instanceof Error ? err.message : String(err) });
4171
+ }
4172
+ }
4173
+ function metered(target, create, byok) {
4174
+ return async (...args) => {
4175
+ const listener = meterScope.getStore();
4176
+ const startedAt = Date.now();
4177
+ const model2 = requestedModel(args[0]);
4178
+ if (listener) safely(() => listener.started());
4179
+ let res;
4180
+ try {
4181
+ res = await Reflect.apply(create, target, args);
4182
+ return res;
4183
+ } finally {
4184
+ if (listener) {
4185
+ const usage = res?.usage;
4186
+ const record = {
4187
+ provider: "openai",
4188
+ model: model2 ?? res?.model ?? "unknown",
4189
+ ...usage ? { usage: { inputTokens: usage.prompt_tokens ?? 0, outputTokens: usage.completion_tokens ?? 0 } } : {},
4190
+ ...byok !== void 0 ? { byok } : {},
4191
+ startedAt,
4192
+ endedAt: Date.now()
4193
+ };
4194
+ safely(() => listener.finished(record));
4195
+ }
4196
+ }
4197
+ };
4198
+ }
4199
+ function meteredClient(client, byok) {
4200
+ const hit = wrapped.get(client);
4201
+ if (hit) return hit;
4202
+ const chat = client.chat;
4203
+ const embeddings = client.embeddings;
4204
+ const completionsCreate = chat ? metered(chat.completions, chat.completions.create, byok) : void 0;
4205
+ const embeddingsCreate = embeddings ? metered(embeddings, embeddings.create, byok) : void 0;
4206
+ const completions = chat ? new Proxy(chat.completions, { get: (t, p, r) => p === "create" ? completionsCreate : Reflect.get(t, p, r) }) : void 0;
4207
+ const chatProxy = chat ? new Proxy(chat, { get: (t, p, r) => p === "completions" ? completions : Reflect.get(t, p, r) }) : void 0;
4208
+ const embeddingsProxy = embeddings ? new Proxy(embeddings, { get: (t, p, r) => p === "create" ? embeddingsCreate : Reflect.get(t, p, r) }) : void 0;
4209
+ const proxy = new Proxy(client, {
4210
+ get(target, prop, receiver) {
4211
+ if (prop === "chat" && chatProxy) return chatProxy;
4212
+ if (prop === "embeddings" && embeddingsProxy) return embeddingsProxy;
4213
+ return Reflect.get(target, prop, receiver);
4214
+ }
4215
+ });
4216
+ wrapped.set(client, proxy);
4217
+ wrapped.set(proxy, proxy);
4218
+ return proxy;
4219
+ }
4220
+ var meterScope, wrapped;
4221
+ var init_model_meter = __esm({
4222
+ "src/runtime/model-meter.ts"() {
4223
+ meterScope = new AsyncLocalStorage();
4224
+ wrapped = /* @__PURE__ */ new WeakMap();
4225
+ }
4226
+ });
4162
4227
  function runWithOpenAIScope(client, fn) {
4163
4228
  return scope.run(client, fn);
4164
4229
  }
4230
+ function helperKeyIsCustomers() {
4231
+ if (credential) return true;
4232
+ return callCredentialSource !== "worker-token";
4233
+ }
4165
4234
  function defaultOpenAIOptions() {
4166
4235
  if (!credential) return {};
4167
4236
  return {
@@ -4169,27 +4238,31 @@ function defaultOpenAIOptions() {
4169
4238
  ...credential.baseURL ? { baseURL: credential.baseURL } : {}
4170
4239
  };
4171
4240
  }
4172
- function internalChatClient(opts) {
4241
+ function internalChatClient() {
4173
4242
  const cred = defaultOpenAIOptions();
4174
4243
  return createChatClient({
4175
4244
  ...cred.apiKey ? { apiKey: cred.apiKey } : {},
4176
- ...cred.baseURL ? { baseURL: cred.baseURL } : {},
4177
- ...{}
4245
+ ...cred.baseURL ? { baseURL: cred.baseURL } : {}
4178
4246
  });
4179
4247
  }
4180
4248
  function defaultOpenAI() {
4181
- return scope.getStore() ?? (cached ??= internalChatClient());
4249
+ const scoped = scope.getStore();
4250
+ if (scoped) return meteredClient(scoped);
4251
+ if (!cached) cached = meteredClient(internalChatClient(), helperKeyIsCustomers());
4252
+ return cached;
4182
4253
  }
4183
4254
  function hasOpenAiKey() {
4184
4255
  return scope.getStore() !== void 0 || credential !== null || Boolean(process.env["OPENAI_API_KEY"]);
4185
4256
  }
4186
- var credential, cached, scope;
4257
+ var credential, cached, scope, callCredentialSource;
4187
4258
  var init_openai_default = __esm({
4188
4259
  "src/runtime/openai-default.ts"() {
4189
4260
  init_dist();
4261
+ init_model_meter();
4190
4262
  credential = null;
4191
4263
  cached = null;
4192
4264
  scope = new AsyncLocalStorage();
4265
+ callCredentialSource = null;
4193
4266
  }
4194
4267
  });
4195
4268
  var init_ssrf = __esm({