npm - vent-hq - Versions diffs - 0.9.1 → 0.9.2 - Mend

vent-hq 0.9.1 → 0.9.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (5) hide show

package/dist/index.mjs +758 -418
package/dist/{package-ON25XLSL.mjs → package-767KASWC.mjs} +5 -5
package/dist/{package-YOCP6D2K.mjs → package-BTQ6XCVY.mjs} +3 -3
package/package.json +5 -5
package/dist/package-YS3UNSUV.mjs +0 -51

package/dist/index.mjs CHANGED Viewed

@@ -73,8 +73,8 @@ import * as path from "node:path";
 import { homedir } from "node:os";
 var CONFIG_DIR = path.join(homedir(), ".vent");
 var CREDENTIALS_FILE = path.join(CONFIG_DIR, "credentials");
-var API_BASE = process.env.VENT_API_URL ?? "https://vent-api.fly.dev";
-var DASHBOARD_URL = process.env.VENT_DASHBOARD_URL ?? "https://ventmcp.dev";
+var API_BASE = process.env.VENT_API_URL ?? "https://api.venthq.dev";
+var DASHBOARD_URL = process.env.VENT_DASHBOARD_URL ?? "https://venthq.dev";
 async function loadAccessToken() {
   if (process.env.VENT_ACCESS_TOKEN) return process.env.VENT_ACCESS_TOKEN;
   try {
@@ -152,372 +152,10 @@ function openBrowser(url) {
 // src/lib/output.ts
 import { writeFileSync } from "node:fs";
-var isTTY = process.stdout.isTTY;
-var _verbose = false;
-function debug(msg) {
-  if (!_verbose) return;
-  const ts = (/* @__PURE__ */ new Date()).toISOString().slice(11, 23);
-  process.stderr.write(`[vent ${ts}] ${msg}
-`);
-}
-function isVerbose() {
-  return _verbose;
-}
-function stdoutSync(data) {
-  if (isTTY) {
-    process.stdout.write(data);
-  } else {
-    try {
-      writeFileSync(1, data);
-    } catch {
-      process.stdout.write(data);
-    }
-  }
-}
-var bold = (s) => isTTY ? `\x1B[1m${s}\x1B[0m` : s;
-var dim = (s) => isTTY ? `\x1B[2m${s}\x1B[0m` : s;
-var green = (s) => isTTY ? `\x1B[32m${s}\x1B[0m` : s;
-var red = (s) => isTTY ? `\x1B[31m${s}\x1B[0m` : s;
-var blue = (s) => isTTY ? `\x1B[34m${s}\x1B[0m` : s;
-function printEvent(event) {
-  if (!isTTY) return;
-  const meta = event.metadata_json ?? {};
-  switch (event.event_type) {
-    case "call_completed":
-      printCallResult(meta);
-      break;
-    case "run_complete":
-      printRunComplete(meta);
-      break;
-    case "call_started": {
-      const name = meta.call_name ?? "call";
-      process.stderr.write(dim(`  \u25B8 ${name}\u2026`) + "\n");
-      break;
-    }
-    default:
-      process.stderr.write(dim(`  [${event.event_type}]`) + "\n");
-  }
-}
-function printCallResult(meta) {
-  const result = meta.result;
-  const callName = result?.name ?? meta.call_name ?? "call";
-  const callStatus = result?.status ?? meta.status;
-  const durationMs = result?.duration_ms ?? meta.duration_ms;
-  const statusIcon = callStatus === "completed" || callStatus === "pass" ? green("\u2714") : red("\u2718");
-  const duration = durationMs != null ? (durationMs / 1e3).toFixed(1) + "s" : "\u2014";
-  const parts = [statusIcon, bold(callName), dim(duration)];
-  if (result?.latency?.p50_response_time_ms != null) {
-    parts.push(`p50: ${result.latency.p50_response_time_ms}ms`);
-  }
-  if (result?.call_metadata?.transfer_attempted) {
-    const transferLabel = result.call_metadata.transfer_completed ? "transfer: completed" : "transfer: attempted";
-    parts.push(transferLabel);
-  }
-  stdoutSync(parts.join("  ") + "\n");
-}
-function printRunComplete(meta) {
-  const status = meta.status;
-  const agg = meta.aggregate;
-  const counts = agg?.conversation_calls;
-  const total = meta.total_calls ?? counts?.total;
-  const passed = meta.passed_calls ?? counts?.passed;
-  const failed = meta.failed_calls ?? counts?.failed;
-  stdoutSync("\n");
-  if (status === "pass") {
-    stdoutSync(green(bold("Run passed")) + "\n");
-  } else {
-    stdoutSync(red(bold("Run failed")) + "\n");
-  }
-  if (total != null) {
-    const parts = [];
-    if (passed) parts.push(green(`${passed} passed`));
-    if (failed) parts.push(red(`${failed} failed`));
-    parts.push(`${total} total`);
-    stdoutSync(parts.join(dim(" \xB7 ")) + "\n");
-  }
-}
-function printSummary(callResults, runComplete, runId) {
-  const allCalls = callResults.map((e) => {
-    const meta = e.metadata_json ?? {};
-    const r = meta.result;
-    if (r) return r;
-    return {
-      name: meta.call_name ?? "call",
-      status: meta.status ?? "unknown",
-      duration_ms: meta.duration_ms,
-      error: null
-    };
-  });
-  const agg = runComplete.aggregate;
-  const counts = agg?.conversation_calls;
-  const summaryData = {
-    run_id: runId,
-    status: runComplete.status,
-    total: runComplete.total_calls ?? counts?.total,
-    passed: runComplete.passed_calls ?? counts?.passed,
-    failed: runComplete.failed_calls ?? counts?.failed,
-    calls: allCalls
-  };
-  if (!isTTY) {
-    stdoutSync(JSON.stringify(summaryData, null, 2) + "\n");
-    return;
-  }
-  const failures = allCalls.filter((t2) => t2.status && t2.status !== "completed" && t2.status !== "pass");
-  if (failures.length > 0) {
-    stdoutSync("\n" + bold("Failed calls:") + "\n");
-    for (const t2 of failures) {
-      const duration = t2.duration_ms != null ? (t2.duration_ms / 1e3).toFixed(1) + "s" : "\u2014";
-      const parts = [red("\u2718"), bold(t2.name ?? "call"), dim(duration)];
-      stdoutSync("  " + parts.join("  ") + "\n");
-    }
-  }
-  process.stderr.write(dim(`Full details: vent status ${runId} --json`) + "\n");
-}
-function printError(message) {
-  const line = red(bold("error")) + ` ${message}
-`;
-  process.stderr.write(line);
-  if (!isTTY) {
-    stdoutSync(line);
-  }
-}
-function printInfo(message, { force } = {}) {
-  if (!force && !isTTY && !_verbose) return;
-  const line = blue("\u25B8") + ` ${message}
-`;
-  process.stderr.write(line);
-  if (!isTTY && force) stdoutSync(line);
-}
-function printSuccess(message, { force } = {}) {
-  if (!force && !isTTY && !_verbose) return;
-  const line = green("\u2714") + ` ${message}
-`;
-  process.stderr.write(line);
-  if (!isTTY && force) stdoutSync(line);
-}
-// src/lib/auth.ts
-var POLL_INTERVAL_MS = 2e3;
-function sleep(ms) {
-  return new Promise((r) => setTimeout(r, ms));
-}
-async function deviceAuthFlow() {
-  let startData;
-  try {
-    const res = await fetch(`${API_BASE}/device/start`, { method: "POST" });
-    if (!res.ok) {
-      return { ok: false, error: `Failed to start device auth: ${res.status}` };
-    }
-    startData = await res.json();
-  } catch {
-    return { ok: false, error: "Could not reach Vent API. Check your connection." };
-  }
-  printInfo(`Your authorization code: ${startData.user_code}`, { force: true });
-  printInfo(`Opening browser to log in...`, { force: true });
-  printInfo(`If the browser doesn't open, visit: ${startData.verification_url}`, { force: true });
-  openBrowser(startData.verification_url);
-  const deadline = new Date(startData.expires_at).getTime();
-  while (Date.now() < deadline) {
-    await sleep(POLL_INTERVAL_MS);
-    try {
-      const res = await fetch(`${API_BASE}/device/exchange`, {
-        method: "POST",
-        headers: { "Content-Type": "application/json" },
-        body: JSON.stringify({ session_id: startData.session_id })
-      });
-      if (!res.ok) continue;
-      const data = await res.json();
-      const accessToken = data.access_token;
-      if (data.status === "approved" && accessToken) {
-        await saveAccessToken(accessToken);
-        return { ok: true, accessToken };
-      }
-      if (data.status === "expired") {
-        return { ok: false, error: "Session expired. Run `npx vent-hq login` again." };
-      }
-      if (data.status === "consumed" || data.status === "invalid") {
-        return { ok: false, error: "Session invalid. Run `npx vent-hq login` again." };
-      }
-    } catch {
-    }
-  }
-  return { ok: false, error: "Login timed out. Run `npx vent-hq login` again." };
-}
-// src/lib/sse.ts
-function log(msg) {
-  if (!isVerbose()) return;
-  const ts = (/* @__PURE__ */ new Date()).toISOString().slice(11, 23);
-  const line = `[vent:sse ${ts}] ${msg}
-`;
-  process.stderr.write(line);
-}
-var MAX_RETRIES = 5;
-var RETRY_DELAY_MS = 2e3;
-async function* streamRunEvents(runId, apiKey, signal) {
-  const url = `${API_BASE}/runs/${runId}/stream`;
-  const seenIds = /* @__PURE__ */ new Set();
-  let retries = 0;
-  while (retries <= MAX_RETRIES) {
-    if (retries > 0) {
-      log(`reconnecting (attempt ${retries}/${MAX_RETRIES}) after ${RETRY_DELAY_MS}ms\u2026`);
-      await new Promise((r) => setTimeout(r, RETRY_DELAY_MS));
-    }
-    log(`connecting to ${url}`);
-    let res;
-    try {
-      res = await fetch(url, {
-        headers: { Authorization: `Bearer ${apiKey}` },
-        signal
-      });
-    } catch (err) {
-      if (err.name === "AbortError") throw err;
-      log(`fetch error: ${err.message}`);
-      retries++;
-      continue;
-    }
-    log(`response: status=${res.status} content-type=${res.headers.get("content-type")}`);
-    if (!res.ok) {
-      const body = await res.text();
-      log(`error body: ${body}`);
-      throw new Error(`SSE stream failed (${res.status}): ${body}`);
-    }
-    if (!res.body) {
-      throw new Error("SSE stream returned no body");
-    }
-    const reader = res.body.getReader();
-    const decoder = new TextDecoder();
-    let buffer = "";
-    let chunkCount = 0;
-    let eventCount = 0;
-    let gotRunComplete = false;
-    let streamError = null;
-    try {
-      while (true) {
-        let readResult;
-        try {
-          readResult = await reader.read();
-        } catch (err) {
-          if (err.name === "AbortError") throw err;
-          streamError = err;
-          log(`read error: ${streamError.message}`);
-          break;
-        }
-        const { done, value } = readResult;
-        if (done) {
-          log(`stream done after ${chunkCount} chunks, ${eventCount} events`);
-          break;
-        }
-        chunkCount++;
-        const chunk = decoder.decode(value, { stream: true });
-        buffer += chunk;
-        if (chunkCount <= 3 || chunkCount % 10 === 0) {
-          log(`chunk #${chunkCount} (${chunk.length} bytes) buffer=${buffer.length} bytes`);
-        }
-        const lines = buffer.split("\n");
-        buffer = lines.pop();
-        for (const line of lines) {
-          if (line.startsWith("data: ")) {
-            const raw = line.slice(6);
-            try {
-              const event = JSON.parse(raw);
-              eventCount++;
-              if (event.id && seenIds.has(event.id)) {
-                log(`skipping duplicate event ${event.id}`);
-                continue;
-              }
-              if (event.id) seenIds.add(event.id);
-              log(`parsed event #${eventCount}: type=${event.event_type}`);
-              yield event;
-              if (event.event_type === "run_complete") {
-                log("run_complete received \u2014 closing stream");
-                gotRunComplete = true;
-                return;
-              }
-            } catch {
-              log(`malformed JSON: ${raw.slice(0, 200)}`);
-            }
-          } else if (line.startsWith(": ")) {
-            if (chunkCount <= 3) {
-              log(`heartbeat: "${line}"`);
-            }
-          }
-        }
-      }
-    } finally {
-      reader.releaseLock();
-      log("reader released");
-    }
-    if (gotRunComplete) return;
-    retries++;
-    if (retries <= MAX_RETRIES) {
-      log(`stream ended without run_complete \u2014 will retry (${retries}/${MAX_RETRIES})`);
-    }
-  }
-  log(`exhausted ${MAX_RETRIES} retries without run_complete`);
-}
-// src/lib/run-history.ts
-import * as fs2 from "node:fs/promises";
-import * as path2 from "node:path";
-import { execSync } from "node:child_process";
-function gitInfo() {
-  try {
-    const sha = execSync("git rev-parse HEAD", { encoding: "utf-8", stdio: ["pipe", "pipe", "pipe"] }).trim();
-    const branch = execSync("git branch --show-current", { encoding: "utf-8", stdio: ["pipe", "pipe", "pipe"] }).trim() || null;
-    const status = execSync("git status --porcelain", { encoding: "utf-8", stdio: ["pipe", "pipe", "pipe"] }).trim();
-    return { sha, branch, dirty: status.length > 0 };
-  } catch {
-    return { sha: null, branch: null, dirty: false };
-  }
-}
-async function saveRunHistory(runId, callResults, runCompleteData) {
-  try {
-    const dir = path2.join(process.cwd(), ".vent", "runs");
-    await fs2.mkdir(dir, { recursive: true });
-    const git = gitInfo();
-    const now = /* @__PURE__ */ new Date();
-    const timestamp = now.toISOString().replace(/[:.]/g, "-").slice(0, 19);
-    const shortId = runId.slice(0, 8);
-    const aggregate = runCompleteData.aggregate;
-    const convCalls = aggregate?.conversation_calls;
-    const total = convCalls?.total ?? 0;
-    const passed = convCalls?.passed ?? 0;
-    const failed = convCalls?.failed ?? 0;
-    const entry = {
-      run_id: runId,
-      timestamp: now.toISOString(),
-      git_sha: git.sha,
-      git_branch: git.branch,
-      git_dirty: git.dirty,
-      summary: {
-        status: runCompleteData.status ?? "unknown",
-        calls_total: total,
-        calls_passed: passed,
-        calls_failed: failed,
-        total_duration_ms: aggregate?.total_duration_ms,
-        total_cost_usd: aggregate?.total_cost_usd
-      },
-      call_results: callResults.map((e) => e.metadata_json ?? {})
-    };
-    const filename = `${timestamp}_${shortId}.json`;
-    const filepath = path2.join(dir, filename);
-    await fs2.writeFile(filepath, JSON.stringify(entry, null, 2) + "\n");
-    return filepath;
-  } catch {
-    return null;
-  }
-}
 // ../shared/src/types.ts
-var AUDIO_CALL_NAMES = [
-  "audio_quality",
-  "latency",
-  "echo"
-];
 var AUDIO_ACTION_TYPES = [
   "interrupt",
-  "silence",
   "inject_noise",
   "split_sentence",
   "noise_on_caller"
@@ -4565,7 +4203,6 @@ var coerce = {
 var NEVER = INVALID;
 // ../shared/src/schemas.ts
-var AudioCallNameSchema = external_exports.enum(AUDIO_CALL_NAMES);
 var AudioActionSchema = external_exports.object({
   at_turn: external_exports.number().int().min(0),
   action: external_exports.enum(AUDIO_ACTION_TYPES),
@@ -4631,6 +4268,7 @@ var ObservedToolCallSchema = external_exports.object({
   arguments: external_exports.record(external_exports.unknown()),
   result: external_exports.unknown().optional(),
   successful: external_exports.boolean().optional(),
+  provider_tool_type: external_exports.string().optional(),
   timestamp_ms: external_exports.number().optional(),
   latency_ms: external_exports.number().optional(),
   turn_index: external_exports.number().int().min(0).optional()
@@ -4732,19 +4370,6 @@ var AudioAnalysisWarningSchema = external_exports.object({
   severity: external_exports.enum(["warning", "critical"]),
   message: external_exports.string()
 });
-var CallDiagnosticsSchema = external_exports.object({
-  error_origin: external_exports.enum(["platform", "agent"]).nullable(),
-  error_detail: external_exports.string().nullable(),
-  timing: external_exports.object({
-    channel_connect_ms: external_exports.number()
-  }),
-  channel: external_exports.object({
-    connected: external_exports.boolean(),
-    error_events: external_exports.array(external_exports.string()),
-    audio_bytes_sent: external_exports.number(),
-    audio_bytes_received: external_exports.number()
-  })
-});
 var ConversationTurnSchema = external_exports.object({
   role: external_exports.enum(["caller", "agent"]),
   text: external_exports.string(),
@@ -4760,7 +4385,8 @@ var ConversationTurnSchema = external_exports.object({
   component_latency: external_exports.object({
     stt_ms: external_exports.number().optional(),
     llm_ms: external_exports.number().optional(),
-    tts_ms: external_exports.number().optional()
+    tts_ms: external_exports.number().optional(),
+    speech_duration_ms: external_exports.number().optional()
   }).optional(),
   platform_transcript: external_exports.string().optional(),
   interrupted: external_exports.boolean().optional(),
@@ -4772,7 +4398,8 @@ var HallucinationEventSchema = external_exports.object({
   hypothesis_text: external_exports.string()
 });
 var TranscriptMetricsSchema = external_exports.object({
-  wer: external_exports.number().min(0).max(1).optional(),
+  wer: external_exports.number().min(0).optional(),
+  cer: external_exports.number().min(0).optional(),
   hallucination_events: external_exports.array(HallucinationEventSchema).optional(),
   repetition_score: external_exports.number().min(0).max(1).optional(),
   reprompt_count: external_exports.number().int().min(0).optional(),
@@ -4807,7 +4434,7 @@ var AudioAnalysisMetricsSchema = external_exports.object({
   talk_ratio_vad: external_exports.number(),
   interruption_rate: external_exports.number().min(0).max(1),
   interruption_count: external_exports.number().int().min(0),
-  barge_in_recovery_time_ms: external_exports.number().min(0).optional(),
+  agent_overtalk_after_barge_in_ms: external_exports.number().min(0).optional(),
   agent_interrupting_user_rate: external_exports.number().min(0).max(1),
   agent_interrupting_user_count: external_exports.number().int().min(0),
   missed_response_windows: external_exports.number().int().min(0),
@@ -4885,6 +4512,11 @@ var CostBreakdownSchema = external_exports.object({
   llm_prompt_tokens: external_exports.number().int().optional(),
   llm_completion_tokens: external_exports.number().int().optional()
 });
+var ProviderWarningSchema = external_exports.object({
+  message: external_exports.string().optional(),
+  code: external_exports.string().optional(),
+  detail: external_exports.unknown().optional()
+});
 var CallTransferSchema = external_exports.object({
   type: external_exports.string(),
   destination: external_exports.string().optional(),
@@ -4894,16 +4526,17 @@ var CallTransferSchema = external_exports.object({
 });
 var CallMetadataSchema = external_exports.object({
   platform: external_exports.string(),
+  provider_call_id: external_exports.string().optional(),
+  provider_session_id: external_exports.string().optional(),
   ended_reason: external_exports.string().optional(),
-  duration_s: external_exports.number().optional(),
   cost_usd: external_exports.number().optional(),
   cost_breakdown: CostBreakdownSchema.optional(),
-  recording_url: external_exports.string().nullable().optional(),
-  summary: external_exports.string().nullable().optional(),
-  success_evaluation: external_exports.string().nullable().optional(),
-  user_sentiment: external_exports.string().nullable().optional(),
-  call_successful: external_exports.boolean().optional(),
+  recording_url: external_exports.string().optional(),
+  recording_variants: external_exports.record(external_exports.string()).optional(),
+  provider_debug_urls: external_exports.record(external_exports.string()).optional(),
   variables: external_exports.record(external_exports.unknown()).optional(),
+  provider_warnings: external_exports.array(ProviderWarningSchema).optional(),
+  provider_metadata: external_exports.record(external_exports.unknown()).optional(),
   transfers: external_exports.array(CallTransferSchema).optional()
 });
 var ConversationMetricsSchema = external_exports.object({
@@ -4920,15 +4553,6 @@ var ConversationMetricsSchema = external_exports.object({
   harness_overhead: HarnessOverheadSchema.optional(),
   component_latency: ComponentLatencyMetricsSchema.optional()
 });
-var AudioCallResultSchema = external_exports.object({
-  call_name: AudioCallNameSchema,
-  status: external_exports.enum(["completed", "error"]),
-  metrics: external_exports.record(external_exports.union([external_exports.number(), external_exports.boolean(), external_exports.array(external_exports.number())])),
-  transcriptions: external_exports.record(external_exports.union([external_exports.string(), external_exports.array(external_exports.string()), external_exports.null()])),
-  duration_ms: external_exports.number(),
-  error: external_exports.string().optional(),
-  diagnostics: CallDiagnosticsSchema.optional()
-});
 var ConversationCallResultSchema = external_exports.object({
   name: external_exports.string().optional(),
   caller_prompt: external_exports.string(),
@@ -4958,6 +4582,653 @@ var RunnerCallbackV2Schema = external_exports.object({
   error_text: external_exports.string().optional()
 });
+// ../shared/src/format-result.ts
+function formatConversationResult(raw, options = {}) {
+  if (!raw || typeof raw !== "object") return null;
+  const r = raw;
+  if (typeof r.caller_prompt !== "string") return null;
+  const debug2 = options.verbose ? formatDebug(r) : void 0;
+  return {
+    name: r.name ?? null,
+    status: r.status,
+    caller_prompt: r.caller_prompt,
+    duration_ms: r.duration_ms,
+    error: r.error ?? null,
+    transcript: formatTranscript(r.transcript, options),
+    latency: r.metrics?.latency ? formatLatency(r.metrics.latency, r.metrics) : null,
+    transcript_quality: r.metrics?.transcript && hasContent(r.metrics.transcript) ? r.metrics.transcript : null,
+    audio_analysis: r.metrics?.audio_analysis && hasContent(r.metrics.audio_analysis) ? formatAudioAnalysis(r.metrics.audio_analysis) : null,
+    tool_calls: formatToolCalls(r.metrics?.tool_calls, r.observed_tool_calls),
+    component_latency: formatComponentLatency(r.metrics?.component_latency),
+    call_metadata: formatCallMetadata(r.call_metadata),
+    warnings: dedupeStrings([
+      ...(r.metrics?.audio_analysis_warnings ?? []).map((w) => w.message),
+      ...(r.metrics?.prosody_warnings ?? []).map((w) => w.message),
+      ...formatProviderWarningMessages(r.call_metadata?.provider_warnings)
+    ]),
+    audio_actions: r.audio_action_results ?? [],
+    emotion: r.metrics?.prosody ? formatEmotion(r.metrics.prosody) : null,
+    ...debug2 ? { debug: debug2 } : {}
+  };
+}
+function formatTranscript(turns, options) {
+  if (!turns) return [];
+  return turns.map((t2) => {
+    const turn = {
+      role: t2.role,
+      text: t2.text
+    };
+    if (t2.ttfb_ms != null) turn.ttfb_ms = t2.ttfb_ms;
+    if (t2.ttfw_ms != null) turn.ttfw_ms = t2.ttfw_ms;
+    if (t2.audio_duration_ms != null) turn.audio_duration_ms = t2.audio_duration_ms;
+    if (t2.interrupted != null) turn.interrupted = t2.interrupted;
+    if (t2.is_interruption != null) turn.is_interruption = t2.is_interruption;
+    if (options.verbose) {
+      const debug2 = compactUnknownRecord({
+        timestamp_ms: t2.timestamp_ms,
+        caller_decision_mode: t2.caller_decision_mode,
+        silence_pad_ms: t2.silence_pad_ms,
+        stt_confidence: t2.stt_confidence,
+        harness_tts_ms: t2.tts_ms,
+        harness_stt_ms: t2.stt_ms,
+        component_latency: t2.component_latency,
+        platform_transcript: t2.platform_transcript
+      });
+      if (debug2 && Object.keys(debug2).length > 0) {
+        turn.debug = debug2;
+      }
+    }
+    return turn;
+  });
+}
+function formatLatency(latency, metrics) {
+  const hasTtfw = metrics.mean_ttfw_ms != null && latency.p50_ttfw_ms != null && latency.p95_ttfw_ms != null;
+  const responseTimeSource = hasTtfw ? "ttfw" : "ttfb";
+  const result = {
+    response_time_ms: hasTtfw ? metrics.mean_ttfw_ms : metrics.mean_ttfb_ms,
+    response_time_source: responseTimeSource,
+    p50_response_time_ms: hasTtfw ? latency.p50_ttfw_ms : latency.p50_ttfb_ms,
+    p90_response_time_ms: hasTtfw ? latency.p90_ttfw_ms ?? latency.p90_ttfb_ms : latency.p90_ttfb_ms,
+    p95_response_time_ms: hasTtfw ? latency.p95_ttfw_ms : latency.p95_ttfb_ms,
+    p99_response_time_ms: hasTtfw ? latency.p99_ttfw_ms ?? latency.p99_ttfb_ms : latency.p99_ttfb_ms,
+    first_response_time_ms: hasTtfw ? latency.first_turn_ttfw_ms ?? latency.first_turn_ttfb_ms : latency.first_turn_ttfb_ms,
+    total_silence_ms: latency.total_silence_ms,
+    mean_turn_gap_ms: latency.mean_turn_gap_ms
+  };
+  if (hasTtfw) {
+    result.mean_ttfw_ms = metrics.mean_ttfw_ms;
+    result.p50_ttfw_ms = latency.p50_ttfw_ms;
+    result.p90_ttfw_ms = latency.p90_ttfw_ms ?? latency.p90_ttfb_ms;
+    result.p95_ttfw_ms = latency.p95_ttfw_ms;
+    result.p99_ttfw_ms = latency.p99_ttfw_ms ?? latency.p99_ttfb_ms;
+    result.first_turn_ttfw_ms = latency.first_turn_ttfw_ms ?? latency.first_turn_ttfb_ms;
+  }
+  if (latency.drift_slope_ms_per_turn != null) result.drift_slope_ms_per_turn = latency.drift_slope_ms_per_turn;
+  if (latency.mean_silence_pad_ms != null) result.mean_silence_pad_ms = latency.mean_silence_pad_ms;
+  if (latency.mouth_to_ear_est_ms != null) result.mouth_to_ear_est_ms = latency.mouth_to_ear_est_ms;
+  return result;
+}
+function formatAudioAnalysis(audio) {
+  return {
+    caller_talk_time_ms: audio.caller_talk_time_ms,
+    agent_talk_time_ms: audio.agent_talk_time_ms,
+    agent_speech_ratio: audio.agent_speech_ratio,
+    talk_ratio_vad: audio.talk_ratio_vad,
+    interruption_rate: audio.interruption_rate,
+    interruption_count: audio.interruption_count,
+    agent_overtalk_after_barge_in_ms: audio.agent_overtalk_after_barge_in_ms,
+    agent_interrupting_user_rate: audio.agent_interrupting_user_rate,
+    agent_interrupting_user_count: audio.agent_interrupting_user_count,
+    missed_response_windows: audio.missed_response_windows,
+    longest_monologue_ms: audio.longest_monologue_ms,
+    silence_gaps_over_2s: audio.silence_gaps_over_2s,
+    total_internal_silence_ms: audio.total_internal_silence_ms,
+    mean_agent_speech_segment_ms: audio.mean_agent_speech_segment_ms
+  };
+}
+function formatToolCalls(summary, observed) {
+  return {
+    total: summary?.total ?? observed?.length ?? 0,
+    successful: summary?.successful ?? observed?.filter((c) => c.successful).length ?? 0,
+    failed: summary?.failed ?? observed?.filter((c) => c.successful === false).length ?? 0,
+    mean_latency_ms: summary?.mean_latency_ms,
+    names: summary?.names ?? [...new Set((observed ?? []).map((c) => c.name))],
+    observed: (observed ?? []).map((c) => ({
+      name: c.name,
+      arguments: c.arguments,
+      result: c.result,
+      successful: c.successful,
+      provider_tool_type: c.provider_tool_type,
+      latency_ms: c.latency_ms,
+      turn_index: c.turn_index
+    }))
+  };
+}
+function formatEmotion(prosody) {
+  return {
+    naturalness: prosody.naturalness,
+    mean_calmness: prosody.mean_calmness,
+    mean_confidence: prosody.mean_confidence,
+    peak_frustration: prosody.peak_frustration,
+    emotion_trajectory: prosody.emotion_trajectory
+  };
+}
+function formatComponentLatency(cl) {
+  if (!cl) return null;
+  const speechDurations = cl.per_turn.map((t2) => t2.speech_duration_ms).filter((v) => v != null);
+  const meanSpeech = speechDurations.length > 0 ? Math.round(speechDurations.reduce((a, b) => a + b, 0) / speechDurations.length) : void 0;
+  return {
+    mean_stt_ms: cl.mean_stt_ms,
+    mean_llm_ms: cl.mean_llm_ms,
+    mean_tts_ms: cl.mean_tts_ms,
+    p95_stt_ms: cl.p95_stt_ms,
+    p95_llm_ms: cl.p95_llm_ms,
+    p95_tts_ms: cl.p95_tts_ms,
+    mean_speech_duration_ms: meanSpeech,
+    bottleneck: cl.bottleneck
+  };
+}
+function formatCallMetadata(meta) {
+  if (!meta) return null;
+  const transfers = meta.transfers?.map((transfer) => {
+    const formattedTransfer = {
+      type: transfer.type,
+      destination: transfer.destination,
+      status: transfer.status,
+      sources: transfer.sources
+    };
+    if (transfer.timestamp_ms != null) {
+      formattedTransfer.timestamp_ms = transfer.timestamp_ms;
+    }
+    return formattedTransfer;
+  });
+  const result = {
+    platform: meta.platform,
+    provider_call_id: meta.provider_call_id,
+    provider_session_id: meta.provider_session_id,
+    ended_reason: meta.ended_reason,
+    cost_usd: meta.cost_usd,
+    cost_breakdown: meta.cost_breakdown,
+    recording_url: meta.recording_url,
+    recording_variants: meta.recording_variants,
+    provider_debug_urls: meta.provider_debug_urls,
+    variables: meta.variables
+  };
+  if (transfers && transfers.length > 0) {
+    const completedTransferCount = transfers.filter((transfer) => transfer.status === "completed").length;
+    const transferCompleted = completedTransferCount > 0;
+    result.transfer_attempted = true;
+    result.transfer_completed = transferCompleted;
+    result.escalated = transferCompleted;
+    result.transfer_count = transfers.length;
+    result.completed_transfer_count = completedTransferCount;
+    result.transfers = transfers;
+  }
+  return result;
+}
+function formatDebug(result) {
+  const debug2 = compactUnknownRecord({
+    signal_quality: result.metrics?.signal_quality,
+    harness_overhead: result.metrics?.harness_overhead,
+    prosody: result.metrics?.prosody,
+    audio_analysis_warnings: nonEmptyArray(result.metrics?.audio_analysis_warnings),
+    prosody_warnings: nonEmptyArray(result.metrics?.prosody_warnings),
+    provider_warnings: nonEmptyArray(result.call_metadata?.provider_warnings),
+    component_latency_per_turn: nonEmptyArray(result.metrics?.component_latency?.per_turn),
+    observed_tool_calls: formatDebugToolCalls(result.observed_tool_calls),
+    provider_metadata: result.call_metadata?.provider_metadata
+  });
+  return debug2 && Object.keys(debug2).length > 0 ? debug2 : void 0;
+}
+function formatDebugToolCalls(observed) {
+  if (!observed || observed.length === 0) return void 0;
+  return observed.map((call) => ({
+    name: call.name,
+    arguments: call.arguments,
+    result: call.result,
+    successful: call.successful,
+    provider_tool_type: call.provider_tool_type,
+    timestamp_ms: call.timestamp_ms,
+    latency_ms: call.latency_ms,
+    turn_index: call.turn_index
+  }));
+}
+function nonEmptyArray(value) {
+  return value && value.length > 0 ? value : void 0;
+}
+function formatProviderWarningMessages(warnings) {
+  if (!warnings || warnings.length === 0) return [];
+  return warnings.map((warning) => warning.message ?? warning.code).filter((message) => typeof message === "string" && message.length > 0);
+}
+function dedupeStrings(values) {
+  return [...new Set(values)];
+}
+function compactUnknownRecord(record) {
+  const entries = Object.entries(record).filter(([, value]) => value != null);
+  return entries.length > 0 ? Object.fromEntries(entries) : void 0;
+}
+function hasContent(obj) {
+  return Object.values(obj).some((v) => v != null);
+}
+// src/lib/output.ts
+var isTTY = process.stdout.isTTY;
+var _verbose = false;
+function debug(msg) {
+  if (!_verbose) return;
+  const ts = (/* @__PURE__ */ new Date()).toISOString().slice(11, 23);
+  stdoutSync(`[vent ${ts}] ${msg}
+`);
+}
+function isVerbose() {
+  return _verbose;
+}
+function stdoutSync(data) {
+  if (isTTY) {
+    process.stdout.write(data);
+  } else {
+    try {
+      writeFileSync(1, data);
+    } catch {
+      process.stdout.write(data);
+    }
+  }
+}
+function writeJsonStdout(value) {
+  stdoutSync(JSON.stringify(value, null, 2) + "\n");
+}
+var bold = (s) => isTTY ? `\x1B[1m${s}\x1B[0m` : s;
+var dim = (s) => isTTY ? `\x1B[2m${s}\x1B[0m` : s;
+var green = (s) => isTTY ? `\x1B[32m${s}\x1B[0m` : s;
+var red = (s) => isTTY ? `\x1B[31m${s}\x1B[0m` : s;
+var yellow = (s) => isTTY ? `\x1B[33m${s}\x1B[0m` : s;
+var blue = (s) => isTTY ? `\x1B[34m${s}\x1B[0m` : s;
+function printEvent(event) {
+  if (!isTTY) return;
+  const meta = event.metadata_json ?? {};
+  switch (event.event_type) {
+    case "call_completed":
+      printCallResult(meta);
+      break;
+    case "run_complete":
+      printRunComplete(meta);
+      break;
+    case "call_started": {
+      const name = meta.call_name ?? "call";
+      stdoutSync(dim(`  \u25B8 ${name}\u2026`) + "\n");
+      break;
+    }
+    default:
+      stdoutSync(dim(`  [${event.event_type}]`) + "\n");
+  }
+}
+function printCallResult(meta) {
+  const result = meta.result;
+  const callName = result?.name ?? meta.call_name ?? "call";
+  const callStatus = result?.status ?? meta.status;
+  const durationMs = result?.duration_ms ?? meta.duration_ms;
+  const statusIcon = callStatus === "completed" || callStatus === "pass" ? green("\u2714") : red("\u2718");
+  const duration = durationMs != null ? (durationMs / 1e3).toFixed(1) + "s" : "\u2014";
+  const parts = [statusIcon, bold(callName), dim(duration)];
+  if (result?.latency?.p50_response_time_ms != null) {
+    parts.push(`p50: ${result.latency.p50_response_time_ms}ms`);
+  }
+  if (result?.call_metadata?.transfer_attempted) {
+    const transferLabel = result.call_metadata.transfer_completed ? "transfer: completed" : "transfer: attempted";
+    parts.push(transferLabel);
+  }
+  stdoutSync(parts.join("  ") + "\n");
+  const providerCallId = result?.call_metadata?.provider_call_id;
+  const providerSessionId = result?.call_metadata?.provider_session_id;
+  if (providerCallId) {
+    stdoutSync(dim(`    provider id: ${providerCallId}`) + "\n");
+  } else if (providerSessionId) {
+    stdoutSync(dim(`    provider session: ${providerSessionId}`) + "\n");
+  }
+  const recordingUrl = result?.call_metadata?.recording_url;
+  if (recordingUrl) {
+    stdoutSync(dim(`    recording: ${recordingUrl}`) + "\n");
+  }
+  const debugUrls = result?.call_metadata?.provider_debug_urls;
+  if (debugUrls) {
+    for (const [label, url] of Object.entries(debugUrls)) {
+      stdoutSync(dim(`    ${label}: ${url}`) + "\n");
+    }
+  }
+}
+function printRunComplete(meta) {
+  const status = meta.status;
+  const agg = meta.aggregate;
+  const counts = agg?.conversation_calls;
+  const total = meta.total_calls ?? counts?.total;
+  const passed = meta.passed_calls ?? counts?.passed;
+  const failed = meta.failed_calls ?? counts?.failed;
+  stdoutSync("\n");
+  if (status === "pass") {
+    stdoutSync(green(bold("Run passed")) + "\n");
+  } else {
+    stdoutSync(red(bold("Run failed")) + "\n");
+  }
+  if (total != null) {
+    const parts = [];
+    if (passed) parts.push(green(`${passed} passed`));
+    if (failed) parts.push(red(`${failed} failed`));
+    parts.push(`${total} total`);
+    stdoutSync(parts.join(dim(" \xB7 ")) + "\n");
+  }
+}
+function printSummary(callResults, runComplete, runId, options = {}) {
+  const allCalls = options.rawCalls ? formatRawCalls(options.rawCalls, options.verbose ?? false) : callResults.map((e) => {
+    const meta = e.metadata_json ?? {};
+    const r = meta.result;
+    if (r) return r;
+    return {
+      name: meta.call_name ?? "call",
+      status: meta.status ?? "unknown",
+      duration_ms: meta.duration_ms,
+      error: null
+    };
+  });
+  const agg = runComplete.aggregate;
+  const counts = agg?.conversation_calls;
+  const summaryData = buildRunSummaryJson({
+    runId,
+    status: runComplete.status,
+    total: runComplete.total_calls ?? counts?.total,
+    passed: runComplete.passed_calls ?? counts?.passed,
+    failed: runComplete.failed_calls ?? counts?.failed,
+    formattedCalls: allCalls,
+    verbose: options.verbose,
+    runDetails: options.runDetails ?? { aggregate: runComplete.aggregate }
+  });
+  if (!isTTY) {
+    stdoutSync(JSON.stringify(summaryData, null, 2) + "\n");
+    return;
+  }
+  const failures = allCalls.filter((t2) => t2.status && t2.status !== "completed" && t2.status !== "pass");
+  if (failures.length > 0) {
+    stdoutSync("\n" + bold("Failed calls:") + "\n");
+    for (const t2 of failures) {
+      const duration = t2.duration_ms != null ? (t2.duration_ms / 1e3).toFixed(1) + "s" : "\u2014";
+      const parts = [red("\u2718"), bold(t2.name ?? "call"), dim(duration)];
+      stdoutSync("  " + parts.join("  ") + "\n");
+    }
+  }
+  stdoutSync(dim(`Full details: vent status ${runId}${options.verbose ? " --verbose" : ""}`) + "\n");
+}
+function buildRunSummaryJson(options) {
+  const calls = options.rawCalls ? formatRawCalls(options.rawCalls, options.verbose ?? false) : options.formattedCalls ?? [];
+  const summaryData = {
+    run_id: options.runId,
+    status: options.status,
+    total: options.total,
+    passed: options.passed,
+    failed: options.failed,
+    calls
+  };
+  const details = options.runDetails;
+  if (details?.created_at != null) summaryData["created_at"] = details.created_at;
+  if (details?.started_at != null) summaryData["started_at"] = details.started_at;
+  if (details?.finished_at != null) summaryData["finished_at"] = details.finished_at;
+  if (details?.duration_ms != null) summaryData["duration_ms"] = details.duration_ms;
+  if (details?.error_text != null) summaryData["error_text"] = details.error_text;
+  if (details?.aggregate != null) summaryData["aggregate"] = details.aggregate;
+  return summaryData;
+}
+function formatRawCalls(rawCalls, verbose) {
+  return rawCalls.map((raw) => {
+    const formatted = formatConversationResult(raw, { verbose });
+    if (formatted) return formatted;
+    const fallback = raw;
+    return {
+      name: typeof fallback["name"] === "string" ? fallback["name"] : "call",
+      status: typeof fallback["status"] === "string" ? fallback["status"] : "unknown",
+      duration_ms: typeof fallback["duration_ms"] === "number" ? fallback["duration_ms"] : void 0,
+      error: typeof fallback["error"] === "string" ? fallback["error"] : null
+    };
+  });
+}
+function printError(message) {
+  const line = red(bold("error")) + ` ${message}
+`;
+  stdoutSync(line);
+}
+function printInfo(message, { force } = {}) {
+  if (!force && !isTTY && !_verbose) return;
+  const line = blue("\u25B8") + ` ${message}
+`;
+  stdoutSync(line);
+}
+function printSuccess(message, { force } = {}) {
+  if (!force && !isTTY && !_verbose) return;
+  const line = green("\u2714") + ` ${message}
+`;
+  stdoutSync(line);
+}
+function printWarn(message, { force } = {}) {
+  if (!force && !isTTY && !_verbose) return;
+  const line = yellow("\u26A0") + ` ${message}
+`;
+  stdoutSync(line);
+}
+// src/lib/auth.ts
+var POLL_INTERVAL_MS = 2e3;
+function sleep(ms) {
+  return new Promise((r) => setTimeout(r, ms));
+}
+async function deviceAuthFlow() {
+  let startData;
+  try {
+    const res = await fetch(`${API_BASE}/device/start`, { method: "POST" });
+    if (!res.ok) {
+      return { ok: false, error: `Failed to start device auth: ${res.status}` };
+    }
+    startData = await res.json();
+  } catch {
+    return { ok: false, error: "Could not reach Vent API. Check your connection." };
+  }
+  printInfo(`Your authorization code: ${startData.user_code}`, { force: true });
+  printInfo(`Opening browser to log in...`, { force: true });
+  printInfo(`If the browser doesn't open, visit: ${startData.verification_url}`, { force: true });
+  openBrowser(startData.verification_url);
+  const deadline = new Date(startData.expires_at).getTime();
+  while (Date.now() < deadline) {
+    await sleep(POLL_INTERVAL_MS);
+    try {
+      const res = await fetch(`${API_BASE}/device/exchange`, {
+        method: "POST",
+        headers: { "Content-Type": "application/json" },
+        body: JSON.stringify({ session_id: startData.session_id })
+      });
+      if (!res.ok) continue;
+      const data = await res.json();
+      const accessToken = data.access_token;
+      if (data.status === "approved" && accessToken) {
+        await saveAccessToken(accessToken);
+        return { ok: true, accessToken };
+      }
+      if (data.status === "expired") {
+        return { ok: false, error: "Session expired. Run `npx vent-hq login` again." };
+      }
+      if (data.status === "consumed" || data.status === "invalid") {
+        return { ok: false, error: "Session invalid. Run `npx vent-hq login` again." };
+      }
+    } catch {
+    }
+  }
+  return { ok: false, error: "Login timed out. Run `npx vent-hq login` again." };
+}
+// src/lib/sse.ts
+function log(msg) {
+  if (!isVerbose()) return;
+  const ts = (/* @__PURE__ */ new Date()).toISOString().slice(11, 23);
+  const line = `[vent:sse ${ts}] ${msg}
+`;
+  process.stdout.write(line);
+}
+var MAX_RETRIES = 5;
+var RETRY_DELAY_MS = 2e3;
+async function* streamRunEvents(runId, apiKey, signal) {
+  const url = `${API_BASE}/runs/${runId}/stream`;
+  const seenIds = /* @__PURE__ */ new Set();
+  let retries = 0;
+  while (retries <= MAX_RETRIES) {
+    if (retries > 0) {
+      log(`reconnecting (attempt ${retries}/${MAX_RETRIES}) after ${RETRY_DELAY_MS}ms\u2026`);
+      await new Promise((r) => setTimeout(r, RETRY_DELAY_MS));
+    }
+    log(`connecting to ${url}`);
+    let res;
+    try {
+      res = await fetch(url, {
+        headers: { Authorization: `Bearer ${apiKey}` },
+        signal
+      });
+    } catch (err) {
+      if (err.name === "AbortError") throw err;
+      log(`fetch error: ${err.message}`);
+      retries++;
+      continue;
+    }
+    log(`response: status=${res.status} content-type=${res.headers.get("content-type")}`);
+    if (!res.ok) {
+      const body = await res.text();
+      log(`error body: ${body}`);
+      throw new Error(`SSE stream failed (${res.status}): ${body}`);
+    }
+    if (!res.body) {
+      throw new Error("SSE stream returned no body");
+    }
+    const reader = res.body.getReader();
+    const decoder = new TextDecoder();
+    let buffer = "";
+    let chunkCount = 0;
+    let eventCount = 0;
+    let gotRunComplete = false;
+    let streamError = null;
+    try {
+      while (true) {
+        let readResult;
+        try {
+          readResult = await reader.read();
+        } catch (err) {
+          if (err.name === "AbortError") throw err;
+          streamError = err;
+          log(`read error: ${streamError.message}`);
+          break;
+        }
+        const { done, value } = readResult;
+        if (done) {
+          log(`stream done after ${chunkCount} chunks, ${eventCount} events`);
+          break;
+        }
+        chunkCount++;
+        const chunk = decoder.decode(value, { stream: true });
+        buffer += chunk;
+        if (chunkCount <= 3 || chunkCount % 10 === 0) {
+          log(`chunk #${chunkCount} (${chunk.length} bytes) buffer=${buffer.length} bytes`);
+        }
+        const lines = buffer.split("\n");
+        buffer = lines.pop();
+        for (const line of lines) {
+          if (line.startsWith("data: ")) {
+            const raw = line.slice(6);
+            try {
+              const event = JSON.parse(raw);
+              eventCount++;
+              if (event.id && seenIds.has(event.id)) {
+                log(`skipping duplicate event ${event.id}`);
+                continue;
+              }
+              if (event.id) seenIds.add(event.id);
+              log(`parsed event #${eventCount}: type=${event.event_type}`);
+              yield event;
+              if (event.event_type === "run_complete") {
+                log("run_complete received \u2014 closing stream");
+                gotRunComplete = true;
+                return;
+              }
+            } catch {
+              log(`malformed JSON: ${raw.slice(0, 200)}`);
+            }
+          } else if (line.startsWith(": ")) {
+            if (chunkCount <= 3) {
+              log(`heartbeat: "${line}"`);
+            }
+          }
+        }
+      }
+    } finally {
+      reader.releaseLock();
+      log("reader released");
+    }
+    if (gotRunComplete) return;
+    retries++;
+    if (retries <= MAX_RETRIES) {
+      log(`stream ended without run_complete \u2014 will retry (${retries}/${MAX_RETRIES})`);
+    }
+  }
+  log(`exhausted ${MAX_RETRIES} retries without run_complete`);
+  yield {
+    event_type: "error",
+    message: `Stream lost after ${MAX_RETRIES} reconnect attempts without receiving run_complete`
+  };
+}
+// src/lib/run-history.ts
+import * as fs2 from "node:fs/promises";
+import * as path2 from "node:path";
+import { execSync } from "node:child_process";
+function gitInfo() {
+  try {
+    const sha = execSync("git rev-parse HEAD", { encoding: "utf-8", stdio: ["pipe", "pipe", "pipe"] }).trim();
+    const branch = execSync("git branch --show-current", { encoding: "utf-8", stdio: ["pipe", "pipe", "pipe"] }).trim() || null;
+    const status = execSync("git status --porcelain", { encoding: "utf-8", stdio: ["pipe", "pipe", "pipe"] }).trim();
+    return { sha, branch, dirty: status.length > 0 };
+  } catch {
+    return { sha: null, branch: null, dirty: false };
+  }
+}
+async function saveRunHistory(runId, callResults, runCompleteData) {
+  try {
+    const dir = path2.join(process.cwd(), ".vent", "runs");
+    await fs2.mkdir(dir, { recursive: true });
+    const git = gitInfo();
+    const now = /* @__PURE__ */ new Date();
+    const timestamp = now.toISOString().replace(/[:.]/g, "-").slice(0, 19);
+    const shortId = runId.slice(0, 8);
+    const aggregate = runCompleteData.aggregate;
+    const convCalls = aggregate?.conversation_calls;
+    const total = convCalls?.total ?? 0;
+    const passed = convCalls?.passed ?? 0;
+    const failed = convCalls?.failed ?? 0;
+    const entry = {
+      run_id: runId,
+      timestamp: now.toISOString(),
+      git_sha: git.sha,
+      git_branch: git.branch,
+      git_dirty: git.dirty,
+      summary: {
+        status: runCompleteData.status ?? "unknown",
+        calls_total: total,
+        calls_passed: passed,
+        calls_failed: failed,
+        total_duration_ms: aggregate?.total_duration_ms,
+        total_cost_usd: aggregate?.total_cost_usd
+      },
+      call_results: callResults.map((e) => e.metadata_json ?? {})
+    };
+    const filename = `${timestamp}_${shortId}.json`;
+    const filepath = path2.join(dir, filename);
+    await fs2.writeFile(filepath, JSON.stringify(entry, null, 2) + "\n");
+    return filepath;
+  } catch {
+    return null;
+  }
+}
 // src/lib/platform-connections.ts
 var PLATFORM_ENV_MAP = {
   vapi: { vapi_api_key: "VAPI_API_KEY", vapi_assistant_id: "VAPI_ASSISTANT_ID" },
@@ -5220,6 +5491,10 @@ async function runCommand(args) {
         exitCode = status === "pass" ? 0 : 1;
         debug(`run_complete: status=${status} exitCode=${exitCode}`);
       }
+      if (event.event_type === "error") {
+        printError(event.message ?? "Stream connection lost");
+        exitCode = 2;
+      }
     }
     debug(`SSE stream ended \u2014 received ${eventCount} events total`);
   } catch (err) {
@@ -5236,7 +5511,28 @@ async function runCommand(args) {
   }
   debug(`summary: callResults=${callResults.length} runComplete=${!!runCompleteData} exitCode=${exitCode}`);
   if (runCompleteData) {
-    printSummary(callResults, runCompleteData, run_id);
+    let rawRunDetails = null;
+    if (args.verbose && !isTTY2) {
+      try {
+        const res = await apiFetch(`/runs/${run_id}`, activeAccessToken);
+        rawRunDetails = await res.json();
+      } catch (err) {
+        debug(`verbose status fetch failed: ${err.message}`);
+        printWarn("Verbose result fetch failed; falling back to streamed summary.");
+      }
+    }
+    printSummary(callResults, runCompleteData, run_id, {
+      verbose: args.verbose,
+      rawCalls: Array.isArray(rawRunDetails?.["results"]) ? rawRunDetails["results"] : void 0,
+      runDetails: rawRunDetails ? {
+        created_at: rawRunDetails["created_at"],
+        started_at: rawRunDetails["started_at"],
+        finished_at: rawRunDetails["finished_at"],
+        duration_ms: rawRunDetails["duration_ms"],
+        error_text: rawRunDetails["error_text"],
+        aggregate: rawRunDetails["aggregate"]
+      } : void 0
+    });
   }
   if (runCompleteData) {
     const savedPath = await saveRunHistory(run_id, callResults, runCompleteData);
@@ -5357,10 +5653,10 @@ var RelayClient = class {
       this.controlWs.send(JSON.stringify(msg));
     }
   }
-  sendBinaryFrame(connId, payload) {
+  sendDataFrame(connId, payload, frameType) {
     if (!this.controlWs || this.controlWs.readyState !== WebSocket.OPEN) return;
     const header = new Uint8Array(37);
-    header[0] = 1;
+    header[0] = frameType;
     const connIdBytes = new TextEncoder().encode(connId);
     header.set(connIdBytes, 1);
     const frame = new Uint8Array(37 + payload.byteLength);
@@ -5372,12 +5668,18 @@ var RelayClient = class {
     ws.addEventListener("message", (event) => {
       if (event.data instanceof ArrayBuffer) {
         const data = new Uint8Array(event.data);
-        if (data.length < 37 || data[0] !== 1) return;
+        if (data.length < 37) return;
+        const frameType = data[0];
+        if (frameType !== 1 && frameType !== 2) return;
         const connId = new TextDecoder().decode(data.subarray(1, 37));
         const payload = data.subarray(37);
         const conn = this.localConnections.get(connId);
         if (conn?.local.readyState === WebSocket.OPEN) {
-          conn.local.send(payload);
+          if (frameType === 2) {
+            conn.local.send(new TextDecoder().decode(payload));
+          } else {
+            conn.local.send(payload);
+          }
         }
         return;
       }
@@ -5425,8 +5727,11 @@ var RelayClient = class {
         this.localConnections.set(connId, { local: localWs, connId });
       });
       localWs.addEventListener("message", (event) => {
-        const payload = event.data instanceof ArrayBuffer ? new Uint8Array(event.data) : new TextEncoder().encode(event.data);
-        this.sendBinaryFrame(connId, payload);
+        if (event.data instanceof ArrayBuffer) {
+          this.sendDataFrame(connId, new Uint8Array(event.data), 1);
+        } else {
+          this.sendDataFrame(connId, new TextEncoder().encode(event.data), 2);
+        }
       });
       let cleaned = false;
       const cleanup = (reason) => {
@@ -5460,7 +5765,7 @@ async function startAgentSession(relayConfig) {
   };
   const client = new RelayClient(clientConfig);
   client.on("log", (msg) => {
-    if (isVerbose()) process.stderr.write(`${msg}
+    if (isVerbose()) process.stdout.write(`${msg}
 `);
   });
   await client.connect();
@@ -5473,13 +5778,13 @@ async function startAgentSession(relayConfig) {
       env
     });
     agentProcess.stdout?.on("data", (data) => {
-      if (isVerbose()) process.stderr.write(`[agent] ${data}`);
+      if (isVerbose()) process.stdout.write(`[agent] ${data}`);
     });
     agentProcess.stderr?.on("data", (data) => {
-      if (isVerbose()) process.stderr.write(`[agent] ${data}`);
+      if (isVerbose()) process.stdout.write(`[agent] ${data}`);
     });
     agentProcess.on("error", (err) => {
-      process.stderr.write(`Agent process error: ${err.message}
+      process.stdout.write(`Agent process error: ${err.message}
 `);
     });
   }
@@ -5661,7 +5966,27 @@ async function statusCommand(args) {
   try {
     const res = await apiFetch(`/runs/${args.runId}`, accessToken);
     const data = await res.json();
-    process.stdout.write(JSON.stringify(data, null, 2) + "\n");
+    const aggregate = data.aggregate;
+    const counts = aggregate?.conversation_calls;
+    const results = Array.isArray(data.results) ? data.results : [];
+    const summary = buildRunSummaryJson({
+      runId: typeof data.id === "string" ? data.id : args.runId,
+      status: data.status,
+      total: counts?.total,
+      passed: counts?.passed,
+      failed: counts?.failed,
+      rawCalls: results,
+      verbose: args.verbose,
+      runDetails: {
+        created_at: data.created_at,
+        started_at: data.started_at,
+        finished_at: data.finished_at,
+        duration_ms: data.duration_ms,
+        error_text: data.error_text,
+        aggregate: data.aggregate
+      }
+    });
+    writeJsonStdout(summary);
     const status = data.status;
     return status === "pass" ? 0 : status === "fail" ? 1 : 0;
   } catch (err) {
@@ -6417,13 +6742,13 @@ var ze = { light: I2("\u2500", "-"), heavy: I2("\u2501", "="), block: I2("\u2588
 var Qe = `${t("gray", h)}  `;
 // src/skills/claude-code.md
-var claude_code_default = '---\nname: vent\ndescription: Voice agent calls \u2014 run calls against your voice agent, get pass/fail results with latency and latency metrics\nallowed-tools: Bash(npx vent-hq *)\n---\n\n# Vent \u2014 Voice Agent Calls\n\nCall voice agents from the terminal. Calls run in the cloud \u2014 results stream back.\n\n## When to Call (read this first)\n\n- After modifying voice agent code (system prompt, tools, handlers): ALWAYS run calls\n- After changing audio config: run calls\n- Before marking a task complete that touches agent behavior: run calls\n- Do NOT skip call verification \u2014 voice agents are non-deterministic, small changes can break flows\n\n## Setup (if no Vent access token)\n\nIf `~/.vent/credentials` does not exist and `VENT_ACCESS_TOKEN` is not set:\n\n```bash\nnpx vent-hq init\n```\n\nOne command. No email, no browser, no human interaction needed. You get 10 free runs instantly.\n\n## Commands\n\n| Command | Purpose |\n|---------|---------|\n| `npx vent-hq init` | First-time setup (creates account + installs skills) |\n| `npx vent-hq agent start -f .vent/suite.<adapter>.json` | Start one shared local agent session (required for `start_command`) |\n| `npx vent-hq agent stop <session-id>` | Close a shared local agent session |\n| `npx vent-hq run -f .vent/suite.<adapter>.json` | Run a call from suite file (auto-selects if only one call) |\n| `npx vent-hq run -f .vent/suite.<adapter>.json --call <name>` | Run a specific named call |\n| `npx vent-hq stop <run-id>` | Cancel a queued or running call |\n| `npx vent-hq status <run-id>` | Check results of a previous run |\n\n\n## Critical Rules\n\n1. **5-minute timeout** \u2014 Set `timeout: 300000` on each Bash call. Individual calls can still take up to 5 minutes.\n2. **If a call gets backgrounded** \u2014 Wait for it to complete before proceeding. Never end your response without the result.\n3. **This skill is self-contained** \u2014 The full config schema is below. Do NOT re-read this file.\n4. **Always analyze results** \u2014 The run command outputs complete JSON with full transcript, latency, and tool calls. Analyze this output directly.\n\n## Workflow\n\n### First time: create the call suite\n\n1. Read the voice agent\'s codebase \u2014 understand its system prompt, tools, intents, and domain.\n2. Read the **Full Config Schema** section below for all available fields.\n3. Create the suite file in `.vent/` using the naming convention: `.vent/suite.<adapter>.json` (e.g., `.vent/suite.vapi.json`, `.vent/suite.websocket.json`, `.vent/suite.retell.json`). This prevents confusion when multiple adapters are tested in the same project.\n   - Name calls after specific flows (e.g., `"reschedule-appointment"`, not `"call-1"`)\n   - Write `caller_prompt` as a realistic persona with a specific goal, based on the agent\'s domain\n   - Set `max_turns` based on the flow complexity (simple FAQ: 4-6, booking: 8-12, complex: 12-20)\n\n### Multiple suite files\n\nIf `.vent/` contains more than one suite file, **always check which adapter each suite uses before running**. Read the `connection.adapter` field in each file. Never run a suite intended for a different adapter \u2014 results will be meaningless or fail. When reporting results, always state which suite file produced them (e.g., "Results from `.vent/suite.vapi.json`:").\n\n### Run calls\n\n1. If the suite uses `start_command`, start the shared local session first:\n   ```bash\n   npx vent-hq agent start -f .vent/suite.<adapter>.json\n   ```\n\n2. Run calls:\n   ```bash\n   # suite with one call (auto-selects)\n   npx vent-hq run -f .vent/suite.<adapter>.json\n\n   # suite with multiple calls \u2014 pick one by name\n   npx vent-hq run -f .vent/suite.<adapter>.json --call happy-path\n\n   # local start_command \u2014 add --session\n   npx vent-hq run -f .vent/suite.<adapter>.json --call happy-path --session <session-id>\n   ```\n\n3. To run multiple calls from the same suite, run each as a separate command:\n   ```bash\n   npx vent-hq run -f .vent/suite.vapi.json --call happy-path\n   npx vent-hq run -f .vent/suite.vapi.json --call edge-case\n   ```\n\n4. Analyze each result, identify failures, correlate with the codebase, and fix.\n\n5. **Compare with previous run** \u2014 Vent saves full result JSON to `.vent/runs/` after every run. Read the second-most-recent JSON in `.vent/runs/` and compare it against the current run:\n   - Status flips: pass\u2192fail (obvious regression)\n   - Latency: TTFW p50/p95 increased >20%\n   - Tool calls: success count dropped\n   - Cost: cost_usd increased >30%\n   - Transcripts: agent responses diverged significantly\n   Report what regressed and correlate with the code diff (`git diff` between the two runs\' git SHAs). If no previous run exists, skip \u2014 this is the baseline.\n\n### After modifying voice agent code\n\nRe-run the existing suite \u2014 no need to recreate it.\n\n## Connection\n\n- **BYO agent runtime**: your agent owns its own provider credentials. Use `start_command` for a local agent or `agent_url` for a hosted custom endpoint.\n- **Platform-direct runtime**: use adapter `vapi | retell | elevenlabs | bland | livekit`. This is the only mode where Vent itself needs provider credentials and saved platform connections apply.\n\n## WebSocket Protocol (BYO agents)\n\nWhen using `adapter: "websocket"`, Vent communicates with the agent over a single WebSocket connection:\n\n- **Binary frames** \u2192 PCM audio (16-bit mono, configurable sample rate)\n- **Text frames** \u2192 optional JSON events the agent can send for better test accuracy:\n\n| Event | Format | Purpose |\n|-------|--------|---------|\n| `speech-update` | `{"type":"speech-update","status":"started"\\|"stopped"}` | Enables platform-assisted turn detection (more accurate than VAD alone) |\n| `tool_call` | `{"type":"tool_call","name":"...","arguments":{...},"result":...,"successful":bool,"duration_ms":number}` | Reports tool calls for observability |\n| `vent:timing` | `{"type":"vent:timing","stt_ms":number,"llm_ms":number,"tts_ms":number}` | Reports component latency breakdown per turn |\n\nVent sends `{"type":"end-call"}` to the agent when the test is done.\n\nAll text frames are optional \u2014 audio-only agents work fine with VAD-based turn detection.\n\n## Full Config Schema\n\n- ALL calls MUST reference the agent\'s real context (system prompt, tools, knowledge base) from the codebase.\n\n<vent_run>\n{\n  "connection": { ... },\n  "calls": {\n    "happy-path": { ... },\n    "edge-case": { ... }\n  }\n}\n</vent_run>\n\nOne suite file per platform/adapter. `connection` is declared once, `calls` is a named map of call specs. Each key becomes the call name. Run one call at a time with `--call <name>`.\n\n<config_connection>\n{\n  "connection": {\n    "adapter": "required -- websocket | livekit | vapi | retell | elevenlabs | bland",\n    "start_command": "shell command to start agent (relay only, required for local)",\n    "health_endpoint": "health check path after start_command (default: /health, relay only, required for local)",\n    "agent_url": "hosted custom agent URL (wss:// or https://). Use for BYO hosted agents.",\n    "agent_port": "local agent port (default: 3001, required for local)",\n    "platform": "optional authoring convenience for platform-direct adapters only. The CLI resolves this locally, creates/updates a saved platform connection, and strips raw provider secrets before submit. Do not use for websocket start_command or agent_url runs."\n  }\n}\n\n<credential_resolution>\nIMPORTANT: How to handle platform credentials (API keys, secrets, agent IDs):\n\nThere are two product modes:\n- `BYO agent runtime`: your agent owns its own provider credentials. This covers both `start_command` (local) and `agent_url` (hosted custom endpoint).\n- `Platform-direct runtime`: Vent talks to `vapi`, `retell`, `elevenlabs`, `bland`, or `livekit` directly. This is the only mode that uses saved platform connections.\n\n1. For `start_command` and `agent_url` runs, do NOT put Deepgram / ElevenLabs / OpenAI / other provider keys into Vent config unless the Vent adapter itself needs them. Those credentials belong to the user\'s local or hosted agent runtime.\n2. For platform-direct adapters (`vapi`, `retell`, `elevenlabs`, `bland`, `livekit`), the CLI auto-resolves credentials from `.env.local`, `.env`, and the current shell env. If those env vars already exist, you can omit credential fields from the config JSON entirely.\n3. If you include credential fields in the config, put the ACTUAL VALUE, NOT the env var name. WRONG: `"vapi_api_key": "VAPI_API_KEY"`. RIGHT: `"vapi_api_key": "sk-abc123..."` or omit the field.\n4. The CLI uses the resolved provider config to create or update a saved platform connection server-side, then submits only `platform_connection_id`. Users should not manually author `platform_connection_id`.\n5. To check whether credentials are already available, inspect `.env.local`, `.env`, and any relevant shell env visible to the CLI process.\n\nAuto-resolved env vars per platform:\n| Platform | Config field | Env var (auto-resolved from `.env.local`, `.env`, or shell env) |\n|----------|-------------|-----------------------------------|\n| Vapi | vapi_api_key | VAPI_API_KEY |\n| Vapi | vapi_assistant_id | VAPI_ASSISTANT_ID |\n| Bland | bland_api_key | BLAND_API_KEY |\n| Bland | bland_pathway_id | BLAND_PATHWAY_ID |\n| LiveKit | livekit_api_key | LIVEKIT_API_KEY |\n| LiveKit | livekit_api_secret | LIVEKIT_API_SECRET |\n| LiveKit | livekit_url | LIVEKIT_URL |\n| Retell | retell_api_key | RETELL_API_KEY |\n| Retell | retell_agent_id | RETELL_AGENT_ID |\n| ElevenLabs | elevenlabs_api_key | ELEVENLABS_API_KEY |\n| ElevenLabs | elevenlabs_agent_id | ELEVENLABS_AGENT_ID |\n\nThe CLI strips raw platform secrets before `/runs/submit`. Platform-direct runs go through a saved `platform_connection_id` automatically. BYO agent runs (`start_command` and `agent_url`) do not.\n</credential_resolution>\n\n<config_adapter_rules>\nWebSocket (local agent via relay):\n{\n  "connection": {\n    "adapter": "websocket",\n    "start_command": "npm run start",\n    "health_endpoint": "/health",\n    "agent_port": 3001\n  }\n}\n\nWebSocket (hosted custom agent):\n{\n  "connection": {\n    "adapter": "websocket",\n    "agent_url": "https://my-agent.fly.dev"\n  }\n}\n\nRetell:\n{\n  "connection": {\n    "adapter": "retell",\n    "platform": { "provider": "retell" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: RETELL_API_KEY, RETELL_AGENT_ID. Only add retell_api_key/retell_agent_id to the JSON if those env vars are not already available.\n\nBland:\n{\n  "connection": {\n    "adapter": "bland",\n    "platform": { "provider": "bland" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: BLAND_API_KEY, BLAND_PATHWAY_ID. Only add bland_api_key/bland_pathway_id to the JSON if those env vars are not already available.\nNote: All agent config (voice, model, tools, etc.) is set on the pathway itself, not in Vent config.\n\nVapi:\n{\n  "connection": {\n    "adapter": "vapi",\n    "platform": { "provider": "vapi" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: VAPI_API_KEY, VAPI_ASSISTANT_ID. Only add vapi_api_key/vapi_assistant_id to the JSON if those env vars are not already available.\nmax_concurrency for Vapi: Starter=10, Growth=50, Enterprise=100+. Ask the user which tier they\'re on. If unknown, default to 10.\nAll assistant config (voice, model, transcriber, interruption settings, etc.) is set on the Vapi assistant itself, not in Vent config.\n\nElevenLabs:\n{\n  "connection": {\n    "adapter": "elevenlabs",\n    "platform": { "provider": "elevenlabs" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: ELEVENLABS_API_KEY, ELEVENLABS_AGENT_ID. Only add elevenlabs_api_key/elevenlabs_agent_id to the JSON if those env vars are not already available.\n\nLiveKit:\n{\n  "connection": {\n    "adapter": "livekit",\n    "platform": {\n      "provider": "livekit",\n      "livekit_agent_name": "my-agent",\n      "max_concurrency": 5\n    }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: LIVEKIT_API_KEY, LIVEKIT_API_SECRET, LIVEKIT_URL. Only add these to the JSON if those env vars are not already available.\nlivekit_agent_name is optional -- only needed if the agent registers with an explicit agent_name in WorkerOptions. Omit for automatic dispatch.\nmax_concurrency: Free/Build=5, Ship=20, Scale=50+. Ask the user which tier they\'re on. If unknown, default to 5.\n</config_adapter_rules>\n</config_connection>\n\n\n<call_config>\n<tool_call_capture>\nvapi/retell/elevenlabs/bland: automatic via platform API (no user code needed).\nWebSocket/WebRTC: user\'s agent must emit tool calls:\n  WebSocket \u2014 JSON text frame: {"type":"tool_call","name":"...","arguments":{},"result":{},"successful":true,"duration_ms":150}\n  WebRTC/LiveKit \u2014 publishData() or sendText() on topic "vent:tool-calls". Same JSON.\n</tool_call_capture>\n\n<component_timing>\nPlatform adapters (vapi/retell/elevenlabs/bland/livekit) get STT/LLM/TTS breakdown automatically.\nWebSocket agents can opt in by sending a JSON text frame after each agent turn:\n  {"type":"vent:timing","stt_ms":120,"llm_ms":450,"tts_ms":80}\nAll fields optional. Send one per agent response. Without this, component_latency is omitted from results.\nWhen modifying a WebSocket agent\'s code, add this text frame after TTS completes to enable component latency reporting.\n</component_timing>\n\n<config_call>\nEach call in the `calls` map. The key is the call name (e.g. `"reschedule-appointment"`, not `"call-1"`).\n{\n      "caller_prompt": "required \u2014 caller persona and behavior (name -> goal -> emotion -> conditional behavior)",\n      "max_turns": "required \u2014 default 6",\n      "silence_threshold_ms": "optional \u2014 end-of-turn threshold ms (default 800, 200-10000). 800-1200 FAQ, 2000-3000 tool calls, 3000-5000 complex reasoning.",\n      "persona": "optional \u2014 caller behavior controls",\n      {\n        "pace": "slow | normal | fast",\n        "clarity": "clear | vague | rambling",\n        "disfluencies": "true | false",\n        "cooperation": "cooperative | reluctant | hostile",\n        "emotion": "neutral | cheerful | confused | frustrated | skeptical | rushed",\n        "interruption_style": "low (~3/10 turns) | high (~7/10 turns)",\n        "memory": "reliable | unreliable",\n        "intent_clarity": "clear | indirect | vague",\n        "confirmation_style": "explicit | vague"\n      },\n      "audio_actions": "optional \u2014 per-turn audio stress calls",\n      [\n        { "action": "interrupt", "at_turn": "N", "prompt": "what caller says" },\n        { "action": "silence", "at_turn": "N", "duration_ms": "1000-30000" },\n        { "action": "inject_noise", "at_turn": "N", "noise_type": "babble | white | pink", "snr_db": "0-40" },\n        { "action": "split_sentence", "at_turn": "N", "split": { "part_a": "...", "part_b": "...", "pause_ms": "500-5000" } },\n        { "action": "noise_on_caller", "at_turn": "N" }\n      ],\n      "prosody": "optional \u2014 Hume emotion analysis (default false)",\n      "caller_audio": "optional \u2014 omit for clean audio",\n      {\n        "noise": { "type": "babble | white | pink", "snr_db": "0-40" },\n        "speed": "0.5-2.0 (1.0 = normal)",\n        "speakerphone": "true | false",\n        "mic_distance": "close | normal | far",\n        "clarity": "0.0-1.0 (1.0 = perfect)",\n        "accent": "american | british | australian | filipino | spanish_mexican | spanish_peninsular | spanish_colombian | spanish_argentine | german | french | italian | dutch | japanese",\n        "packet_loss": "0.0-0.3",\n        "jitter_ms": "0-100"\n      },\n      "language": "optional \u2014 ISO 639-1: en, es, fr, de, it, nl, ja"\n}\n\n<examples_call>\n<simple_suite_example>\n{\n  "connection": {\n    "adapter": "vapi",\n    "platform": { "provider": "vapi" }\n  },\n  "calls": {\n    "reschedule-appointment": {\n      "caller_prompt": "You are Maria, calling to reschedule her dentist appointment from Thursday to next Tuesday. She\'s in a hurry and wants this done quickly.",\n      "max_turns": 8\n    },\n    "cancel-appointment": {\n      "caller_prompt": "You are Tom, calling to cancel his appointment for Friday. He\'s calm and just wants confirmation.",\n      "max_turns": 6\n    }\n  }\n}\n</simple_suite_example>\n\n<advanced_call_example>\nA call entry with advanced options (persona, audio actions, prosody):\n{\n  "noisy-interruption-booking": {\n    "caller_prompt": "You are James, an impatient customer calling from a loud coffee shop to book a plumber for tomorrow morning. You interrupt the agent mid-sentence when they start listing availability \u2014 you just want the earliest slot.",\n    "max_turns": 12,\n    "persona": { "pace": "fast", "cooperation": "reluctant", "emotion": "rushed", "interruption_style": "high" },\n    "audio_actions": [\n      { "action": "interrupt", "at_turn": 3, "prompt": "Just give me the earliest one!" },\n      { "action": "inject_noise", "at_turn": 1, "noise_type": "babble", "snr_db": 15 }\n    ],\n    "caller_audio": { "noise": { "type": "babble", "snr_db": 20 }, "speed": 1.3 },\n    "prosody": true\n  }\n}\n</advanced_call_example>\n\n</examples_call>\n</config_call>\n\n<output_conversation_test>\n{\n  "name": "sarah-hotel-booking",\n  "status": "completed",\n  "caller_prompt": "You are Sarah, calling to book...",\n  "duration_ms": 45200,\n  "error": null,\n  "transcript": [\n    { "role": "caller", "text": "Hi, I\'d like to book..." },\n    { "role": "agent", "text": "Sure! What date?", "ttfb_ms": 650, "ttfw_ms": 780, "audio_duration_ms": 2400 },\n    { "role": "agent", "text": "Let me check avail\u2014", "ttfb_ms": 540, "ttfw_ms": 620, "audio_duration_ms": 1400, "interrupted": true },\n    { "role": "caller", "text": "Just the earliest slot please", "audio_duration_ms": 900, "is_interruption": true },\n    { "role": "agent", "text": "Sure, the earliest is 9 AM tomorrow.", "ttfb_ms": 220, "ttfw_ms": 260, "audio_duration_ms": 2100 }\n  ],\n  "latency": {\n    "mean_ttfw_ms": 890, "p50_ttfw_ms": 850, "p95_ttfw_ms": 1400, "p99_ttfw_ms": 1550,\n    "first_turn_ttfw_ms": 1950, "total_silence_ms": 4200, "mean_turn_gap_ms": 380,\n    "drift_slope_ms_per_turn": -45.2, "mean_silence_pad_ms": 128, "mouth_to_ear_est_ms": 1020\n  },\n  "transcript_quality": {\n    "wer": 0.04,\n    "hallucination_events": [\n      { "error_count": 5, "reference_text": "triple five one two", "hypothesis_text": "five five five nine two" }\n    ],\n    "repetition_score": 0.05,\n    "reprompt_count": 0,\n    "filler_word_rate": 0.8,\n    "words_per_minute": 148\n  },\n  "audio_analysis": {\n    "agent_speech_ratio": 0.72,\n    "interruption_rate": 0.25,\n    "interruption_count": 1,\n    "barge_in_recovery_time_ms": 280,\n    "agent_interrupting_user_rate": 0.0,\n    "agent_interrupting_user_count": 0,\n    "missed_response_windows": 0,\n    "longest_monologue_ms": 5800,\n    "silence_gaps_over_2s": 1,\n    "total_internal_silence_ms": 2400,\n    "mean_agent_speech_segment_ms": 3450\n  },\n  "tool_calls": {\n    "total": 2, "successful": 2, "failed": 0, "mean_latency_ms": 340,\n    "names": ["check_availability", "book_appointment"],\n    "observed": [{ "name": "check_availability", "arguments": { "date": "2026-03-12" }, "result": { "slots": ["09:00", "10:00"] }, "successful": true, "latency_ms": 280, "turn_index": 3 }]\n  },\n  "call_metadata": {\n    "platform": "vapi",\n    "recording_url": "https://example.com/recording"\n  },\n  "warnings": [],\n  "audio_actions": [\n    { "at_turn": 5, "action": "silence", "metrics": { "agent_prompted": false, "unprompted_utterance_count": 0, "silence_duration_ms": 8000 } }\n  ],\n  "emotion": {\n    "naturalness": 0.72, "mean_calmness": 0.65, "mean_confidence": 0.58, "peak_frustration": 0.08, "emotion_trajectory": "stable"\n  }\n}\n\nAll fields optional except name, status, caller_prompt, duration_ms, transcript. Fields appear only when relevant analysis ran (e.g., emotion requires prosody: true).\n\n### Result presentation\n\nWhen you report a conversation result to the user, always include:\n\n1. **Summary** \u2014 the overall verdict and the 1-3 most important findings.\n2. **Transcript summary** \u2014 a short narrative of what happened in the call.\n3. **Recording URL** \u2014 include `call_metadata.recording_url` when present; explicitly say when it is unavailable.\n4. **Next steps** \u2014 concrete fixes, follow-up tests, or why no change is needed.\n\nUse metrics to support the summary, not as the whole answer. Do not dump raw numbers without interpretation.\n\nWhen `call_metadata.transfer_attempted` is present, explicitly say whether the transfer only appeared attempted or was mechanically verified as completed. If `call_metadata.transfers[*].verification` is present, use it to mention second-leg observation, connect latency, transcript/context summary, and whether context passing was verified.\n\n### Judging guidance\n\nUse the transcript, metrics, test scenario, and relevant agent instructions/system prompt to judge:\n\n| Dimension | What to check |\n|--------|----------------|\n| **Hallucination detection** | Check whether the agent stated anything not grounded in its instructions, tools, or the conversation itself. Treat `transcript_quality.hallucination_events` only as a speech-recognition warning signal, not proof of agent hallucination. |\n| **Instruction following** | Compare the agent\'s behavior against its system prompt and the test\'s expected constraints. |\n| **Context retention** | Check whether the agent forgot or contradicted information established earlier in the call. |\n| **Semantic accuracy** | Check whether the agent correctly understood the caller\'s intent and responded to the real request. |\n| **Goal completion** | Decide whether the agent achieved what the test scenario was designed to verify. |\n| **Transfer correctness** | For transfer scenarios, judge whether transfer was appropriate, whether it completed, whether it went to the expected destination, and whether enough context was passed during the handoff. |\n\n### Interruption evaluation\n\nWhen the transcript contains `interrupted: true` / `is_interruption: true` turns, evaluate these metrics by reading the transcript:\n\n| Metric | How to evaluate | Target |\n|--------|----------------|--------|\n| **Recovery rate** | For each interrupted turn: does the post-interrupt agent response acknowledge or address the interruption? (e.g., "Sure, the earliest is 9 AM" after being cut off mid-availability-list) | >90% |\n| **Context retention** | After the interruption, does the agent remember pre-interrupt conversation state? (e.g., still knows the caller\'s name, booking details, etc.) | >95% |\n| **Barge-in recovery time** | Use `audio_analysis.barge_in_recovery_time_ms` when available. Lower is better because it measures how long the agent kept speaking after the caller cut in. | <500ms acceptable |\n| **Agent interrupting user rate** | Use `audio_analysis.agent_interrupting_user_rate` and the transcript to see whether the agent starts speaking before the caller finished. | 0 ideal |\n\nReport these alongside standard metrics when interruption calls run. Flag any turn where the agent ignores the interruption, repeats itself from scratch, or loses context.\n</output_conversation_test>\n</call_config>\n\n\n## Output\n\n- **Exit codes**: 0=pass, 1=fail, 2=error\n- The `run` command outputs **complete results as pretty-printed JSON** \u2014 including full transcript, latency metrics, tool calls, and audio analysis for every call. Do NOT run a separate `vent status` command \u2014 all data is already in the output.\n\n## Vent Access Token\n\nRun `npx vent-hq login` or set `VENT_ACCESS_TOKEN` env var.\nVent provides DEEPGRAM_API_KEY and ANTHROPIC_API_KEY automatically.\n';
+var claude_code_default = '---\nname: vent\ndescription: Voice agent calls \u2014 run calls against your voice agent, get pass/fail results with latency and latency metrics\nallowed-tools: Bash(npx vent-hq *)\n---\n\n# Vent \u2014 Voice Agent Calls\n\nCall voice agents from the terminal. Calls run in the cloud \u2014 results stream back.\n\n## When to Call (read this first)\n\n- After modifying voice agent code (system prompt, tools, handlers): ALWAYS run calls\n- After changing audio config: run calls\n- Before marking a task complete that touches agent behavior: run calls\n- Do NOT skip call verification \u2014 voice agents are non-deterministic, small changes can break flows\n\n## Setup (if no Vent access token)\n\nIf `~/.vent/credentials` does not exist and `VENT_ACCESS_TOKEN` is not set:\n\n```bash\nnpx vent-hq init\n```\n\nOne command. No email, no browser, no human interaction needed. You get 10 free runs instantly.\n\n## Commands\n\n| Command | Purpose |\n|---------|---------|\n| `npx vent-hq init` | First-time setup (creates account + installs skills) |\n| `npx vent-hq agent start -f .vent/suite.<adapter>.json` | Start one shared local agent session (required for `start_command`) |\n| `npx vent-hq agent stop <session-id>` | Close a shared local agent session |\n| `npx vent-hq run -f .vent/suite.<adapter>.json` | Run a call from suite file (auto-selects if only one call) |\n| `npx vent-hq run -f .vent/suite.<adapter>.json --verbose` | Include debug fields in the result JSON |\n| `npx vent-hq run -f .vent/suite.<adapter>.json --call <name>` | Run a specific named call |\n| `npx vent-hq stop <run-id>` | Cancel a queued or running call |\n| `npx vent-hq status <run-id>` | Check results of a previous run |\n| `npx vent-hq status <run-id> --verbose` | Re-print a run with debug fields included |\n\n## When To Use `--verbose`\n\nDefault output is enough for most work. It already includes:\n- transcript\n- latency\n- transcript quality (`wer` / `cer`)\n- audio analysis\n- tool calls\n- summary cost / recording / transfers\n\nUse `--verbose` only when you need debugging detail that is not in the default result:\n- per-turn debug fields: timestamps, caller decision mode, silence pad, STT confidence, platform transcript\n- raw signal analysis: `debug.signal_quality`\n- harness timings: `debug.harness_overhead`\n- raw prosody payload and warnings\n- raw provider warnings\n- per-turn component latency arrays\n- raw observed tool-call timeline\n- provider-specific metadata in `debug.provider_metadata`\n\nTrigger `--verbose` when:\n- transcript accuracy looks wrong and you need to inspect `platform_transcript`\n- latency is bad and you need per-turn/component breakdowns\n- interruptions/barge-in behavior looks wrong\n- tool-call execution looks inconsistent or missing\n- the provider returned warnings/errors or you need provider-native artifacts\n\nSkip `--verbose` when:\n- you only need pass/fail, transcript, latency, tool calls, recording, or summary\n- you are doing quick iteration on prompt wording and the normal result already explains the failure\n\n## Normalization Contract\n\nVent always returns one normalized result shape on `stdout` across adapters. Treat these as the stable categories:\n- `transcript`\n- `latency`\n- `transcript_quality`\n- `audio_analysis`\n- `tool_calls`\n- `component_latency`\n- `call_metadata`\n- `warnings`\n- `audio_actions`\n- `emotion`\n\nSource-of-truth policy:\n- Vent computes transcript, latency, and audio-quality metrics itself.\n- Hosted adapters choose the best source per category, usually provider post-call data for tool calls, call metadata, transfers, provider transcripts, and recordings.\n- Realtime provider events are fallback or enrichment only when post-call data is missing, delayed, weaker for that category, or provider-specific.\n- `LiveKit` helper events are the provider-native path for rich in-agent observability.\n- `websocket`/custom agents are realtime-native but still map into the same normalized categories.\n- Keep adapter-specific details in `call_metadata.provider_metadata` or `debug.provider_metadata`, not in new top-level fields.\n\n\n## Critical Rules\n\n1. **5-minute timeout** \u2014 Set `timeout: 300000` on each Bash call. Individual calls can still take up to 5 minutes.\n2. **If a call gets backgrounded** \u2014 Wait for it to complete before proceeding. Never end your response without the result.\n3. **This skill is self-contained** \u2014 The full config schema is below. Do NOT re-read this file.\n4. **Always analyze results** \u2014 The run command outputs complete JSON with full transcript, latency, and tool calls. Use `--verbose` only when the default result is not enough to explain the failure. Analyze this output directly.\n\n## Workflow\n\n### First time: create the call suite\n\n1. Read the voice agent\'s codebase \u2014 understand its system prompt, tools, intents, and domain.\n2. Read the **Full Config Schema** section below for all available fields.\n3. Create the suite file in `.vent/` using the naming convention: `.vent/suite.<adapter>.json` (e.g., `.vent/suite.vapi.json`, `.vent/suite.websocket.json`, `.vent/suite.retell.json`). This prevents confusion when multiple adapters are tested in the same project.\n   - Name calls after specific flows (e.g., `"reschedule-appointment"`, not `"call-1"`)\n   - Write `caller_prompt` as a realistic persona with a specific goal, based on the agent\'s domain\n   - Set `max_turns` based on the flow complexity (simple FAQ: 4-6, booking: 8-12, complex: 12-20)\n\n### Multiple suite files\n\nIf `.vent/` contains more than one suite file, **always check which adapter each suite uses before running**. Read the `connection.adapter` field in each file. Never run a suite intended for a different adapter \u2014 results will be meaningless or fail. When reporting results, always state which suite file produced them (e.g., "Results from `.vent/suite.vapi.json`:").\n\n### Run calls\n\n1. If the suite uses `start_command`, start the shared local session first:\n   ```bash\n   npx vent-hq agent start -f .vent/suite.<adapter>.json\n   ```\n\n2. Run calls:\n   ```bash\n   # suite with one call (auto-selects)\n   npx vent-hq run -f .vent/suite.<adapter>.json\n\n   # suite with multiple calls \u2014 pick one by name\n   npx vent-hq run -f .vent/suite.<adapter>.json --call happy-path\n\n   # local start_command \u2014 add --session\n   npx vent-hq run -f .vent/suite.<adapter>.json --call happy-path --session <session-id>\n   ```\n\n3. To run multiple calls from the same suite, run each as a separate command:\n   ```bash\n   npx vent-hq run -f .vent/suite.vapi.json --call happy-path\n   npx vent-hq run -f .vent/suite.vapi.json --call edge-case\n   ```\n\n4. Analyze each result, identify failures, correlate with the codebase, and fix.\n\n5. **Compare with previous run** \u2014 Vent saves full result JSON to `.vent/runs/` after every run. Read the second-most-recent JSON in `.vent/runs/` and compare it against the current run:\n   - Status flips: pass\u2192fail (obvious regression)\n   - Latency: TTFW p50/p95 increased >20%\n   - Tool calls: success count dropped\n   - Cost: cost_usd increased >30%\n   - Transcripts: agent responses diverged significantly\n   Report what regressed and correlate with the code diff (`git diff` between the two runs\' git SHAs). If no previous run exists, skip \u2014 this is the baseline.\n\n### After modifying voice agent code\n\nRe-run the existing suite \u2014 no need to recreate it.\n\n## Connection\n\n- **BYO agent runtime**: your agent owns its own provider credentials. Use `start_command` for a local agent or `agent_url` for a hosted custom endpoint.\n- **Platform-direct runtime**: use adapter `vapi | retell | elevenlabs | bland | livekit`. This is the only mode where Vent itself needs provider credentials and saved platform connections apply.\n\n## WebSocket Protocol (BYO agents)\n\nWhen using `adapter: "websocket"`, Vent communicates with the agent over a single WebSocket connection:\n\n- **Binary frames** \u2192 PCM audio (16-bit mono, configurable sample rate)\n- **Text frames** \u2192 optional JSON events the agent can send for better test accuracy:\n\n| Event | Format | Purpose |\n|-------|--------|---------|\n| `speech-update` | `{"type":"speech-update","status":"started"\\|"stopped"}` | Enables platform-assisted turn detection (more accurate than VAD alone) |\n| `tool_call` | `{"type":"tool_call","name":"...","arguments":{...},"result":...,"successful":bool,"duration_ms":number}` | Reports tool calls for observability |\n| `vent:timing` | `{"type":"vent:timing","stt_ms":number,"llm_ms":number,"tts_ms":number}` | Reports component latency breakdown per turn |\n| `vent:session` | `{"type":"vent:session","platform":"custom","provider_call_id":"...","provider_session_id":"..."}` | Reports stable provider/session identifiers |\n| `vent:call-metadata` | `{"type":"vent:call-metadata","call_metadata":{...}}` | Reports post-call metadata such as cost, recordings, variables, and provider-specific artifacts |\n| `vent:transcript` | `{"type":"vent:transcript","role":"caller"\\|"agent","text":"...","turn_index":0}` | Reports platform/native transcript text for caller or agent |\n| `vent:transfer` | `{"type":"vent:transfer","destination":"...","status":"attempted"\\|"completed"}` | Reports transfer attempts and outcomes |\n| `vent:debug-url` | `{"type":"vent:debug-url","label":"log","url":"https://..."}` | Reports provider debug/deep-link URLs |\n| `vent:warning` | `{"type":"vent:warning","message":"...","code":"..."}` | Reports provider/runtime warnings worth preserving in run metadata |\n\nVent sends `{"type":"end-call"}` to the agent when the test is done.\n\nAll text frames are optional \u2014 audio-only agents work fine with VAD-based turn detection.\n\n## Full Config Schema\n\n- ALL calls MUST reference the agent\'s real context (system prompt, tools, knowledge base) from the codebase.\n\n<vent_run>\n{\n  "connection": { ... },\n  "calls": {\n    "happy-path": { ... },\n    "edge-case": { ... }\n  }\n}\n</vent_run>\n\nOne suite file per platform/adapter. `connection` is declared once, `calls` is a named map of call specs. Each key becomes the call name. Run one call at a time with `--call <name>`.\n\n<config_connection>\n{\n  "connection": {\n    "adapter": "required -- websocket | livekit | vapi | retell | elevenlabs | bland",\n    "start_command": "shell command to start agent (relay only, required for local)",\n    "health_endpoint": "health check path after start_command (default: /health, relay only, required for local)",\n    "agent_url": "hosted custom agent URL (wss:// or https://). Use for BYO hosted agents.",\n    "agent_port": "local agent port (default: 3001, required for local)",\n    "platform": "optional authoring convenience for platform-direct adapters only. The CLI resolves this locally, creates/updates a saved platform connection, and strips raw provider secrets before submit. Do not use for websocket start_command or agent_url runs."\n  }\n}\n\n<credential_resolution>\nIMPORTANT: How to handle platform credentials (API keys, secrets, agent IDs):\n\nThere are two product modes:\n- `BYO agent runtime`: your agent owns its own provider credentials. This covers both `start_command` (local) and `agent_url` (hosted custom endpoint).\n- `Platform-direct runtime`: Vent talks to `vapi`, `retell`, `elevenlabs`, `bland`, or `livekit` directly. This is the only mode that uses saved platform connections.\n\n1. For `start_command` and `agent_url` runs, do NOT put Deepgram / ElevenLabs / OpenAI / other provider keys into Vent config unless the Vent adapter itself needs them. Those credentials belong to the user\'s local or hosted agent runtime.\n2. For platform-direct adapters (`vapi`, `retell`, `elevenlabs`, `bland`, `livekit`), the CLI auto-resolves credentials from `.env.local`, `.env`, and the current shell env. If those env vars already exist, you can omit credential fields from the config JSON entirely.\n3. If you include credential fields in the config, put the ACTUAL VALUE, NOT the env var name. WRONG: `"vapi_api_key": "VAPI_API_KEY"`. RIGHT: `"vapi_api_key": "sk-abc123..."` or omit the field.\n4. The CLI uses the resolved provider config to create or update a saved platform connection server-side, then submits only `platform_connection_id`. Users should not manually author `platform_connection_id`.\n5. To check whether credentials are already available, inspect `.env.local`, `.env`, and any relevant shell env visible to the CLI process.\n\nAuto-resolved env vars per platform:\n| Platform | Config field | Env var (auto-resolved from `.env.local`, `.env`, or shell env) |\n|----------|-------------|-----------------------------------|\n| Vapi | vapi_api_key | VAPI_API_KEY |\n| Vapi | vapi_assistant_id | VAPI_ASSISTANT_ID |\n| Bland | bland_api_key | BLAND_API_KEY |\n| Bland | bland_pathway_id | BLAND_PATHWAY_ID |\n| LiveKit | livekit_api_key | LIVEKIT_API_KEY |\n| LiveKit | livekit_api_secret | LIVEKIT_API_SECRET |\n| LiveKit | livekit_url | LIVEKIT_URL |\n| Retell | retell_api_key | RETELL_API_KEY |\n| Retell | retell_agent_id | RETELL_AGENT_ID |\n| ElevenLabs | elevenlabs_api_key | ELEVENLABS_API_KEY |\n| ElevenLabs | elevenlabs_agent_id | ELEVENLABS_AGENT_ID |\n\nThe CLI strips raw platform secrets before `/runs/submit`. Platform-direct runs go through a saved `platform_connection_id` automatically. BYO agent runs (`start_command` and `agent_url`) do not.\n</credential_resolution>\n\n<config_adapter_rules>\nWebSocket (local agent via relay):\n{\n  "connection": {\n    "adapter": "websocket",\n    "start_command": "npm run start",\n    "health_endpoint": "/health",\n    "agent_port": 3001\n  }\n}\n\nWebSocket (hosted custom agent):\n{\n  "connection": {\n    "adapter": "websocket",\n    "agent_url": "https://my-agent.fly.dev"\n  }\n}\n\nRetell:\n{\n  "connection": {\n    "adapter": "retell",\n    "platform": { "provider": "retell" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: RETELL_API_KEY, RETELL_AGENT_ID. Only add retell_api_key/retell_agent_id to the JSON if those env vars are not already available.\n\nBland:\n{\n  "connection": {\n    "adapter": "bland",\n    "platform": { "provider": "bland" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: BLAND_API_KEY, BLAND_PATHWAY_ID. Only add bland_api_key/bland_pathway_id to the JSON if those env vars are not already available.\nNote: All agent config (voice, model, tools, etc.) is set on the pathway itself, not in Vent config.\n\nVapi:\n{\n  "connection": {\n    "adapter": "vapi",\n    "platform": { "provider": "vapi" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: VAPI_API_KEY, VAPI_ASSISTANT_ID. Only add vapi_api_key/vapi_assistant_id to the JSON if those env vars are not already available.\nmax_concurrency for Vapi: Starter=10, Growth=50, Enterprise=100+. Ask the user which tier they\'re on. If unknown, default to 10.\nAll assistant config (voice, model, transcriber, interruption settings, etc.) is set on the Vapi assistant itself, not in Vent config.\n\nElevenLabs:\n{\n  "connection": {\n    "adapter": "elevenlabs",\n    "platform": { "provider": "elevenlabs" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: ELEVENLABS_API_KEY, ELEVENLABS_AGENT_ID. Only add elevenlabs_api_key/elevenlabs_agent_id to the JSON if those env vars are not already available.\n\nLiveKit:\n{\n  "connection": {\n    "adapter": "livekit",\n    "platform": {\n      "provider": "livekit",\n      "livekit_agent_name": "my-agent",\n      "max_concurrency": 5\n    }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: LIVEKIT_API_KEY, LIVEKIT_API_SECRET, LIVEKIT_URL. Only add these to the JSON if those env vars are not already available.\nlivekit_agent_name is optional -- only needed if the agent registers with an explicit agent_name in WorkerOptions. Omit for automatic dispatch.\nThe livekit adapter requires the LiveKit Agents SDK. It depends on Agents SDK signals (lk.agent.state, lk.transcription) for readiness detection, turn timing, and component latency. Custom LiveKit participants not using the Agents SDK should use the websocket adapter with a relay instead.\nmax_concurrency: Free/Build=5, Ship=20, Scale=50+. Ask the user which tier they\'re on. If unknown, default to 5.\n</config_adapter_rules>\n</config_connection>\n\n\n<call_config>\n<tool_call_capture>\nvapi/retell/elevenlabs/bland: automatic via platform API (no user code needed).\nWebSocket/WebRTC: user\'s agent must emit tool calls:\n  WebSocket \u2014 JSON text frame: {"type":"tool_call","name":"...","arguments":{},"result":{},"successful":true,"duration_ms":150}\n  WebRTC/LiveKit \u2014 publishData() or sendText() on topic "vent:tool-calls". Same JSON.\n</tool_call_capture>\n\n<component_timing>\nPlatform adapters (vapi/retell/elevenlabs/bland/livekit) get STT/LLM/TTS breakdown automatically.\nWebSocket agents can opt in by sending a JSON text frame after each agent turn:\n  {"type":"vent:timing","stt_ms":120,"llm_ms":450,"tts_ms":80}\nAll fields optional. Send one per agent response. Without this, component_latency is omitted from results.\nWhen modifying a WebSocket agent\'s code, add this text frame after TTS completes to enable component latency reporting.\n</component_timing>\n\n<metadata_capture>\nWebSocket and LiveKit/WebRTC agents can also emit richer observability metadata:\n  {"type":"vent:session","platform":"custom","provider_call_id":"call_123","provider_session_id":"session_abc"}\n  {"type":"vent:call-metadata","call_metadata":{"recording_url":"https://...","cost_usd":0.12,"provider_debug_urls":{"log":"https://..."}}}\n  {"type":"vent:debug-url","label":"trace","url":"https://..."}\n  {"type":"vent:session-report","report":{"room_name":"room-123","events":[...],"metrics":[...]}}\n  {"type":"vent:metrics","event":"metrics_collected","metric_type":"eou","metrics":{"speechId":"speech_123","endOfUtteranceDelayMs":420}}\n  {"type":"vent:function-tools-executed","event":"function_tools_executed","hasAgentHandoff":true,"tool_calls":[{"name":"lookup_customer","arguments":{"id":"123"}}]}\n  {"type":"vent:conversation-item","event":"conversation_item_added","item":{"type":"agent_handoff","newAgentId":"billing-agent"}}\n  {"type":"vent:session-usage","usage":{"llm":{"promptTokens":123,"completionTokens":45}}}\nTransport:\n  WebSocket \u2014 send JSON text frames with these payloads. WebSocket agents may also emit {"type":"vent:transcript","role":"caller","text":"I need to reschedule","turn_index":0} when they have native transcript text.\n  WebRTC/LiveKit \u2014 publishData() or sendText() on the matching "vent:*" topic, e.g. topic "vent:call-metadata" with the JSON body above.\nFor LiveKit, transcript and timing stay authoritative from native room signals (`lk.transcription`, `lk.agent.state`). Do not emit `vent:transcript` from LiveKit agents.\nFor LiveKit Node agents, prefer the first-party helper instead of manual forwarding:\n```ts\nimport { instrumentLiveKitAgent } from "@vent-hq/livekit";\n\nconst vent = instrumentLiveKitAgent({\n  ctx,\n  session,\n});\n```\nThis helper must run inside the LiveKit agent runtime with the existing Agents SDK `session` and `ctx` objects. It is the Vent integration layer on top of the Agents SDK, not a replacement for it.\nInstall it with `npm install @vent-hq/livekit` after the package is published to the `vent-hq` npm org. Until then, use the workspace package from this repo.\nThis automatically publishes only the in-agent-only LiveKit signals: `metrics_collected`, `function_tools_executed`, `conversation_item_added`, and a session report on close/shutdown.\nDo not use it to mirror room-visible signals like transcript, agent state timing, or room/session ID \u2014 Vent already gets those from LiveKit itself.\nFor LiveKit inside-agent forwarding, prefer sending the raw LiveKit event payloads on:\n  `vent:metrics`\n  `vent:function-tools-executed`\n  `vent:conversation-item`\n  `vent:session-usage`\nUse these metadata events when the agent runtime already knows native IDs, recordings, warnings, debug links, session reports, metrics events, or handoff artifacts. This gives custom and LiveKit agents parity with hosted adapters without needing a LiveKit Cloud connector.\n</metadata_capture>\n\n<config_call>\nEach call in the `calls` map. The key is the call name (e.g. `"reschedule-appointment"`, not `"call-1"`).\n{\n      "caller_prompt": "required \u2014 caller persona and behavior (name -> goal -> emotion -> conditional behavior)",\n      "max_turns": "required \u2014 default 6",\n      "silence_threshold_ms": "optional \u2014 end-of-turn threshold ms (default 800, 200-10000). 800-1200 FAQ, 2000-3000 tool calls, 3000-5000 complex reasoning.",\n      "persona": "optional \u2014 caller behavior controls",\n      {\n        "pace": "slow | normal | fast",\n        "clarity": "clear | vague | rambling",\n        "disfluencies": "true | false",\n        "cooperation": "cooperative | reluctant | hostile",\n        "emotion": "neutral | cheerful | confused | frustrated | skeptical | rushed",\n        "interruption_style": "optional preplanned interrupt tendency: low | high. If set, Vent may pre-plan a caller cut-in before the agent turn starts. It does NOT make a mid-turn interrupt LLM call.",\n        "memory": "reliable | unreliable",\n        "intent_clarity": "clear | indirect | vague",\n        "confirmation_style": "explicit | vague"\n      },\n      "audio_actions": "optional \u2014 per-turn audio stress calls",\n      [\n        { "action": "interrupt", "at_turn": "N", "prompt": "what caller says" },\n        { "action": "inject_noise", "at_turn": "N", "noise_type": "babble | white | pink", "snr_db": "0-40" },\n        { "action": "split_sentence", "at_turn": "N", "split": { "part_a": "...", "part_b": "...", "pause_ms": "500-5000" } },\n        { "action": "noise_on_caller", "at_turn": "N" }\n      ],\n      "prosody": "optional \u2014 Hume emotion analysis (default false)",\n      "caller_audio": "optional \u2014 omit for clean audio",\n      {\n        "noise": { "type": "babble | white | pink", "snr_db": "0-40" },\n        "speed": "0.5-2.0 (1.0 = normal)",\n        "speakerphone": "true | false",\n        "mic_distance": "close | normal | far",\n        "clarity": "0.0-1.0 (1.0 = perfect)",\n        "accent": "american | british | australian | filipino | spanish_mexican | spanish_peninsular | spanish_colombian | spanish_argentine | german | french | italian | dutch | japanese",\n        "packet_loss": "0.0-0.3",\n        "jitter_ms": "0-100"\n      },\n      "language": "optional \u2014 ISO 639-1: en, es, fr, de, it, nl, ja"\n}\n\nInterruption rules:\n- `audio_actions: [{ "action": "interrupt", ... }]` is the deterministic per-turn interrupt test. Prefer this for evaluation.\n- `persona.interruption_style` is only a preplanned caller tendency. If used, Vent decides before the agent response starts whether this turn may cut in.\n- Vent no longer pauses mid-turn to ask a second LLM whether to interrupt.\n- For production-faithful testing, prefer explicit `audio_actions.interrupt` over persona interruption.\n\n<examples_call>\n<simple_suite_example>\n{\n  "connection": {\n    "adapter": "vapi",\n    "platform": { "provider": "vapi" }\n  },\n  "calls": {\n    "reschedule-appointment": {\n      "caller_prompt": "You are Maria, calling to reschedule her dentist appointment from Thursday to next Tuesday. She\'s in a hurry and wants this done quickly.",\n      "max_turns": 8\n    },\n    "cancel-appointment": {\n      "caller_prompt": "You are Tom, calling to cancel his appointment for Friday. He\'s calm and just wants confirmation.",\n      "max_turns": 6\n    }\n  }\n}\n</simple_suite_example>\n\n<advanced_call_example>\nA call entry with advanced options (persona, audio actions, prosody):\n{\n  "noisy-interruption-booking": {\n    "caller_prompt": "You are James, an impatient customer calling from a loud coffee shop to book a plumber for tomorrow morning. You interrupt the agent mid-sentence when they start listing availability \u2014 you just want the earliest slot.",\n    "max_turns": 12,\n    "persona": { "pace": "fast", "cooperation": "reluctant", "emotion": "rushed", "interruption_style": "high" },\n    "audio_actions": [\n      { "action": "interrupt", "at_turn": 3, "prompt": "Just give me the earliest one!" },\n      { "action": "inject_noise", "at_turn": 1, "noise_type": "babble", "snr_db": 15 }\n    ],\n    "caller_audio": { "noise": { "type": "babble", "snr_db": 20 }, "speed": 1.3 },\n    "prosody": true\n  }\n}\n</advanced_call_example>\n\n</examples_call>\n</config_call>\n\n<output_conversation_test>\n{\n  "name": "sarah-hotel-booking",\n  "status": "completed",\n  "caller_prompt": "You are Sarah, calling to book...",\n  "duration_ms": 45200,\n  "error": null,\n  "transcript": [\n    { "role": "caller", "text": "Hi, I\'d like to book..." },\n    { "role": "agent", "text": "Sure! What date?", "ttfb_ms": 650, "ttfw_ms": 780, "audio_duration_ms": 2400 },\n    { "role": "agent", "text": "Let me check avail\u2014", "ttfb_ms": 540, "ttfw_ms": 620, "audio_duration_ms": 1400, "interrupted": true },\n    { "role": "caller", "text": "Just the earliest slot please", "audio_duration_ms": 900, "is_interruption": true },\n    { "role": "agent", "text": "Sure, the earliest is 9 AM tomorrow.", "ttfb_ms": 220, "ttfw_ms": 260, "audio_duration_ms": 2100 }\n  ],\n  "latency": {\n    "response_time_ms": 890, "response_time_source": "ttfw",\n    "p50_response_time_ms": 850, "p90_response_time_ms": 1100, "p95_response_time_ms": 1400, "p99_response_time_ms": 1550,\n    "first_response_time_ms": 1950,\n    "mean_ttfw_ms": 890, "p50_ttfw_ms": 850, "p95_ttfw_ms": 1400, "p99_ttfw_ms": 1550,\n    "first_turn_ttfw_ms": 1950, "total_silence_ms": 4200, "mean_turn_gap_ms": 380,\n    "drift_slope_ms_per_turn": -45.2, "mean_silence_pad_ms": 128, "mouth_to_ear_est_ms": 1020\n  },\n  "transcript_quality": {\n    "wer": 0.04,\n    "hallucination_events": [\n      { "error_count": 5, "reference_text": "triple five one two", "hypothesis_text": "five five five nine two" }\n    ],\n    "repetition_score": 0.05,\n    "reprompt_count": 0,\n    "filler_word_rate": 0.8,\n    "words_per_minute": 148\n  },\n  "audio_analysis": {\n    "caller_talk_time_ms": 12400,\n    "agent_talk_time_ms": 28500,\n    "agent_speech_ratio": 0.72,\n    "talk_ratio_vad": 0.69,\n    "interruption_rate": 0.25,\n    "interruption_count": 1,\n    "agent_overtalk_after_barge_in_ms": 280,\n    "agent_interrupting_user_rate": 0.0,\n    "agent_interrupting_user_count": 0,\n    "missed_response_windows": 0,\n    "longest_monologue_ms": 5800,\n    "silence_gaps_over_2s": 1,\n    "total_internal_silence_ms": 2400,\n    "mean_agent_speech_segment_ms": 3450\n  },\n  "tool_calls": {\n    "total": 2, "successful": 2, "failed": 0, "mean_latency_ms": 340,\n    "names": ["check_availability", "book_appointment"],\n    "observed": [{ "name": "check_availability", "arguments": { "date": "2026-03-12" }, "result": { "slots": ["09:00", "10:00"] }, "successful": true, "latency_ms": 280, "turn_index": 3 }]\n  },\n  "component_latency": {\n    "mean_stt_ms": 120, "mean_llm_ms": 450, "mean_tts_ms": 80,\n    "p95_stt_ms": 180, "p95_llm_ms": 620, "p95_tts_ms": 110,\n    "mean_speech_duration_ms": 2100,\n    "bottleneck": "llm"\n  },\n  "call_metadata": {\n    "platform": "vapi",\n    "cost_usd": 0.08,\n    "recording_url": "https://example.com/recording",\n    "ended_reason": "customer_ended_call",\n    "transfers": []\n  },\n  "warnings": [],\n  "audio_actions": [],\n  "emotion": {\n    "naturalness": 0.72, "mean_calmness": 0.65, "mean_confidence": 0.58, "peak_frustration": 0.08, "emotion_trajectory": "stable"\n  }\n}\n\nAlways present: name, status, caller_prompt, duration_ms, error, transcript, tool_calls, warnings, audio_actions. Nullable when analysis didn\'t run: latency, transcript_quality, audio_analysis, component_latency, call_metadata, emotion (requires prosody: true), debug (requires --verbose).\n\n### Result presentation\n\nWhen you report a conversation result to the user, always include:\n\n1. **Summary** \u2014 the overall verdict and the 1-3 most important findings.\n2. **Transcript summary** \u2014 a short narrative of what happened in the call.\n3. **Recording URL** \u2014 include `call_metadata.recording_url` when present; explicitly say when it is unavailable.\n4. **Next steps** \u2014 concrete fixes, follow-up tests, or why no change is needed.\n\nUse metrics to support the summary, not as the whole answer. Do not dump raw numbers without interpretation.\n\nWhen `call_metadata.transfer_attempted` is present, explicitly say whether the transfer only appeared attempted or was mechanically verified as completed (`call_metadata.transfer_completed`). Use `call_metadata.transfers[]` to report transfer type, destination, status, and sources.\n\n### Judging guidance\n\nUse the transcript, metrics, test scenario, and relevant agent instructions/system prompt to judge:\n\n| Dimension | What to check |\n|--------|----------------|\n| **Hallucination detection** | Check whether the agent stated anything not grounded in its instructions, tools, or the conversation itself. Treat `transcript_quality.hallucination_events` only as a speech-recognition warning signal, not proof of agent hallucination. |\n| **Instruction following** | Compare the agent\'s behavior against its system prompt and the test\'s expected constraints. |\n| **Context retention** | Check whether the agent forgot or contradicted information established earlier in the call. |\n| **Semantic accuracy** | Check whether the agent correctly understood the caller\'s intent and responded to the real request. |\n| **Goal completion** | Decide whether the agent achieved what the test scenario was designed to verify. |\n| **Transfer correctness** | For transfer scenarios, judge whether transfer was appropriate, whether it completed, whether it went to the expected destination, and whether enough context was passed during the handoff. |\n\n### Interruption evaluation\n\nWhen the transcript contains `interrupted: true` / `is_interruption: true` turns, evaluate these metrics by reading the transcript:\n\n| Metric | How to evaluate | Target |\n|--------|----------------|--------|\n| **Recovery rate** | For each interrupted turn: does the post-interrupt agent response acknowledge or address the interruption? (e.g., "Sure, the earliest is 9 AM" after being cut off mid-availability-list) | >90% |\n| **Context retention** | After the interruption, does the agent remember pre-interrupt conversation state? (e.g., still knows the caller\'s name, booking details, etc.) | >95% |\n| **Agent overtalk after barge-in** | Use `audio_analysis.agent_overtalk_after_barge_in_ms` when available. Lower is better because it measures how long the agent kept speaking after the caller cut in. | <500ms acceptable |\n| **Agent interrupting user rate** | Use `audio_analysis.agent_interrupting_user_rate` and the transcript to see whether the agent starts speaking before the caller finished. | 0 ideal |\n\nReport these alongside standard metrics when interruption calls run. Flag any turn where the agent ignores the interruption, repeats itself from scratch, or loses context.\n</output_conversation_test>\n</call_config>\n\n\n## Output\n\n- **Exit codes**: 0=pass, 1=fail, 2=error\n- The `run` command outputs **complete results as pretty-printed JSON** \u2014 including full transcript, latency metrics, tool calls, and audio analysis for every call. Do NOT run a separate `vent status` command \u2014 all data is already in the output.\n\n## Vent Access Token\n\nRun `npx vent-hq login` or set `VENT_ACCESS_TOKEN` env var.\nVent provides DEEPGRAM_API_KEY and ANTHROPIC_API_KEY automatically.\n';
 // src/skills/cursor.md
-var cursor_default = '---\ndescription: Vent \u2014 Voice agent calls. Run calls against your voice agent, get pass/fail results. Use when the user asks to run voice-agent calls or when you need to verify voice agent behavior after code changes.\nalwaysApply: true\n---\n\n# Vent \u2014 Voice Agent Calls\n\nCall voice agents from the terminal. Calls run in the cloud (30-120s).\n\n## When to Call (read this first)\n\n- After modifying voice agent code (system prompt, tools, handlers): ALWAYS run calls\n- After changing audio config: run calls\n- Before marking a task complete that touches agent behavior: run calls\n- Do NOT skip call verification \u2014 voice agents are non-deterministic, small changes can break flows\n\n## Setup (if no Vent access token)\n\nIf `~/.vent/credentials` does not exist and `VENT_ACCESS_TOKEN` is not set:\n\n```bash\nnpx vent-hq init\n```\n\nOne command. No email, no browser, no human interaction needed. You get 10 free runs instantly.\n\n## Commands\n\n| Command | Purpose |\n|---------|---------|\n| `npx vent-hq init` | First-time setup (creates account + installs skills) |\n| `npx vent-hq agent start -f .vent/suite.<adapter>.json` | Start one shared local agent session (required for `start_command`) |\n| `npx vent-hq agent stop <session-id>` | Close a shared local agent session |\n| `npx vent-hq run -f .vent/suite.<adapter>.json` | Run a call from suite file (auto-selects if only one call) |\n| `npx vent-hq run -f .vent/suite.<adapter>.json --call <name>` | Run a specific named call |\n| `npx vent-hq stop <run-id>` | Cancel a queued or running call |\n| `npx vent-hq status <run-id>` | Check results of a previous run |\n\n\n## Critical Rules\n\n1. **Set timeout on shell calls** \u2014 Calls take 30-120s but can reach 5 minutes. Always set a 300-second (5 min) timeout on shell commands that run calls.\n2. **Handle backgrounded commands** \u2014 If a call command gets moved to background by the system, wait for it to complete before proceeding. Never end your response without delivering call results.\n3. **Output format** \u2014 In non-TTY mode (when run by an agent), every SSE event is written to stdout as a JSON line. Results are always in stdout.\n4. **This skill is self-contained** \u2014 The full config schema is below. Do NOT re-read this file.\n5. **Always analyze results** \u2014 The run command outputs complete JSON with full transcript, latency, and tool calls. Analyze this output directly \u2014 do NOT run `vent status` afterwards, the data is already there.\n\n## Workflow\n\n### First time: create the call suite\n\n1. Read the voice agent\'s codebase \u2014 understand its system prompt, tools, intents, and domain.\n2. Read the **Full Config Schema** section below for all available fields.\n3. Create the suite file in `.vent/` using the naming convention: `.vent/suite.<adapter>.json` (e.g., `.vent/suite.vapi.json`, `.vent/suite.websocket.json`, `.vent/suite.retell.json`). This prevents confusion when multiple adapters are tested in the same project.\n   - Name calls after specific flows (e.g., `"reschedule-appointment"`, not `"call-1"`)\n   - Write `caller_prompt` as a realistic persona with a specific goal, based on the agent\'s domain\n   - Set `max_turns` based on the flow complexity (simple FAQ: 4-6, booking: 8-12, complex: 12-20)\n\n### Multiple suite files\n\nIf `.vent/` contains more than one suite file, **always check which adapter each suite uses before running**. Read the `connection.adapter` field in each file. Never run a suite intended for a different adapter \u2014 results will be meaningless or fail. When reporting results, always state which suite file produced them (e.g., "Results from `.vent/suite.vapi.json`:").\n\n### Subsequent runs \u2014 reuse the existing suite\n\nA matching `.vent/suite.<adapter>.json` already exists? Just re-run it. No need to recreate.\n\n### Run calls\n\n1. If the suite uses `start_command`, start the shared local session first:\n   ```\n   npx vent-hq agent start -f .vent/suite.<adapter>.json\n   ```\n\n2. Run calls:\n   ```\n   # suite with one call (auto-selects)\n   npx vent-hq run -f .vent/suite.<adapter>.json\n\n   # suite with multiple calls \u2014 pick one by name\n   npx vent-hq run -f .vent/suite.<adapter>.json --call happy-path\n\n   # local start_command \u2014 add --session\n   npx vent-hq run -f .vent/suite.<adapter>.json --call happy-path --session <session-id>\n   ```\n\n3. To run multiple calls from the same suite, run each as a separate command:\n   ```\n   npx vent-hq run -f .vent/suite.vapi.json --call happy-path\n   npx vent-hq run -f .vent/suite.vapi.json --call edge-case\n   ```\n\n4. Analyze each result, identify failures, correlate with the codebase, and fix.\n5. **Compare with previous run** \u2014 Vent saves full result JSON to `.vent/runs/` after every run. Read the second-most-recent JSON in `.vent/runs/` and compare against the current run: status flips, TTFW p50/p95 changes >20%, tool call count drops, cost increases >30%, transcript divergence. Correlate with `git diff` between the two runs\' git SHAs. Skip if no previous run exists.\n\n## Connection\n\n- **BYO agent runtime**: your agent owns its own provider credentials. Use `start_command` for a local agent or `agent_url` for a hosted custom endpoint.\n- **Platform-direct runtime**: use adapter `vapi | retell | elevenlabs | bland | livekit`. This is the only mode where Vent itself needs provider credentials and saved platform connections apply.\n\n## WebSocket Protocol (BYO agents)\n\nWhen using `adapter: "websocket"`, Vent communicates with the agent over a single WebSocket connection:\n\n- **Binary frames** \u2192 PCM audio (16-bit mono, configurable sample rate)\n- **Text frames** \u2192 optional JSON events the agent can send for better test accuracy:\n\n| Event | Format | Purpose |\n|-------|--------|---------|\n| `speech-update` | `{"type":"speech-update","status":"started"\\|"stopped"}` | Enables platform-assisted turn detection (more accurate than VAD alone) |\n| `tool_call` | `{"type":"tool_call","name":"...","arguments":{...},"result":...,"successful":bool,"duration_ms":number}` | Reports tool calls for observability |\n| `vent:timing` | `{"type":"vent:timing","stt_ms":number,"llm_ms":number,"tts_ms":number}` | Reports component latency breakdown per turn |\n\nVent sends `{"type":"end-call"}` to the agent when the test is done.\n\nAll text frames are optional \u2014 audio-only agents work fine with VAD-based turn detection.\n\n## Full Config Schema\n\n- ALL calls MUST reference the agent\'s real context (system prompt, tools, knowledge base) from the codebase.\n\n<vent_run>\n{\n  "connection": { ... },\n  "calls": {\n    "happy-path": { ... },\n    "edge-case": { ... }\n  }\n}\n</vent_run>\n\nOne suite file per platform/adapter. `connection` is declared once, `calls` is a named map of call specs. Each key becomes the call name. Run one call at a time with `--call <name>`.\n\n<config_connection>\n{\n  "connection": {\n    "adapter": "required -- websocket | livekit | vapi | retell | elevenlabs | bland",\n    "start_command": "shell command to start agent (relay only, required for local)",\n    "health_endpoint": "health check path after start_command (default: /health, relay only, required for local)",\n    "agent_url": "hosted custom agent URL (wss:// or https://). Use for BYO hosted agents.",\n    "agent_port": "local agent port (default: 3001, required for local)",\n    "platform": "optional authoring convenience for platform-direct adapters only. The CLI resolves this locally, creates/updates a saved platform connection, and strips raw provider secrets before submit. Do not use for websocket start_command or agent_url runs."\n  }\n}\n\n<credential_resolution>\nIMPORTANT: How to handle platform credentials (API keys, secrets, agent IDs):\n\nThere are two product modes:\n- `BYO agent runtime`: your agent owns its own provider credentials. This covers both `start_command` (local) and `agent_url` (hosted custom endpoint).\n- `Platform-direct runtime`: Vent talks to `vapi`, `retell`, `elevenlabs`, `bland`, or `livekit` directly. This is the only mode that uses saved platform connections.\n\n1. For `start_command` and `agent_url` runs, do NOT put Deepgram / ElevenLabs / OpenAI / other provider keys into Vent config unless the Vent adapter itself needs them. Those credentials belong to the user\'s local or hosted agent runtime.\n2. For platform-direct adapters (`vapi`, `retell`, `elevenlabs`, `bland`, `livekit`), the CLI auto-resolves credentials from `.env.local`, `.env`, and the current shell env. If those env vars already exist, you can omit credential fields from the config JSON entirely.\n3. If you include credential fields in the config, put the ACTUAL VALUE, NOT the env var name. WRONG: `"vapi_api_key": "VAPI_API_KEY"`. RIGHT: `"vapi_api_key": "sk-abc123..."` or omit the field.\n4. The CLI uses the resolved provider config to create or update a saved platform connection server-side, then submits only `platform_connection_id`. Users should not manually author `platform_connection_id`.\n5. To check whether credentials are already available, inspect `.env.local`, `.env`, and any relevant shell env visible to the CLI process.\n\nAuto-resolved env vars per platform:\n| Platform | Config field | Env var (auto-resolved from `.env.local`, `.env`, or shell env) |\n|----------|-------------|-----------------------------------|\n| Vapi | vapi_api_key | VAPI_API_KEY |\n| Vapi | vapi_assistant_id | VAPI_ASSISTANT_ID |\n| Bland | bland_api_key | BLAND_API_KEY |\n| Bland | bland_pathway_id | BLAND_PATHWAY_ID |\n| LiveKit | livekit_api_key | LIVEKIT_API_KEY |\n| LiveKit | livekit_api_secret | LIVEKIT_API_SECRET |\n| LiveKit | livekit_url | LIVEKIT_URL |\n| Retell | retell_api_key | RETELL_API_KEY |\n| Retell | retell_agent_id | RETELL_AGENT_ID |\n| ElevenLabs | elevenlabs_api_key | ELEVENLABS_API_KEY |\n| ElevenLabs | elevenlabs_agent_id | ELEVENLABS_AGENT_ID |\n\nThe CLI strips raw platform secrets before `/runs/submit`. Platform-direct runs go through a saved `platform_connection_id` automatically. BYO agent runs (`start_command` and `agent_url`) do not.\n</credential_resolution>\n\n<config_adapter_rules>\nWebSocket (local agent via relay):\n{\n  "connection": {\n    "adapter": "websocket",\n    "start_command": "npm run start",\n    "health_endpoint": "/health",\n    "agent_port": 3001\n  }\n}\n\nWebSocket (hosted custom agent):\n{\n  "connection": {\n    "adapter": "websocket",\n    "agent_url": "https://my-agent.fly.dev"\n  }\n}\n\nRetell:\n{\n  "connection": {\n    "adapter": "retell",\n    "platform": { "provider": "retell" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: RETELL_API_KEY, RETELL_AGENT_ID. Only add retell_api_key/retell_agent_id to the JSON if those env vars are not already available.\n\nBland:\n{\n  "connection": {\n    "adapter": "bland",\n    "platform": { "provider": "bland" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: BLAND_API_KEY, BLAND_PATHWAY_ID. Only add bland_api_key/bland_pathway_id to the JSON if those env vars are not already available.\nNote: All agent config (voice, model, tools, etc.) is set on the pathway itself, not in Vent config.\n\nVapi:\n{\n  "connection": {\n    "adapter": "vapi",\n    "platform": { "provider": "vapi" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: VAPI_API_KEY, VAPI_ASSISTANT_ID. Only add vapi_api_key/vapi_assistant_id to the JSON if those env vars are not already available.\nmax_concurrency for Vapi: Starter=10, Growth=50, Enterprise=100+. Ask the user which tier they\'re on. If unknown, default to 10.\nAll assistant config (voice, model, transcriber, interruption settings, etc.) is set on the Vapi assistant itself, not in Vent config.\n\nElevenLabs:\n{\n  "connection": {\n    "adapter": "elevenlabs",\n    "platform": { "provider": "elevenlabs" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: ELEVENLABS_API_KEY, ELEVENLABS_AGENT_ID. Only add elevenlabs_api_key/elevenlabs_agent_id to the JSON if those env vars are not already available.\n\nLiveKit:\n{\n  "connection": {\n    "adapter": "livekit",\n    "platform": {\n      "provider": "livekit",\n      "livekit_agent_name": "my-agent",\n      "max_concurrency": 5\n    }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: LIVEKIT_API_KEY, LIVEKIT_API_SECRET, LIVEKIT_URL. Only add these to the JSON if those env vars are not already available.\nlivekit_agent_name is optional -- only needed if the agent registers with an explicit agent_name in WorkerOptions. Omit for automatic dispatch.\nmax_concurrency: Free/Build=5, Ship=20, Scale=50+. Ask the user which tier they\'re on. If unknown, default to 5.\n</config_adapter_rules>\n</config_connection>\n\n\n<call_config>\n<tool_call_capture>\nvapi/retell/elevenlabs/bland: automatic via platform API (no user code needed).\nWebSocket/WebRTC: user\'s agent must emit tool calls:\n  WebSocket \u2014 JSON text frame: {"type":"tool_call","name":"...","arguments":{},"result":{},"successful":true,"duration_ms":150}\n  WebRTC/LiveKit \u2014 publishData() or sendText() on topic "vent:tool-calls". Same JSON.\n</tool_call_capture>\n\n<component_timing>\nPlatform adapters (vapi/retell/elevenlabs/bland/livekit) get STT/LLM/TTS breakdown automatically.\nWebSocket agents can opt in by sending a JSON text frame after each agent turn:\n  {"type":"vent:timing","stt_ms":120,"llm_ms":450,"tts_ms":80}\nAll fields optional. Send one per agent response. Without this, component_latency is omitted from results.\nWhen modifying a WebSocket agent\'s code, add this text frame after TTS completes to enable component latency reporting.\n</component_timing>\n\n<config_call>\nEach call in the `calls` map. The key is the call name (e.g. `"reschedule-appointment"`, not `"call-1"`).\n{\n      "caller_prompt": "required \u2014 caller persona and behavior (name -> goal -> emotion -> conditional behavior)",\n    "max_turns": "required \u2014 default 6",\n    "silence_threshold_ms": "optional \u2014 end-of-turn threshold ms (default 800, 200-10000). 800-1200 FAQ, 2000-3000 tool calls, 3000-5000 complex reasoning.",\n    "persona": "optional \u2014 caller behavior controls",\n    {\n      "pace": "slow | normal | fast",\n      "clarity": "clear | vague | rambling",\n      "disfluencies": "true | false",\n      "cooperation": "cooperative | reluctant | hostile",\n      "emotion": "neutral | cheerful | confused | frustrated | skeptical | rushed",\n      "interruption_style": "low (~3/10 turns) | high (~7/10 turns)",\n      "memory": "reliable | unreliable",\n      "intent_clarity": "clear | indirect | vague",\n      "confirmation_style": "explicit | vague"\n    },\n    "audio_actions": "optional \u2014 per-turn audio stress calls",\n    [\n      { "action": "interrupt", "at_turn": "N", "prompt": "what caller says" },\n      { "action": "silence", "at_turn": "N", "duration_ms": "1000-30000" },\n      { "action": "inject_noise", "at_turn": "N", "noise_type": "babble | white | pink", "snr_db": "0-40" },\n      { "action": "split_sentence", "at_turn": "N", "split": { "part_a": "...", "part_b": "...", "pause_ms": "500-5000" } },\n      { "action": "noise_on_caller", "at_turn": "N" }\n    ],\n    "prosody": "optional \u2014 Hume emotion analysis (default false)",\n    "caller_audio": "optional \u2014 omit for clean audio",\n    {\n      "noise": { "type": "babble | white | pink", "snr_db": "0-40" },\n      "speed": "0.5-2.0 (1.0 = normal)",\n      "speakerphone": "true | false",\n      "mic_distance": "close | normal | far",\n      "clarity": "0.0-1.0 (1.0 = perfect)",\n      "accent": "american | british | australian | filipino | spanish_mexican | spanish_peninsular | spanish_colombian | spanish_argentine | german | french | italian | dutch | japanese",\n      "packet_loss": "0.0-0.3",\n      "jitter_ms": "0-100"\n    },\n    "language": "optional \u2014 ISO 639-1: en, es, fr, de, it, nl, ja"\n}\n\n<examples_call>\n<simple_suite_example>\n{\n  "connection": {\n    "adapter": "vapi",\n    "platform": { "provider": "vapi" }\n  },\n  "calls": {\n    "reschedule-appointment": {\n      "caller_prompt": "You are Maria, calling to reschedule her dentist appointment from Thursday to next Tuesday. She\'s in a hurry and wants this done quickly.",\n      "max_turns": 8\n    },\n    "cancel-appointment": {\n      "caller_prompt": "You are Tom, calling to cancel his appointment for Friday. He\'s calm and just wants confirmation.",\n      "max_turns": 6\n    }\n  }\n}\n</simple_suite_example>\n\n<advanced_call_example>\nA call entry with advanced options (persona, audio actions, prosody):\n{\n  "noisy-interruption-booking": {\n    "caller_prompt": "You are James, an impatient customer calling from a loud coffee shop to book a plumber for tomorrow morning. You interrupt the agent mid-sentence when they start listing availability \u2014 you just want the earliest slot.",\n    "max_turns": 12,\n    "persona": { "pace": "fast", "cooperation": "reluctant", "emotion": "rushed", "interruption_style": "high" },\n    "audio_actions": [\n      { "action": "interrupt", "at_turn": 3, "prompt": "Just give me the earliest one!" },\n      { "action": "inject_noise", "at_turn": 1, "noise_type": "babble", "snr_db": 15 }\n    ],\n    "caller_audio": { "noise": { "type": "babble", "snr_db": 20 }, "speed": 1.3 },\n    "prosody": true\n  }\n}\n</advanced_call_example>\n\n</examples_call>\n</config_call>\n\n<output_conversation_test>\n{\n  "name": "sarah-hotel-booking",\n  "status": "completed",\n  "caller_prompt": "You are Sarah, calling to book...",\n  "duration_ms": 45200,\n  "error": null,\n  "transcript": [\n    { "role": "caller", "text": "Hi, I\'d like to book..." },\n    { "role": "agent", "text": "Sure! What date?", "ttfb_ms": 650, "ttfw_ms": 780, "audio_duration_ms": 2400 },\n    { "role": "agent", "text": "Let me check avail\u2014", "ttfb_ms": 540, "ttfw_ms": 620, "audio_duration_ms": 1400, "interrupted": true },\n    { "role": "caller", "text": "Just the earliest slot please", "audio_duration_ms": 900, "is_interruption": true },\n    { "role": "agent", "text": "Sure, the earliest is 9 AM tomorrow.", "ttfb_ms": 220, "ttfw_ms": 260, "audio_duration_ms": 2100 }\n  ],\n  "latency": {\n    "mean_ttfw_ms": 890, "p50_ttfw_ms": 850, "p95_ttfw_ms": 1400, "p99_ttfw_ms": 1550,\n    "first_turn_ttfw_ms": 1950, "total_silence_ms": 4200, "mean_turn_gap_ms": 380,\n    "drift_slope_ms_per_turn": -45.2, "mean_silence_pad_ms": 128, "mouth_to_ear_est_ms": 1020\n  },\n  "transcript_quality": {\n    "wer": 0.04,\n    "hallucination_events": [\n      { "error_count": 5, "reference_text": "triple five one two", "hypothesis_text": "five five five nine two" }\n    ],\n    "repetition_score": 0.05,\n    "reprompt_count": 0,\n    "filler_word_rate": 0.8,\n    "words_per_minute": 148\n  },\n  "audio_analysis": {\n    "agent_speech_ratio": 0.72,\n    "interruption_rate": 0.25,\n    "interruption_count": 1,\n    "barge_in_recovery_time_ms": 280,\n    "agent_interrupting_user_rate": 0.0,\n    "agent_interrupting_user_count": 0,\n    "missed_response_windows": 0,\n    "longest_monologue_ms": 5800,\n    "silence_gaps_over_2s": 1,\n    "total_internal_silence_ms": 2400,\n    "mean_agent_speech_segment_ms": 3450\n  },\n  "tool_calls": {\n    "total": 2, "successful": 2, "failed": 0, "mean_latency_ms": 340,\n    "names": ["check_availability", "book_appointment"],\n    "observed": [{ "name": "check_availability", "arguments": { "date": "2026-03-12" }, "result": { "slots": ["09:00", "10:00"] }, "successful": true, "latency_ms": 280, "turn_index": 3 }]\n  },\n  "call_metadata": {\n    "platform": "vapi",\n    "recording_url": "https://example.com/recording"\n  },\n  "warnings": [],\n  "audio_actions": [\n    { "at_turn": 5, "action": "silence", "metrics": { "agent_prompted": false, "unprompted_utterance_count": 0, "silence_duration_ms": 8000 } }\n  ],\n  "emotion": {\n    "naturalness": 0.72, "mean_calmness": 0.65, "mean_confidence": 0.58, "peak_frustration": 0.08, "emotion_trajectory": "stable"\n  }\n}\n\nAll fields optional except name, status, caller_prompt, duration_ms, transcript. Fields appear only when relevant analysis ran (e.g., emotion requires prosody: true).\n\n### Result presentation\n\nWhen you report a conversation result to the user, always include:\n\n1. **Summary** \u2014 the overall verdict and the 1-3 most important findings.\n2. **Transcript summary** \u2014 a short narrative of what happened in the call.\n3. **Recording URL** \u2014 include `call_metadata.recording_url` when present; explicitly say when it is unavailable.\n4. **Next steps** \u2014 concrete fixes, follow-up tests, or why no change is needed.\n\nUse metrics to support the summary, not as the whole answer. Do not dump raw numbers without interpretation.\n\nWhen `call_metadata.transfer_attempted` is present, explicitly say whether the transfer only appeared attempted or was mechanically verified as completed. If `call_metadata.transfers[*].verification` is present, use it to mention second-leg observation, connect latency, transcript/context summary, and whether context passing was verified.\n\n### Judging guidance\n\nUse the transcript, metrics, test scenario, and relevant agent instructions/system prompt to judge:\n\n| Dimension | What to check |\n|--------|----------------|\n| **Hallucination detection** | Check whether the agent stated anything not grounded in its instructions, tools, or the conversation itself. Treat `transcript_quality.hallucination_events` only as a speech-recognition warning signal, not proof of agent hallucination. |\n| **Instruction following** | Compare the agent\'s behavior against its system prompt and the test\'s expected constraints. |\n| **Context retention** | Check whether the agent forgot or contradicted information established earlier in the call. |\n| **Semantic accuracy** | Check whether the agent correctly understood the caller\'s intent and responded to the real request. |\n| **Goal completion** | Decide whether the agent achieved what the test scenario was designed to verify. |\n| **Transfer correctness** | For transfer scenarios, judge whether transfer was appropriate, whether it completed, whether it went to the expected destination, and whether enough context was passed during the handoff. |\n\n### Interruption evaluation\n\nWhen the transcript contains `interrupted: true` / `is_interruption: true` turns, evaluate these metrics by reading the transcript:\n\n| Metric | How to evaluate | Target |\n|--------|----------------|--------|\n| **Recovery rate** | For each interrupted turn: does the post-interrupt agent response acknowledge or address the interruption? | >90% |\n| **Context retention** | After the interruption, does the agent remember pre-interrupt conversation state? | >95% |\n| **Barge-in recovery time** | Use `audio_analysis.barge_in_recovery_time_ms` when available. Lower is better because it measures how long the agent kept speaking after the caller cut in. | <500ms acceptable |\n| **Agent interrupting user rate** | Use `audio_analysis.agent_interrupting_user_rate` and the transcript to see whether the agent starts speaking before the caller finished. | 0 ideal |\n\nReport these alongside standard metrics when interruption calls run.\n</output_conversation_test>\n</call_config>\n\n\n## Exit Codes\n\n0=pass, 1=fail, 2=error\n\n## Vent Access Token\n\nSet `VENT_ACCESS_TOKEN` env var or run `npx vent-hq login`.\n';
+var cursor_default = '---\ndescription: Vent \u2014 Voice agent calls. Run calls against your voice agent, get pass/fail results. Use when the user asks to run voice-agent calls or when you need to verify voice agent behavior after code changes.\nalwaysApply: true\n---\n\n# Vent \u2014 Voice Agent Calls\n\nCall voice agents from the terminal. Calls run in the cloud (30-120s).\n\n## When to Call (read this first)\n\n- After modifying voice agent code (system prompt, tools, handlers): ALWAYS run calls\n- After changing audio config: run calls\n- Before marking a task complete that touches agent behavior: run calls\n- Do NOT skip call verification \u2014 voice agents are non-deterministic, small changes can break flows\n\n## Setup (if no Vent access token)\n\nIf `~/.vent/credentials` does not exist and `VENT_ACCESS_TOKEN` is not set:\n\n```bash\nnpx vent-hq init\n```\n\nOne command. No email, no browser, no human interaction needed. You get 10 free runs instantly.\n\n## Commands\n\n| Command | Purpose |\n|---------|---------|\n| `npx vent-hq init` | First-time setup (creates account + installs skills) |\n| `npx vent-hq agent start -f .vent/suite.<adapter>.json` | Start one shared local agent session (required for `start_command`) |\n| `npx vent-hq agent stop <session-id>` | Close a shared local agent session |\n| `npx vent-hq run -f .vent/suite.<adapter>.json` | Run a call from suite file (auto-selects if only one call) |\n| `npx vent-hq run -f .vent/suite.<adapter>.json --verbose` | Include debug fields in the result JSON |\n| `npx vent-hq run -f .vent/suite.<adapter>.json --call <name>` | Run a specific named call |\n| `npx vent-hq stop <run-id>` | Cancel a queued or running call |\n| `npx vent-hq status <run-id>` | Check results of a previous run |\n| `npx vent-hq status <run-id> --verbose` | Re-print a run with debug fields included |\n\n## When To Use `--verbose`\n\nDefault output is enough for most work. It already includes:\n- transcript\n- latency\n- transcript quality (`wer` / `cer`)\n- audio analysis\n- tool calls\n- summary cost / recording / transfers\n\nUse `--verbose` only when you need debugging detail that is not in the default result:\n- per-turn debug fields: timestamps, caller decision mode, silence pad, STT confidence, platform transcript\n- raw signal analysis: `debug.signal_quality`\n- harness timings: `debug.harness_overhead`\n- raw prosody payload and warnings\n- raw provider warnings\n- per-turn component latency arrays\n- raw observed tool-call timeline\n- provider-specific metadata in `debug.provider_metadata`\n\nTrigger `--verbose` when:\n- transcript accuracy looks wrong and you need to inspect `platform_transcript`\n- latency is bad and you need per-turn/component breakdowns\n- interruptions/barge-in behavior looks wrong\n- tool-call execution looks inconsistent or missing\n- the provider returned warnings/errors or you need provider-native artifacts\n\nSkip `--verbose` when:\n- you only need pass/fail, transcript, latency, tool calls, recording, or summary\n- you are doing quick iteration on prompt wording and the normal result already explains the failure\n\n## Normalization Contract\n\nVent always returns one normalized result shape on `stdout` across adapters. Treat these as the stable categories:\n- `transcript`\n- `latency`\n- `transcript_quality`\n- `audio_analysis`\n- `tool_calls`\n- `component_latency`\n- `call_metadata`\n- `warnings`\n- `audio_actions`\n- `emotion`\n\nSource-of-truth policy:\n- Vent computes transcript, latency, and audio-quality metrics itself.\n- Hosted adapters choose the best source per category, usually provider post-call data for tool calls, call metadata, transfers, provider transcripts, and recordings.\n- Realtime provider events are fallback or enrichment only when post-call data is missing, delayed, weaker for that category, or provider-specific.\n- `LiveKit` helper events are the provider-native path for rich in-agent observability.\n- `websocket`/custom agents are realtime-native but still map into the same normalized categories.\n- Keep adapter-specific details in `call_metadata.provider_metadata` or `debug.provider_metadata`, not in new top-level fields.\n\n\n## Critical Rules\n\n1. **Set timeout on shell calls** \u2014 Calls take 30-120s but can reach 5 minutes. Always set a 300-second (5 min) timeout on shell commands that run calls.\n2. **Handle backgrounded commands** \u2014 If a call command gets moved to background by the system, wait for it to complete before proceeding. Never end your response without delivering call results.\n3. **Output format** \u2014 In non-TTY mode (when run by an agent), every SSE event is written to stdout as a JSON line. Results are always in stdout.\n4. **This skill is self-contained** \u2014 The full config schema is below. Do NOT re-read this file.\n5. **Always analyze results** \u2014 The run command outputs complete JSON with full transcript, latency, and tool calls. Use `--verbose` only when the default result is not enough to explain the failure. Analyze this output directly \u2014 do NOT run `vent status` afterwards unless you are re-checking a past run.\n\n## Workflow\n\n### First time: create the call suite\n\n1. Read the voice agent\'s codebase \u2014 understand its system prompt, tools, intents, and domain.\n2. Read the **Full Config Schema** section below for all available fields.\n3. Create the suite file in `.vent/` using the naming convention: `.vent/suite.<adapter>.json` (e.g., `.vent/suite.vapi.json`, `.vent/suite.websocket.json`, `.vent/suite.retell.json`). This prevents confusion when multiple adapters are tested in the same project.\n   - Name calls after specific flows (e.g., `"reschedule-appointment"`, not `"call-1"`)\n   - Write `caller_prompt` as a realistic persona with a specific goal, based on the agent\'s domain\n   - Set `max_turns` based on the flow complexity (simple FAQ: 4-6, booking: 8-12, complex: 12-20)\n\n### Multiple suite files\n\nIf `.vent/` contains more than one suite file, **always check which adapter each suite uses before running**. Read the `connection.adapter` field in each file. Never run a suite intended for a different adapter \u2014 results will be meaningless or fail. When reporting results, always state which suite file produced them (e.g., "Results from `.vent/suite.vapi.json`:").\n\n### Subsequent runs \u2014 reuse the existing suite\n\nA matching `.vent/suite.<adapter>.json` already exists? Just re-run it. No need to recreate.\n\n### Run calls\n\n1. If the suite uses `start_command`, start the shared local session first:\n   ```\n   npx vent-hq agent start -f .vent/suite.<adapter>.json\n   ```\n\n2. Run calls:\n   ```\n   # suite with one call (auto-selects)\n   npx vent-hq run -f .vent/suite.<adapter>.json\n\n   # suite with multiple calls \u2014 pick one by name\n   npx vent-hq run -f .vent/suite.<adapter>.json --call happy-path\n\n   # local start_command \u2014 add --session\n   npx vent-hq run -f .vent/suite.<adapter>.json --call happy-path --session <session-id>\n   ```\n\n3. To run multiple calls from the same suite, run each as a separate command:\n   ```\n   npx vent-hq run -f .vent/suite.vapi.json --call happy-path\n   npx vent-hq run -f .vent/suite.vapi.json --call edge-case\n   ```\n\n4. Analyze each result, identify failures, correlate with the codebase, and fix.\n5. **Compare with previous run** \u2014 Vent saves full result JSON to `.vent/runs/` after every run. Read the second-most-recent JSON in `.vent/runs/` and compare against the current run: status flips, TTFW p50/p95 changes >20%, tool call count drops, cost increases >30%, transcript divergence. Correlate with `git diff` between the two runs\' git SHAs. Skip if no previous run exists.\n\n## Connection\n\n- **BYO agent runtime**: your agent owns its own provider credentials. Use `start_command` for a local agent or `agent_url` for a hosted custom endpoint.\n- **Platform-direct runtime**: use adapter `vapi | retell | elevenlabs | bland | livekit`. This is the only mode where Vent itself needs provider credentials and saved platform connections apply.\n\n## WebSocket Protocol (BYO agents)\n\nWhen using `adapter: "websocket"`, Vent communicates with the agent over a single WebSocket connection:\n\n- **Binary frames** \u2192 PCM audio (16-bit mono, configurable sample rate)\n- **Text frames** \u2192 optional JSON events the agent can send for better test accuracy:\n\n| Event | Format | Purpose |\n|-------|--------|---------|\n| `speech-update` | `{"type":"speech-update","status":"started"\\|"stopped"}` | Enables platform-assisted turn detection (more accurate than VAD alone) |\n| `tool_call` | `{"type":"tool_call","name":"...","arguments":{...},"result":...,"successful":bool,"duration_ms":number}` | Reports tool calls for observability |\n| `vent:timing` | `{"type":"vent:timing","stt_ms":number,"llm_ms":number,"tts_ms":number}` | Reports component latency breakdown per turn |\n| `vent:session` | `{"type":"vent:session","platform":"custom","provider_call_id":"...","provider_session_id":"..."}` | Reports stable provider/session identifiers |\n| `vent:call-metadata` | `{"type":"vent:call-metadata","call_metadata":{...}}` | Reports post-call metadata such as cost, recordings, variables, and provider-specific artifacts |\n| `vent:transcript` | `{"type":"vent:transcript","role":"caller"\\|"agent","text":"...","turn_index":0}` | Reports platform/native transcript text for caller or agent |\n| `vent:transfer` | `{"type":"vent:transfer","destination":"...","status":"attempted"\\|"completed"}` | Reports transfer attempts and outcomes |\n| `vent:debug-url` | `{"type":"vent:debug-url","label":"log","url":"https://..."}` | Reports provider debug/deep-link URLs |\n| `vent:warning` | `{"type":"vent:warning","message":"...","code":"..."}` | Reports provider/runtime warnings worth preserving in run metadata |\n\nVent sends `{"type":"end-call"}` to the agent when the test is done.\n\nAll text frames are optional \u2014 audio-only agents work fine with VAD-based turn detection.\n\n## Full Config Schema\n\n- ALL calls MUST reference the agent\'s real context (system prompt, tools, knowledge base) from the codebase.\n\n<vent_run>\n{\n  "connection": { ... },\n  "calls": {\n    "happy-path": { ... },\n    "edge-case": { ... }\n  }\n}\n</vent_run>\n\nOne suite file per platform/adapter. `connection` is declared once, `calls` is a named map of call specs. Each key becomes the call name. Run one call at a time with `--call <name>`.\n\n<config_connection>\n{\n  "connection": {\n    "adapter": "required -- websocket | livekit | vapi | retell | elevenlabs | bland",\n    "start_command": "shell command to start agent (relay only, required for local)",\n    "health_endpoint": "health check path after start_command (default: /health, relay only, required for local)",\n    "agent_url": "hosted custom agent URL (wss:// or https://). Use for BYO hosted agents.",\n    "agent_port": "local agent port (default: 3001, required for local)",\n    "platform": "optional authoring convenience for platform-direct adapters only. The CLI resolves this locally, creates/updates a saved platform connection, and strips raw provider secrets before submit. Do not use for websocket start_command or agent_url runs."\n  }\n}\n\n<credential_resolution>\nIMPORTANT: How to handle platform credentials (API keys, secrets, agent IDs):\n\nThere are two product modes:\n- `BYO agent runtime`: your agent owns its own provider credentials. This covers both `start_command` (local) and `agent_url` (hosted custom endpoint).\n- `Platform-direct runtime`: Vent talks to `vapi`, `retell`, `elevenlabs`, `bland`, or `livekit` directly. This is the only mode that uses saved platform connections.\n\n1. For `start_command` and `agent_url` runs, do NOT put Deepgram / ElevenLabs / OpenAI / other provider keys into Vent config unless the Vent adapter itself needs them. Those credentials belong to the user\'s local or hosted agent runtime.\n2. For platform-direct adapters (`vapi`, `retell`, `elevenlabs`, `bland`, `livekit`), the CLI auto-resolves credentials from `.env.local`, `.env`, and the current shell env. If those env vars already exist, you can omit credential fields from the config JSON entirely.\n3. If you include credential fields in the config, put the ACTUAL VALUE, NOT the env var name. WRONG: `"vapi_api_key": "VAPI_API_KEY"`. RIGHT: `"vapi_api_key": "sk-abc123..."` or omit the field.\n4. The CLI uses the resolved provider config to create or update a saved platform connection server-side, then submits only `platform_connection_id`. Users should not manually author `platform_connection_id`.\n5. To check whether credentials are already available, inspect `.env.local`, `.env`, and any relevant shell env visible to the CLI process.\n\nAuto-resolved env vars per platform:\n| Platform | Config field | Env var (auto-resolved from `.env.local`, `.env`, or shell env) |\n|----------|-------------|-----------------------------------|\n| Vapi | vapi_api_key | VAPI_API_KEY |\n| Vapi | vapi_assistant_id | VAPI_ASSISTANT_ID |\n| Bland | bland_api_key | BLAND_API_KEY |\n| Bland | bland_pathway_id | BLAND_PATHWAY_ID |\n| LiveKit | livekit_api_key | LIVEKIT_API_KEY |\n| LiveKit | livekit_api_secret | LIVEKIT_API_SECRET |\n| LiveKit | livekit_url | LIVEKIT_URL |\n| Retell | retell_api_key | RETELL_API_KEY |\n| Retell | retell_agent_id | RETELL_AGENT_ID |\n| ElevenLabs | elevenlabs_api_key | ELEVENLABS_API_KEY |\n| ElevenLabs | elevenlabs_agent_id | ELEVENLABS_AGENT_ID |\n\nThe CLI strips raw platform secrets before `/runs/submit`. Platform-direct runs go through a saved `platform_connection_id` automatically. BYO agent runs (`start_command` and `agent_url`) do not.\n</credential_resolution>\n\n<config_adapter_rules>\nWebSocket (local agent via relay):\n{\n  "connection": {\n    "adapter": "websocket",\n    "start_command": "npm run start",\n    "health_endpoint": "/health",\n    "agent_port": 3001\n  }\n}\n\nWebSocket (hosted custom agent):\n{\n  "connection": {\n    "adapter": "websocket",\n    "agent_url": "https://my-agent.fly.dev"\n  }\n}\n\nRetell:\n{\n  "connection": {\n    "adapter": "retell",\n    "platform": { "provider": "retell" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: RETELL_API_KEY, RETELL_AGENT_ID. Only add retell_api_key/retell_agent_id to the JSON if those env vars are not already available.\n\nBland:\n{\n  "connection": {\n    "adapter": "bland",\n    "platform": { "provider": "bland" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: BLAND_API_KEY, BLAND_PATHWAY_ID. Only add bland_api_key/bland_pathway_id to the JSON if those env vars are not already available.\nNote: All agent config (voice, model, tools, etc.) is set on the pathway itself, not in Vent config.\n\nVapi:\n{\n  "connection": {\n    "adapter": "vapi",\n    "platform": { "provider": "vapi" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: VAPI_API_KEY, VAPI_ASSISTANT_ID. Only add vapi_api_key/vapi_assistant_id to the JSON if those env vars are not already available.\nmax_concurrency for Vapi: Starter=10, Growth=50, Enterprise=100+. Ask the user which tier they\'re on. If unknown, default to 10.\nAll assistant config (voice, model, transcriber, interruption settings, etc.) is set on the Vapi assistant itself, not in Vent config.\n\nElevenLabs:\n{\n  "connection": {\n    "adapter": "elevenlabs",\n    "platform": { "provider": "elevenlabs" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: ELEVENLABS_API_KEY, ELEVENLABS_AGENT_ID. Only add elevenlabs_api_key/elevenlabs_agent_id to the JSON if those env vars are not already available.\n\nLiveKit:\n{\n  "connection": {\n    "adapter": "livekit",\n    "platform": {\n      "provider": "livekit",\n      "livekit_agent_name": "my-agent",\n      "max_concurrency": 5\n    }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: LIVEKIT_API_KEY, LIVEKIT_API_SECRET, LIVEKIT_URL. Only add these to the JSON if those env vars are not already available.\nlivekit_agent_name is optional -- only needed if the agent registers with an explicit agent_name in WorkerOptions. Omit for automatic dispatch.\nThe livekit adapter requires the LiveKit Agents SDK. It depends on Agents SDK signals (lk.agent.state, lk.transcription) for readiness detection, turn timing, and component latency. Custom LiveKit participants not using the Agents SDK should use the websocket adapter with a relay instead.\nmax_concurrency: Free/Build=5, Ship=20, Scale=50+. Ask the user which tier they\'re on. If unknown, default to 5.\n</config_adapter_rules>\n</config_connection>\n\n\n<call_config>\n<tool_call_capture>\nvapi/retell/elevenlabs/bland: automatic via platform API (no user code needed).\nWebSocket/WebRTC: user\'s agent must emit tool calls:\n  WebSocket \u2014 JSON text frame: {"type":"tool_call","name":"...","arguments":{},"result":{},"successful":true,"duration_ms":150}\n  WebRTC/LiveKit \u2014 publishData() or sendText() on topic "vent:tool-calls". Same JSON.\n</tool_call_capture>\n\n<component_timing>\nPlatform adapters (vapi/retell/elevenlabs/bland/livekit) get STT/LLM/TTS breakdown automatically.\nWebSocket agents can opt in by sending a JSON text frame after each agent turn:\n  {"type":"vent:timing","stt_ms":120,"llm_ms":450,"tts_ms":80}\nAll fields optional. Send one per agent response. Without this, component_latency is omitted from results.\nWhen modifying a WebSocket agent\'s code, add this text frame after TTS completes to enable component latency reporting.\n</component_timing>\n\n<metadata_capture>\nWebSocket and LiveKit/WebRTC agents can also emit richer observability metadata:\n  {"type":"vent:session","platform":"custom","provider_call_id":"call_123","provider_session_id":"session_abc"}\n  {"type":"vent:call-metadata","call_metadata":{"recording_url":"https://...","cost_usd":0.12,"provider_debug_urls":{"log":"https://..."}}}\n  {"type":"vent:debug-url","label":"trace","url":"https://..."}\n  {"type":"vent:session-report","report":{"room_name":"room-123","events":[...],"metrics":[...]}}\n  {"type":"vent:metrics","event":"metrics_collected","metric_type":"eou","metrics":{"speechId":"speech_123","endOfUtteranceDelayMs":420}}\n  {"type":"vent:function-tools-executed","event":"function_tools_executed","hasAgentHandoff":true,"tool_calls":[{"name":"lookup_customer","arguments":{"id":"123"}}]}\n  {"type":"vent:conversation-item","event":"conversation_item_added","item":{"type":"agent_handoff","newAgentId":"billing-agent"}}\n  {"type":"vent:session-usage","usage":{"llm":{"promptTokens":123,"completionTokens":45}}}\nTransport:\n  WebSocket \u2014 send JSON text frames with these payloads. WebSocket agents may also emit {"type":"vent:transcript","role":"caller","text":"I need to reschedule","turn_index":0} when they have native transcript text.\n  WebRTC/LiveKit \u2014 publishData() or sendText() on the matching "vent:*" topic, e.g. topic "vent:call-metadata" with the JSON body above.\nFor LiveKit, transcript and timing stay authoritative from native room signals (`lk.transcription`, `lk.agent.state`). Do not emit `vent:transcript` from LiveKit agents.\nFor LiveKit Node agents, prefer the first-party helper instead of manual forwarding:\n```ts\nimport { instrumentLiveKitAgent } from "@vent-hq/livekit";\n\nconst vent = instrumentLiveKitAgent({\n  ctx,\n  session,\n});\n```\nThis helper must run inside the LiveKit agent runtime with the existing Agents SDK `session` and `ctx` objects. It is the Vent integration layer on top of the Agents SDK, not a replacement for it.\nInstall it with `npm install @vent-hq/livekit` after the package is published to the `vent-hq` npm org. Until then, use the workspace package from this repo.\nThis automatically publishes only the in-agent-only LiveKit signals: `metrics_collected`, `function_tools_executed`, `conversation_item_added`, and a session report on close/shutdown.\nDo not use it to mirror room-visible signals like transcript, agent state timing, or room/session ID \u2014 Vent already gets those from LiveKit itself.\nFor LiveKit inside-agent forwarding, prefer sending the raw LiveKit event payloads on:\n  `vent:metrics`\n  `vent:function-tools-executed`\n  `vent:conversation-item`\n  `vent:session-usage`\nUse these metadata events when the agent runtime already knows native IDs, recordings, warnings, debug links, session reports, metrics events, or handoff artifacts. This gives custom and LiveKit agents parity with hosted adapters without needing a LiveKit Cloud connector.\n</metadata_capture>\n\n<config_call>\nEach call in the `calls` map. The key is the call name (e.g. `"reschedule-appointment"`, not `"call-1"`).\n{\n      "caller_prompt": "required \u2014 caller persona and behavior (name -> goal -> emotion -> conditional behavior)",\n    "max_turns": "required \u2014 default 6",\n    "silence_threshold_ms": "optional \u2014 end-of-turn threshold ms (default 800, 200-10000). 800-1200 FAQ, 2000-3000 tool calls, 3000-5000 complex reasoning.",\n    "persona": "optional \u2014 caller behavior controls",\n    {\n      "pace": "slow | normal | fast",\n      "clarity": "clear | vague | rambling",\n      "disfluencies": "true | false",\n      "cooperation": "cooperative | reluctant | hostile",\n      "emotion": "neutral | cheerful | confused | frustrated | skeptical | rushed",\n      "interruption_style": "optional preplanned interrupt tendency: low | high. If set, Vent may pre-plan a caller cut-in before the agent turn starts. It does NOT make a mid-turn interrupt LLM call.",\n      "memory": "reliable | unreliable",\n      "intent_clarity": "clear | indirect | vague",\n      "confirmation_style": "explicit | vague"\n    },\n    "audio_actions": "optional \u2014 per-turn audio stress calls",\n    [\n      { "action": "interrupt", "at_turn": "N", "prompt": "what caller says" },\n      { "action": "inject_noise", "at_turn": "N", "noise_type": "babble | white | pink", "snr_db": "0-40" },\n      { "action": "split_sentence", "at_turn": "N", "split": { "part_a": "...", "part_b": "...", "pause_ms": "500-5000" } },\n      { "action": "noise_on_caller", "at_turn": "N" }\n    ],\n    "prosody": "optional \u2014 Hume emotion analysis (default false)",\n    "caller_audio": "optional \u2014 omit for clean audio",\n    {\n      "noise": { "type": "babble | white | pink", "snr_db": "0-40" },\n      "speed": "0.5-2.0 (1.0 = normal)",\n      "speakerphone": "true | false",\n      "mic_distance": "close | normal | far",\n      "clarity": "0.0-1.0 (1.0 = perfect)",\n      "accent": "american | british | australian | filipino | spanish_mexican | spanish_peninsular | spanish_colombian | spanish_argentine | german | french | italian | dutch | japanese",\n      "packet_loss": "0.0-0.3",\n      "jitter_ms": "0-100"\n    },\n    "language": "optional \u2014 ISO 639-1: en, es, fr, de, it, nl, ja"\n}\n\nInterruption rules:\n- `audio_actions: [{ "action": "interrupt", ... }]` is the deterministic per-turn interrupt test. Prefer this for evaluation.\n- `persona.interruption_style` is only a preplanned caller tendency. If used, Vent decides before the agent response starts whether this turn may cut in.\n- Vent no longer pauses mid-turn to ask a second LLM whether to interrupt.\n- For production-faithful testing, prefer explicit `audio_actions.interrupt` over persona interruption.\n\n<examples_call>\n<simple_suite_example>\n{\n  "connection": {\n    "adapter": "vapi",\n    "platform": { "provider": "vapi" }\n  },\n  "calls": {\n    "reschedule-appointment": {\n      "caller_prompt": "You are Maria, calling to reschedule her dentist appointment from Thursday to next Tuesday. She\'s in a hurry and wants this done quickly.",\n      "max_turns": 8\n    },\n    "cancel-appointment": {\n      "caller_prompt": "You are Tom, calling to cancel his appointment for Friday. He\'s calm and just wants confirmation.",\n      "max_turns": 6\n    }\n  }\n}\n</simple_suite_example>\n\n<advanced_call_example>\nA call entry with advanced options (persona, audio actions, prosody):\n{\n  "noisy-interruption-booking": {\n    "caller_prompt": "You are James, an impatient customer calling from a loud coffee shop to book a plumber for tomorrow morning. You interrupt the agent mid-sentence when they start listing availability \u2014 you just want the earliest slot.",\n    "max_turns": 12,\n    "persona": { "pace": "fast", "cooperation": "reluctant", "emotion": "rushed", "interruption_style": "high" },\n    "audio_actions": [\n      { "action": "interrupt", "at_turn": 3, "prompt": "Just give me the earliest one!" },\n      { "action": "inject_noise", "at_turn": 1, "noise_type": "babble", "snr_db": 15 }\n    ],\n    "caller_audio": { "noise": { "type": "babble", "snr_db": 20 }, "speed": 1.3 },\n    "prosody": true\n  }\n}\n</advanced_call_example>\n\n</examples_call>\n</config_call>\n\n<output_conversation_test>\n{\n  "name": "sarah-hotel-booking",\n  "status": "completed",\n  "caller_prompt": "You are Sarah, calling to book...",\n  "duration_ms": 45200,\n  "error": null,\n  "transcript": [\n    { "role": "caller", "text": "Hi, I\'d like to book..." },\n    { "role": "agent", "text": "Sure! What date?", "ttfb_ms": 650, "ttfw_ms": 780, "audio_duration_ms": 2400 },\n    { "role": "agent", "text": "Let me check avail\u2014", "ttfb_ms": 540, "ttfw_ms": 620, "audio_duration_ms": 1400, "interrupted": true },\n    { "role": "caller", "text": "Just the earliest slot please", "audio_duration_ms": 900, "is_interruption": true },\n    { "role": "agent", "text": "Sure, the earliest is 9 AM tomorrow.", "ttfb_ms": 220, "ttfw_ms": 260, "audio_duration_ms": 2100 }\n  ],\n  "latency": {\n    "response_time_ms": 890, "response_time_source": "ttfw",\n    "p50_response_time_ms": 850, "p90_response_time_ms": 1100, "p95_response_time_ms": 1400, "p99_response_time_ms": 1550,\n    "first_response_time_ms": 1950,\n    "mean_ttfw_ms": 890, "p50_ttfw_ms": 850, "p95_ttfw_ms": 1400, "p99_ttfw_ms": 1550,\n    "first_turn_ttfw_ms": 1950, "total_silence_ms": 4200, "mean_turn_gap_ms": 380,\n    "drift_slope_ms_per_turn": -45.2, "mean_silence_pad_ms": 128, "mouth_to_ear_est_ms": 1020\n  },\n  "transcript_quality": {\n    "wer": 0.04,\n    "hallucination_events": [\n      { "error_count": 5, "reference_text": "triple five one two", "hypothesis_text": "five five five nine two" }\n    ],\n    "repetition_score": 0.05,\n    "reprompt_count": 0,\n    "filler_word_rate": 0.8,\n    "words_per_minute": 148\n  },\n  "audio_analysis": {\n    "caller_talk_time_ms": 12400,\n    "agent_talk_time_ms": 28500,\n    "agent_speech_ratio": 0.72,\n    "talk_ratio_vad": 0.69,\n    "interruption_rate": 0.25,\n    "interruption_count": 1,\n    "agent_overtalk_after_barge_in_ms": 280,\n    "agent_interrupting_user_rate": 0.0,\n    "agent_interrupting_user_count": 0,\n    "missed_response_windows": 0,\n    "longest_monologue_ms": 5800,\n    "silence_gaps_over_2s": 1,\n    "total_internal_silence_ms": 2400,\n    "mean_agent_speech_segment_ms": 3450\n  },\n  "tool_calls": {\n    "total": 2, "successful": 2, "failed": 0, "mean_latency_ms": 340,\n    "names": ["check_availability", "book_appointment"],\n    "observed": [{ "name": "check_availability", "arguments": { "date": "2026-03-12" }, "result": { "slots": ["09:00", "10:00"] }, "successful": true, "latency_ms": 280, "turn_index": 3 }]\n  },\n  "component_latency": {\n    "mean_stt_ms": 120, "mean_llm_ms": 450, "mean_tts_ms": 80,\n    "p95_stt_ms": 180, "p95_llm_ms": 620, "p95_tts_ms": 110,\n    "mean_speech_duration_ms": 2100,\n    "bottleneck": "llm"\n  },\n  "call_metadata": {\n    "platform": "vapi",\n    "cost_usd": 0.08,\n    "recording_url": "https://example.com/recording",\n    "ended_reason": "customer_ended_call",\n    "transfers": []\n  },\n  "warnings": [],\n  "audio_actions": [],\n  "emotion": {\n    "naturalness": 0.72, "mean_calmness": 0.65, "mean_confidence": 0.58, "peak_frustration": 0.08, "emotion_trajectory": "stable"\n  }\n}\n\nAlways present: name, status, caller_prompt, duration_ms, error, transcript, tool_calls, warnings, audio_actions. Nullable when analysis didn\'t run: latency, transcript_quality, audio_analysis, component_latency, call_metadata, emotion (requires prosody: true), debug (requires --verbose).\n\n### Result presentation\n\nWhen you report a conversation result to the user, always include:\n\n1. **Summary** \u2014 the overall verdict and the 1-3 most important findings.\n2. **Transcript summary** \u2014 a short narrative of what happened in the call.\n3. **Recording URL** \u2014 include `call_metadata.recording_url` when present; explicitly say when it is unavailable.\n4. **Next steps** \u2014 concrete fixes, follow-up tests, or why no change is needed.\n\nUse metrics to support the summary, not as the whole answer. Do not dump raw numbers without interpretation.\n\nWhen `call_metadata.transfer_attempted` is present, explicitly say whether the transfer only appeared attempted or was mechanically verified as completed (`call_metadata.transfer_completed`). Use `call_metadata.transfers[]` to report transfer type, destination, status, and sources.\n\n### Judging guidance\n\nUse the transcript, metrics, test scenario, and relevant agent instructions/system prompt to judge:\n\n| Dimension | What to check |\n|--------|----------------|\n| **Hallucination detection** | Check whether the agent stated anything not grounded in its instructions, tools, or the conversation itself. Treat `transcript_quality.hallucination_events` only as a speech-recognition warning signal, not proof of agent hallucination. |\n| **Instruction following** | Compare the agent\'s behavior against its system prompt and the test\'s expected constraints. |\n| **Context retention** | Check whether the agent forgot or contradicted information established earlier in the call. |\n| **Semantic accuracy** | Check whether the agent correctly understood the caller\'s intent and responded to the real request. |\n| **Goal completion** | Decide whether the agent achieved what the test scenario was designed to verify. |\n| **Transfer correctness** | For transfer scenarios, judge whether transfer was appropriate, whether it completed, whether it went to the expected destination, and whether enough context was passed during the handoff. |\n\n### Interruption evaluation\n\nWhen the transcript contains `interrupted: true` / `is_interruption: true` turns, evaluate these metrics by reading the transcript:\n\n| Metric | How to evaluate | Target |\n|--------|----------------|--------|\n| **Recovery rate** | For each interrupted turn: does the post-interrupt agent response acknowledge or address the interruption? | >90% |\n| **Context retention** | After the interruption, does the agent remember pre-interrupt conversation state? | >95% |\n| **Agent overtalk after barge-in** | Use `audio_analysis.agent_overtalk_after_barge_in_ms` when available. Lower is better because it measures how long the agent kept speaking after the caller cut in. | <500ms acceptable |\n| **Agent interrupting user rate** | Use `audio_analysis.agent_interrupting_user_rate` and the transcript to see whether the agent starts speaking before the caller finished. | 0 ideal |\n\nReport these alongside standard metrics when interruption calls run.\n</output_conversation_test>\n</call_config>\n\n\n## Exit Codes\n\n0=pass, 1=fail, 2=error\n\n## Vent Access Token\n\nSet `VENT_ACCESS_TOKEN` env var or run `npx vent-hq login`.\n';
 // src/skills/codex.md
-var codex_default = '# Vent \u2014 Voice Agent Calls\n\nCall voice agents from the terminal. Calls run in the cloud (30-120s).\n\n## When to Call\n\n- After modifying voice agent code (system prompt, tools, handlers): ALWAYS run calls\n- After changing audio config: run calls\n- Before marking a task complete that touches agent behavior: run calls\n\n## Setup (if no Vent access token)\n\nIf `~/.vent/credentials` does not exist and `VENT_ACCESS_TOKEN` is not set:\n\n```bash\nnpx vent-hq init\n```\n\nOne command. No email, no browser, no human interaction needed. You get 10 free runs instantly.\n\n## Commands\n\n| Command | Purpose |\n|---------|---------|\n| `npx vent-hq init` | First-time setup (creates account + installs skills) |\n| `npx vent-hq agent start -f .vent/suite.<adapter>.json` | Start one shared local agent session (required for `start_command`) |\n| `npx vent-hq agent stop <session-id>` | Close a shared local agent session |\n| `npx vent-hq run -f .vent/suite.<adapter>.json` | Run a call from suite file (auto-selects if only one call) |\n| `npx vent-hq run -f .vent/suite.<adapter>.json --call <name>` | Run a specific named call |\n| `npx vent-hq stop <run-id>` | Cancel a queued or running call |\n| `npx vent-hq status <run-id>` | Get full results for a completed run |\n\n\n## Workflow\n\n1. Read the voice agent\'s codebase \u2014 understand its system prompt, tools, intents, and domain.\n2. Read the config schema below for all available fields.\n3. Create the suite file in `.vent/` using the naming convention: `.vent/suite.<adapter>.json` (e.g., `.vent/suite.vapi.json`, `.vent/suite.websocket.json`, `.vent/suite.retell.json`). This prevents confusion when multiple adapters are tested in the same project.\n4. Run calls:\n   ```\n   # suite with one call (auto-selects)\n   npx vent-hq run -f .vent/suite.<adapter>.json\n\n   # suite with multiple calls \u2014 pick one by name\n   npx vent-hq run -f .vent/suite.<adapter>.json --call happy-path\n\n   # local start_command \u2014 first start relay, then add --session\n   npx vent-hq agent start -f .vent/suite.<adapter>.json\n   npx vent-hq run -f .vent/suite.<adapter>.json --call happy-path --session <session-id>\n   ```\n5. To run multiple calls, run each as a separate command.\n6. After results return, **compare with previous run** \u2014 Vent saves full result JSON to `.vent/runs/` after every run. Compare status flips, TTFW p50/p95 changes >20%, tool call count drops, cost increases >30%. Skip if no previous run exists.\n7. After code changes, re-run the same way.\n\n### Multiple suite files\n\nIf `.vent/` contains more than one suite file, **always check which adapter each suite uses before running**. Read the `connection.adapter` field in each file. Never run a suite intended for a different adapter \u2014 results will be meaningless or fail. When reporting results, always state which suite file produced them (e.g., "Results from `.vent/suite.vapi.json`:").\n\n## Critical Rules\n\n1. **Run calls in parallel with 5min timeout** \u2014 Each call is a separate shell command, run them all at once. Set a 300-second (5 min) timeout on each \u2014 calls can take up to 5 minutes.\n2. **Handle backgrounded commands** \u2014 If a call command gets moved to background by the system, wait for it to complete before proceeding. Never end your response without delivering call results.\n3. **Output format** \u2014 In non-TTY mode (when run by an agent), every SSE event is written to stdout as a JSON line. Results are always in stdout.\n4. **This skill is self-contained** \u2014 The full config schema is below.\n\n## WebSocket Protocol (BYO agents)\n\nWhen using `adapter: "websocket"`, Vent communicates with the agent over a single WebSocket connection:\n\n- **Binary frames** \u2192 PCM audio (16-bit mono, configurable sample rate)\n- **Text frames** \u2192 optional JSON events the agent can send for better test accuracy:\n\n| Event | Format | Purpose |\n|-------|--------|---------|\n| `speech-update` | `{"type":"speech-update","status":"started"\\|"stopped"}` | Enables platform-assisted turn detection (more accurate than VAD alone) |\n| `tool_call` | `{"type":"tool_call","name":"...","arguments":{...},"result":...,"successful":bool,"duration_ms":number}` | Reports tool calls for observability |\n| `vent:timing` | `{"type":"vent:timing","stt_ms":number,"llm_ms":number,"tts_ms":number}` | Reports component latency breakdown per turn |\n\nVent sends `{"type":"end-call"}` to the agent when the test is done.\n\nAll text frames are optional \u2014 audio-only agents work fine with VAD-based turn detection.\n\n## Full Config Schema\n\n- ALL calls MUST reference the agent\'s real context (system prompt, tools, knowledge base) from the codebase.\n\n<vent_run>\n{\n  "connection": { ... },\n  "calls": {\n    "happy-path": { ... },\n    "edge-case": { ... }\n  }\n}\n</vent_run>\n\nOne suite file per platform/adapter. `connection` is declared once, `calls` is a named map of call specs. Each key becomes the call name. Run one call at a time with `--call <name>`.\n\n<config_connection>\n{\n  "connection": {\n    "adapter": "required -- websocket | livekit | vapi | retell | elevenlabs | bland",\n    "start_command": "shell command to start agent (relay only, required for local)",\n    "health_endpoint": "health check path after start_command (default: /health, relay only, required for local)",\n    "agent_url": "hosted custom agent URL (wss:// or https://). Use for BYO hosted agents.",\n    "agent_port": "local agent port (default: 3001, required for local)",\n    "platform": "optional authoring convenience for platform-direct adapters only. The CLI resolves this locally, creates/updates a saved platform connection, and strips raw provider secrets before submit. Do not use for websocket start_command or agent_url runs."\n  }\n}\n\n<credential_resolution>\nIMPORTANT: How to handle platform credentials (API keys, secrets, agent IDs):\n\nThere are two product modes:\n- `BYO agent runtime`: your agent owns its own provider credentials. This covers both `start_command` (local) and `agent_url` (hosted custom endpoint).\n- `Platform-direct runtime`: Vent talks to `vapi`, `retell`, `elevenlabs`, `bland`, or `livekit` directly. This is the only mode that uses saved platform connections.\n\n1. For `start_command` and `agent_url` runs, do NOT put Deepgram / ElevenLabs / OpenAI / other provider keys into Vent config unless the Vent adapter itself needs them. Those credentials belong to the user\'s local or hosted agent runtime.\n2. For platform-direct adapters (`vapi`, `retell`, `elevenlabs`, `bland`, `livekit`), the CLI auto-resolves credentials from `.env.local`, `.env`, and the current shell env. If those env vars already exist, you can omit credential fields from the config JSON entirely.\n3. If you include credential fields in the config, put the ACTUAL VALUE, NOT the env var name. WRONG: `"vapi_api_key": "VAPI_API_KEY"`. RIGHT: `"vapi_api_key": "sk-abc123..."` or omit the field.\n4. The CLI uses the resolved provider config to create or update a saved platform connection server-side, then submits only `platform_connection_id`. Users should not manually author `platform_connection_id`.\n5. To check whether credentials are already available, inspect `.env.local`, `.env`, and any relevant shell env visible to the CLI process.\n\nAuto-resolved env vars per platform:\n| Platform | Config field | Env var (auto-resolved from `.env.local`, `.env`, or shell env) |\n|----------|-------------|-----------------------------------|\n| Vapi | vapi_api_key | VAPI_API_KEY |\n| Vapi | vapi_assistant_id | VAPI_ASSISTANT_ID |\n| Bland | bland_api_key | BLAND_API_KEY |\n| Bland | bland_pathway_id | BLAND_PATHWAY_ID |\n| LiveKit | livekit_api_key | LIVEKIT_API_KEY |\n| LiveKit | livekit_api_secret | LIVEKIT_API_SECRET |\n| LiveKit | livekit_url | LIVEKIT_URL |\n| Retell | retell_api_key | RETELL_API_KEY |\n| Retell | retell_agent_id | RETELL_AGENT_ID |\n| ElevenLabs | elevenlabs_api_key | ELEVENLABS_API_KEY |\n| ElevenLabs | elevenlabs_agent_id | ELEVENLABS_AGENT_ID |\n\nThe CLI strips raw platform secrets before `/runs/submit`. Platform-direct runs go through a saved `platform_connection_id` automatically. BYO agent runs (`start_command` and `agent_url`) do not.\n</credential_resolution>\n\n<config_adapter_rules>\nWebSocket (local agent via relay):\n{\n  "connection": {\n    "adapter": "websocket",\n    "start_command": "npm run start",\n    "health_endpoint": "/health",\n    "agent_port": 3001\n  }\n}\n\nWebSocket (hosted custom agent):\n{\n  "connection": {\n    "adapter": "websocket",\n    "agent_url": "https://my-agent.fly.dev"\n  }\n}\n\nRetell:\n{\n  "connection": {\n    "adapter": "retell",\n    "platform": { "provider": "retell" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: RETELL_API_KEY, RETELL_AGENT_ID. Only add retell_api_key/retell_agent_id to the JSON if those env vars are not already available.\n\nBland:\n{\n  "connection": {\n    "adapter": "bland",\n    "platform": { "provider": "bland" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: BLAND_API_KEY, BLAND_PATHWAY_ID. Only add bland_api_key/bland_pathway_id to the JSON if those env vars are not already available.\nNote: All agent config (voice, model, tools, etc.) is set on the pathway itself, not in Vent config.\n\nVapi:\n{\n  "connection": {\n    "adapter": "vapi",\n    "platform": { "provider": "vapi" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: VAPI_API_KEY, VAPI_ASSISTANT_ID. Only add vapi_api_key/vapi_assistant_id to the JSON if those env vars are not already available.\nmax_concurrency for Vapi: Starter=10, Growth=50, Enterprise=100+. Ask the user which tier they\'re on. If unknown, default to 10.\nAll assistant config (voice, model, transcriber, interruption settings, etc.) is set on the Vapi assistant itself, not in Vent config.\n\nElevenLabs:\n{\n  "connection": {\n    "adapter": "elevenlabs",\n    "platform": { "provider": "elevenlabs" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: ELEVENLABS_API_KEY, ELEVENLABS_AGENT_ID. Only add elevenlabs_api_key/elevenlabs_agent_id to the JSON if those env vars are not already available.\n\nLiveKit:\n{\n  "connection": {\n    "adapter": "livekit",\n    "platform": {\n      "provider": "livekit",\n      "livekit_agent_name": "my-agent",\n      "max_concurrency": 5\n    }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: LIVEKIT_API_KEY, LIVEKIT_API_SECRET, LIVEKIT_URL. Only add these to the JSON if those env vars are not already available.\nlivekit_agent_name is optional -- only needed if the agent registers with an explicit agent_name in WorkerOptions. Omit for automatic dispatch.\nmax_concurrency: Free/Build=5, Ship=20, Scale=50+. Ask the user which tier they\'re on. If unknown, default to 5.\n</config_adapter_rules>\n</config_connection>\n\n\n<call_config>\n<tool_call_capture>\nvapi/retell/elevenlabs/bland: automatic via platform API (no user code needed).\nWebSocket/WebRTC: user\'s agent must emit tool calls:\n  WebSocket \u2014 JSON text frame: {"type":"tool_call","name":"...","arguments":{},"result":{},"successful":true,"duration_ms":150}\n  WebRTC/LiveKit \u2014 publishData() or sendText() on topic "vent:tool-calls". Same JSON.\n</tool_call_capture>\n\n<component_timing>\nPlatform adapters (vapi/retell/elevenlabs/bland/livekit) get STT/LLM/TTS breakdown automatically.\nWebSocket agents can opt in by sending a JSON text frame after each agent turn:\n  {"type":"vent:timing","stt_ms":120,"llm_ms":450,"tts_ms":80}\nAll fields optional. Send one per agent response. Without this, component_latency is omitted from results.\nWhen modifying a WebSocket agent\'s code, add this text frame after TTS completes to enable component latency reporting.\n</component_timing>\n\n<config_call>\nEach call in the `calls` map. The key is the call name (e.g. `"reschedule-appointment"`, not `"call-1"`).\n{\n      "caller_prompt": "required \u2014 caller persona and behavior (name -> goal -> emotion -> conditional behavior)",\n      "max_turns": "required \u2014 default 6",\n      "silence_threshold_ms": "optional \u2014 end-of-turn threshold ms (default 800, 200-10000). 800-1200 FAQ, 2000-3000 tool calls, 3000-5000 complex reasoning.",\n      "persona": "optional \u2014 caller behavior controls",\n      {\n        "pace": "slow | normal | fast",\n        "clarity": "clear | vague | rambling",\n        "disfluencies": "true | false",\n        "cooperation": "cooperative | reluctant | hostile",\n        "emotion": "neutral | cheerful | confused | frustrated | skeptical | rushed",\n        "interruption_style": "low (~3/10 turns) | high (~7/10 turns)",\n        "memory": "reliable | unreliable",\n        "intent_clarity": "clear | indirect | vague",\n        "confirmation_style": "explicit | vague"\n      },\n      "audio_actions": "optional \u2014 per-turn audio stress calls",\n      [\n        { "action": "interrupt", "at_turn": "N", "prompt": "what caller says" },\n        { "action": "silence", "at_turn": "N", "duration_ms": "1000-30000" },\n        { "action": "inject_noise", "at_turn": "N", "noise_type": "babble | white | pink", "snr_db": "0-40" },\n        { "action": "split_sentence", "at_turn": "N", "split": { "part_a": "...", "part_b": "...", "pause_ms": "500-5000" } },\n        { "action": "noise_on_caller", "at_turn": "N" }\n      ],\n      "prosody": "optional \u2014 Hume emotion analysis (default false)",\n      "caller_audio": "optional \u2014 omit for clean audio",\n      {\n        "noise": { "type": "babble | white | pink", "snr_db": "0-40" },\n        "speed": "0.5-2.0 (1.0 = normal)",\n        "speakerphone": "true | false",\n        "mic_distance": "close | normal | far",\n        "clarity": "0.0-1.0 (1.0 = perfect)",\n        "accent": "american | british | australian | filipino | spanish_mexican | spanish_peninsular | spanish_colombian | spanish_argentine | german | french | italian | dutch | japanese",\n        "packet_loss": "0.0-0.3",\n        "jitter_ms": "0-100"\n      },\n      "language": "optional \u2014 ISO 639-1: en, es, fr, de, it, nl, ja"\n}\n\n<examples_call>\n<simple_suite_example>\n{\n  "connection": {\n    "adapter": "vapi",\n    "platform": { "provider": "vapi" }\n  },\n  "calls": {\n    "reschedule-appointment": {\n      "caller_prompt": "You are Maria, calling to reschedule her dentist appointment from Thursday to next Tuesday. She\'s in a hurry and wants this done quickly.",\n      "max_turns": 8\n    },\n    "cancel-appointment": {\n      "caller_prompt": "You are Tom, calling to cancel his appointment for Friday. He\'s calm and just wants confirmation.",\n      "max_turns": 6\n    }\n  }\n}\n</simple_suite_example>\n\n<advanced_call_example>\nA call entry with advanced options (persona, audio actions, prosody):\n{\n  "noisy-interruption-booking": {\n    "caller_prompt": "You are James, an impatient customer calling from a loud coffee shop to book a plumber for tomorrow morning. You interrupt the agent mid-sentence when they start listing availability \u2014 you just want the earliest slot.",\n    "max_turns": 12,\n    "persona": { "pace": "fast", "cooperation": "reluctant", "emotion": "rushed", "interruption_style": "high" },\n    "audio_actions": [\n      { "action": "interrupt", "at_turn": 3, "prompt": "Just give me the earliest one!" },\n      { "action": "inject_noise", "at_turn": 1, "noise_type": "babble", "snr_db": 15 }\n    ],\n    "caller_audio": { "noise": { "type": "babble", "snr_db": 20 }, "speed": 1.3 },\n    "prosody": true\n  }\n}\n</advanced_call_example>\n\n</examples_call>\n</config_call>\n\n<output_conversation_test>\n{\n  "name": "sarah-hotel-booking",\n  "status": "completed",\n  "caller_prompt": "You are Sarah, calling to book...",\n  "duration_ms": 45200,\n  "error": null,\n  "transcript": [\n    { "role": "caller", "text": "Hi, I\'d like to book..." },\n    { "role": "agent", "text": "Sure! What date?", "ttfb_ms": 650, "ttfw_ms": 780, "audio_duration_ms": 2400 },\n    { "role": "agent", "text": "Let me check avail\u2014", "ttfb_ms": 540, "ttfw_ms": 620, "audio_duration_ms": 1400, "interrupted": true },\n    { "role": "caller", "text": "Just the earliest slot please", "audio_duration_ms": 900, "is_interruption": true },\n    { "role": "agent", "text": "Sure, the earliest is 9 AM tomorrow.", "ttfb_ms": 220, "ttfw_ms": 260, "audio_duration_ms": 2100 }\n  ],\n  "latency": {\n    "mean_ttfw_ms": 890, "p50_ttfw_ms": 850, "p95_ttfw_ms": 1400, "p99_ttfw_ms": 1550,\n    "first_turn_ttfw_ms": 1950, "total_silence_ms": 4200, "mean_turn_gap_ms": 380,\n    "drift_slope_ms_per_turn": -45.2, "mean_silence_pad_ms": 128, "mouth_to_ear_est_ms": 1020\n  },\n  "transcript_quality": {\n    "wer": 0.04,\n    "hallucination_events": [\n      { "error_count": 5, "reference_text": "triple five one two", "hypothesis_text": "five five five nine two" }\n    ],\n    "repetition_score": 0.05,\n    "reprompt_count": 0,\n    "filler_word_rate": 0.8,\n    "words_per_minute": 148\n  },\n  "audio_analysis": {\n    "agent_speech_ratio": 0.72,\n    "interruption_rate": 0.25,\n    "interruption_count": 1,\n    "barge_in_recovery_time_ms": 280,\n    "agent_interrupting_user_rate": 0.0,\n    "agent_interrupting_user_count": 0,\n    "missed_response_windows": 0,\n    "longest_monologue_ms": 5800,\n    "silence_gaps_over_2s": 1,\n    "total_internal_silence_ms": 2400,\n    "mean_agent_speech_segment_ms": 3450\n  },\n  "tool_calls": {\n    "total": 2, "successful": 2, "failed": 0, "mean_latency_ms": 340,\n    "names": ["check_availability", "book_appointment"],\n    "observed": [{ "name": "check_availability", "arguments": { "date": "2026-03-12" }, "result": { "slots": ["09:00", "10:00"] }, "successful": true, "latency_ms": 280, "turn_index": 3 }]\n  },\n  "call_metadata": {\n    "platform": "vapi",\n    "recording_url": "https://example.com/recording"\n  },\n  "warnings": [],\n  "audio_actions": [\n    { "at_turn": 5, "action": "silence", "metrics": { "agent_prompted": false, "unprompted_utterance_count": 0, "silence_duration_ms": 8000 } }\n  ],\n  "emotion": {\n    "naturalness": 0.72, "mean_calmness": 0.65, "mean_confidence": 0.58, "peak_frustration": 0.08, "emotion_trajectory": "stable"\n  }\n}\n\nAll fields optional except name, status, caller_prompt, duration_ms, transcript. Fields appear only when relevant analysis ran (e.g., emotion requires prosody: true).\n\n### Result presentation\n\nWhen you report a conversation result to the user, always include:\n\n1. **Summary** \u2014 the overall verdict and the 1-3 most important findings.\n2. **Transcript summary** \u2014 a short narrative of what happened in the call.\n3. **Recording URL** \u2014 include `call_metadata.recording_url` when present; explicitly say when it is unavailable.\n4. **Next steps** \u2014 concrete fixes, follow-up tests, or why no change is needed.\n\nUse metrics to support the summary, not as the whole answer. Do not dump raw numbers without interpretation.\n\nWhen `call_metadata.transfer_attempted` is present, explicitly say whether the transfer only appeared attempted or was mechanically verified as completed. If `call_metadata.transfers[*].verification` is present, use it to mention second-leg observation, connect latency, transcript/context summary, and whether context passing was verified.\n\n### Judging guidance\n\nUse the transcript, metrics, test scenario, and relevant agent instructions/system prompt to judge:\n\n| Dimension | What to check |\n|--------|----------------|\n| **Hallucination detection** | Check whether the agent stated anything not grounded in its instructions, tools, or the conversation itself. Treat `transcript_quality.hallucination_events` only as a speech-recognition warning signal, not proof of agent hallucination. |\n| **Instruction following** | Compare the agent\'s behavior against its system prompt and the test\'s expected constraints. |\n| **Context retention** | Check whether the agent forgot or contradicted information established earlier in the call. |\n| **Semantic accuracy** | Check whether the agent correctly understood the caller\'s intent and responded to the real request. |\n| **Goal completion** | Decide whether the agent achieved what the test scenario was designed to verify. |\n| **Transfer correctness** | For transfer scenarios, judge whether transfer was appropriate, whether it completed, whether it went to the expected destination, and whether enough context was passed during the handoff. |\n\n### Interruption evaluation\n\nWhen the transcript contains `interrupted: true` / `is_interruption: true` turns, evaluate these metrics by reading the transcript:\n\n| Metric | How to evaluate | Target |\n|--------|----------------|--------|\n| **Recovery rate** | For each interrupted turn: does the post-interrupt agent response acknowledge or address the interruption? | >90% |\n| **Context retention** | After the interruption, does the agent remember pre-interrupt conversation state? | >95% |\n| **Barge-in recovery time** | Use `audio_analysis.barge_in_recovery_time_ms` when available. Lower is better because it measures how long the agent kept speaking after the caller cut in. | <500ms acceptable |\n| **Agent interrupting user rate** | Use `audio_analysis.agent_interrupting_user_rate` and the transcript to see whether the agent starts speaking before the caller finished. | 0 ideal |\n\nReport these alongside standard metrics when interruption calls run.\n</output_conversation_test>\n</call_config>\n\n\n## Exit Codes\n\n0=pass, 1=fail, 2=error\n';
+var codex_default = '# Vent \u2014 Voice Agent Calls\n\nCall voice agents from the terminal. Calls run in the cloud (30-120s).\n\n## When to Call\n\n- After modifying voice agent code (system prompt, tools, handlers): ALWAYS run calls\n- After changing audio config: run calls\n- Before marking a task complete that touches agent behavior: run calls\n\n## Setup (if no Vent access token)\n\nIf `~/.vent/credentials` does not exist and `VENT_ACCESS_TOKEN` is not set:\n\n```bash\nnpx vent-hq init\n```\n\nOne command. No email, no browser, no human interaction needed. You get 10 free runs instantly.\n\n## Commands\n\n| Command | Purpose |\n|---------|---------|\n| `npx vent-hq init` | First-time setup (creates account + installs skills) |\n| `npx vent-hq agent start -f .vent/suite.<adapter>.json` | Start one shared local agent session (required for `start_command`) |\n| `npx vent-hq agent stop <session-id>` | Close a shared local agent session |\n| `npx vent-hq run -f .vent/suite.<adapter>.json` | Run a call from suite file (auto-selects if only one call) |\n| `npx vent-hq run -f .vent/suite.<adapter>.json --verbose` | Include debug fields in the result JSON |\n| `npx vent-hq run -f .vent/suite.<adapter>.json --call <name>` | Run a specific named call |\n| `npx vent-hq stop <run-id>` | Cancel a queued or running call |\n| `npx vent-hq status <run-id>` | Get full results for a completed run |\n| `npx vent-hq status <run-id> --verbose` | Re-print a run with debug fields included |\n\n## When To Use `--verbose`\n\nDefault output is enough for most iterations. It already includes:\n- transcript\n- latency\n- transcript quality (`wer` / `cer`)\n- audio analysis\n- tool calls\n- summary cost / recording / transfers\n\nUse `--verbose` only when you need debugging detail that is not in the default result:\n- per-turn debug fields: timestamps, caller decision mode, silence pad, STT confidence, platform transcript\n- raw signal analysis: `debug.signal_quality`\n- harness timings: `debug.harness_overhead`\n- raw prosody payload and warnings\n- raw provider warnings\n- per-turn component latency arrays\n- raw observed tool-call timeline\n- provider-specific metadata in `debug.provider_metadata`\n\nTrigger `--verbose` when:\n- transcript accuracy looks wrong and you need to inspect `platform_transcript`\n- latency is bad and you need per-turn/component breakdowns\n- interruptions/barge-in behavior looks wrong\n- tool-call execution looks inconsistent or missing\n- the provider returned warnings/errors or you need provider-native artifacts\n\nSkip `--verbose` when:\n- you only need pass/fail, transcript, latency, tool calls, recording, or summary\n- you are doing quick iteration on prompt wording and the normal result already explains the failure\n\n## Normalization Contract\n\nVent always returns one normalized result shape on `stdout` across adapters. Treat these as the stable categories:\n- `transcript`\n- `latency`\n- `transcript_quality`\n- `audio_analysis`\n- `tool_calls`\n- `component_latency`\n- `call_metadata`\n- `warnings`\n- `audio_actions`\n- `emotion`\n\nSource-of-truth policy:\n- Vent computes transcript, latency, and audio-quality metrics itself.\n- Hosted adapters choose the best source per category, usually provider post-call data for tool calls, call metadata, transfers, provider transcripts, and recordings.\n- Realtime provider events are fallback or enrichment only when post-call data is missing, delayed, weaker for that category, or provider-specific.\n- `LiveKit` helper events are the provider-native path for rich in-agent observability.\n- `websocket`/custom agents are realtime-native but still map into the same normalized categories.\n- Keep adapter-specific details in `call_metadata.provider_metadata` or `debug.provider_metadata`, not in new top-level fields.\n\n## Workflow\n\n1. Read the voice agent\'s codebase \u2014 understand its system prompt, tools, intents, and domain.\n2. Read the config schema below for all available fields.\n3. Create the suite file in `.vent/` using the naming convention: `.vent/suite.<adapter>.json` (e.g., `.vent/suite.vapi.json`, `.vent/suite.websocket.json`, `.vent/suite.retell.json`). This prevents confusion when multiple adapters are tested in the same project.\n4. Run calls:\n   ```\n   # suite with one call (auto-selects)\n   npx vent-hq run -f .vent/suite.<adapter>.json\n\n   # suite with multiple calls \u2014 pick one by name\n   npx vent-hq run -f .vent/suite.<adapter>.json --call happy-path\n\n   # local start_command \u2014 first start relay, then add --session\n   npx vent-hq agent start -f .vent/suite.<adapter>.json\n   npx vent-hq run -f .vent/suite.<adapter>.json --call happy-path --session <session-id>\n   ```\n5. To run multiple calls, run each as a separate command.\n6. After results return, **compare with previous run** \u2014 Vent saves full result JSON to `.vent/runs/` after every run. Use `--verbose` only when the default result is not enough to explain the failure. Compare status flips, TTFW p50/p95 changes >20%, tool call count drops, cost increases >30%. Skip if no previous run exists.\n7. After code changes, re-run the same way.\n\n### Multiple suite files\n\nIf `.vent/` contains more than one suite file, **always check which adapter each suite uses before running**. Read the `connection.adapter` field in each file. Never run a suite intended for a different adapter \u2014 results will be meaningless or fail. When reporting results, always state which suite file produced them (e.g., "Results from `.vent/suite.vapi.json`:").\n\n## Critical Rules\n\n1. **Run calls in parallel with 5min timeout** \u2014 Each call is a separate shell command, run them all at once. Set a 300-second (5 min) timeout on each \u2014 calls can take up to 5 minutes.\n2. **Handle backgrounded commands** \u2014 If a call command gets moved to background by the system, wait for it to complete before proceeding. Never end your response without delivering call results.\n3. **Output format** \u2014 In non-TTY mode (when run by an agent), every SSE event is written to stdout as a JSON line. Results are always in stdout.\n4. **This skill is self-contained** \u2014 The full config schema is below.\n\n## WebSocket Protocol (BYO agents)\n\nWhen using `adapter: "websocket"`, Vent communicates with the agent over a single WebSocket connection:\n\n- **Binary frames** \u2192 PCM audio (16-bit mono, configurable sample rate)\n- **Text frames** \u2192 optional JSON events the agent can send for better test accuracy:\n\n| Event | Format | Purpose |\n|-------|--------|---------|\n| `speech-update` | `{"type":"speech-update","status":"started"\\|"stopped"}` | Enables platform-assisted turn detection (more accurate than VAD alone) |\n| `tool_call` | `{"type":"tool_call","name":"...","arguments":{...},"result":...,"successful":bool,"duration_ms":number}` | Reports tool calls for observability |\n| `vent:timing` | `{"type":"vent:timing","stt_ms":number,"llm_ms":number,"tts_ms":number}` | Reports component latency breakdown per turn |\n| `vent:session` | `{"type":"vent:session","platform":"custom","provider_call_id":"...","provider_session_id":"..."}` | Reports stable provider/session identifiers |\n| `vent:call-metadata` | `{"type":"vent:call-metadata","call_metadata":{...}}` | Reports post-call metadata such as cost, recordings, variables, and provider-specific artifacts |\n| `vent:transcript` | `{"type":"vent:transcript","role":"caller"\\|"agent","text":"...","turn_index":0}` | Reports platform/native transcript text for caller or agent |\n| `vent:transfer` | `{"type":"vent:transfer","destination":"...","status":"attempted"\\|"completed"}` | Reports transfer attempts and outcomes |\n| `vent:debug-url` | `{"type":"vent:debug-url","label":"log","url":"https://..."}` | Reports provider debug/deep-link URLs |\n| `vent:warning` | `{"type":"vent:warning","message":"...","code":"..."}` | Reports provider/runtime warnings worth preserving in run metadata |\n\nVent sends `{"type":"end-call"}` to the agent when the test is done.\n\nAll text frames are optional \u2014 audio-only agents work fine with VAD-based turn detection.\n\n## Full Config Schema\n\n- ALL calls MUST reference the agent\'s real context (system prompt, tools, knowledge base) from the codebase.\n\n<vent_run>\n{\n  "connection": { ... },\n  "calls": {\n    "happy-path": { ... },\n    "edge-case": { ... }\n  }\n}\n</vent_run>\n\nOne suite file per platform/adapter. `connection` is declared once, `calls` is a named map of call specs. Each key becomes the call name. Run one call at a time with `--call <name>`.\n\n<config_connection>\n{\n  "connection": {\n    "adapter": "required -- websocket | livekit | vapi | retell | elevenlabs | bland",\n    "start_command": "shell command to start agent (relay only, required for local)",\n    "health_endpoint": "health check path after start_command (default: /health, relay only, required for local)",\n    "agent_url": "hosted custom agent URL (wss:// or https://). Use for BYO hosted agents.",\n    "agent_port": "local agent port (default: 3001, required for local)",\n    "platform": "optional authoring convenience for platform-direct adapters only. The CLI resolves this locally, creates/updates a saved platform connection, and strips raw provider secrets before submit. Do not use for websocket start_command or agent_url runs."\n  }\n}\n\n<credential_resolution>\nIMPORTANT: How to handle platform credentials (API keys, secrets, agent IDs):\n\nThere are two product modes:\n- `BYO agent runtime`: your agent owns its own provider credentials. This covers both `start_command` (local) and `agent_url` (hosted custom endpoint).\n- `Platform-direct runtime`: Vent talks to `vapi`, `retell`, `elevenlabs`, `bland`, or `livekit` directly. This is the only mode that uses saved platform connections.\n\n1. For `start_command` and `agent_url` runs, do NOT put Deepgram / ElevenLabs / OpenAI / other provider keys into Vent config unless the Vent adapter itself needs them. Those credentials belong to the user\'s local or hosted agent runtime.\n2. For platform-direct adapters (`vapi`, `retell`, `elevenlabs`, `bland`, `livekit`), the CLI auto-resolves credentials from `.env.local`, `.env`, and the current shell env. If those env vars already exist, you can omit credential fields from the config JSON entirely.\n3. If you include credential fields in the config, put the ACTUAL VALUE, NOT the env var name. WRONG: `"vapi_api_key": "VAPI_API_KEY"`. RIGHT: `"vapi_api_key": "sk-abc123..."` or omit the field.\n4. The CLI uses the resolved provider config to create or update a saved platform connection server-side, then submits only `platform_connection_id`. Users should not manually author `platform_connection_id`.\n5. To check whether credentials are already available, inspect `.env.local`, `.env`, and any relevant shell env visible to the CLI process.\n\nAuto-resolved env vars per platform:\n| Platform | Config field | Env var (auto-resolved from `.env.local`, `.env`, or shell env) |\n|----------|-------------|-----------------------------------|\n| Vapi | vapi_api_key | VAPI_API_KEY |\n| Vapi | vapi_assistant_id | VAPI_ASSISTANT_ID |\n| Bland | bland_api_key | BLAND_API_KEY |\n| Bland | bland_pathway_id | BLAND_PATHWAY_ID |\n| LiveKit | livekit_api_key | LIVEKIT_API_KEY |\n| LiveKit | livekit_api_secret | LIVEKIT_API_SECRET |\n| LiveKit | livekit_url | LIVEKIT_URL |\n| Retell | retell_api_key | RETELL_API_KEY |\n| Retell | retell_agent_id | RETELL_AGENT_ID |\n| ElevenLabs | elevenlabs_api_key | ELEVENLABS_API_KEY |\n| ElevenLabs | elevenlabs_agent_id | ELEVENLABS_AGENT_ID |\n\nThe CLI strips raw platform secrets before `/runs/submit`. Platform-direct runs go through a saved `platform_connection_id` automatically. BYO agent runs (`start_command` and `agent_url`) do not.\n</credential_resolution>\n\n<config_adapter_rules>\nWebSocket (local agent via relay):\n{\n  "connection": {\n    "adapter": "websocket",\n    "start_command": "npm run start",\n    "health_endpoint": "/health",\n    "agent_port": 3001\n  }\n}\n\nWebSocket (hosted custom agent):\n{\n  "connection": {\n    "adapter": "websocket",\n    "agent_url": "https://my-agent.fly.dev"\n  }\n}\n\nRetell:\n{\n  "connection": {\n    "adapter": "retell",\n    "platform": { "provider": "retell" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: RETELL_API_KEY, RETELL_AGENT_ID. Only add retell_api_key/retell_agent_id to the JSON if those env vars are not already available.\n\nBland:\n{\n  "connection": {\n    "adapter": "bland",\n    "platform": { "provider": "bland" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: BLAND_API_KEY, BLAND_PATHWAY_ID. Only add bland_api_key/bland_pathway_id to the JSON if those env vars are not already available.\nNote: All agent config (voice, model, tools, etc.) is set on the pathway itself, not in Vent config.\n\nVapi:\n{\n  "connection": {\n    "adapter": "vapi",\n    "platform": { "provider": "vapi" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: VAPI_API_KEY, VAPI_ASSISTANT_ID. Only add vapi_api_key/vapi_assistant_id to the JSON if those env vars are not already available.\nmax_concurrency for Vapi: Starter=10, Growth=50, Enterprise=100+. Ask the user which tier they\'re on. If unknown, default to 10.\nAll assistant config (voice, model, transcriber, interruption settings, etc.) is set on the Vapi assistant itself, not in Vent config.\n\nElevenLabs:\n{\n  "connection": {\n    "adapter": "elevenlabs",\n    "platform": { "provider": "elevenlabs" }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: ELEVENLABS_API_KEY, ELEVENLABS_AGENT_ID. Only add elevenlabs_api_key/elevenlabs_agent_id to the JSON if those env vars are not already available.\n\nLiveKit:\n{\n  "connection": {\n    "adapter": "livekit",\n    "platform": {\n      "provider": "livekit",\n      "livekit_agent_name": "my-agent",\n      "max_concurrency": 5\n    }\n  }\n}\nCredentials auto-resolve from `.env.local`, `.env`, or shell env: LIVEKIT_API_KEY, LIVEKIT_API_SECRET, LIVEKIT_URL. Only add these to the JSON if those env vars are not already available.\nlivekit_agent_name is optional -- only needed if the agent registers with an explicit agent_name in WorkerOptions. Omit for automatic dispatch.\nThe livekit adapter requires the LiveKit Agents SDK. It depends on Agents SDK signals (lk.agent.state, lk.transcription) for readiness detection, turn timing, and component latency. Custom LiveKit participants not using the Agents SDK should use the websocket adapter with a relay instead.\nmax_concurrency: Free/Build=5, Ship=20, Scale=50+. Ask the user which tier they\'re on. If unknown, default to 5.\n</config_adapter_rules>\n</config_connection>\n\n\n<call_config>\n<tool_call_capture>\nvapi/retell/elevenlabs/bland: automatic via platform API (no user code needed).\nWebSocket/WebRTC: user\'s agent must emit tool calls:\n  WebSocket \u2014 JSON text frame: {"type":"tool_call","name":"...","arguments":{},"result":{},"successful":true,"duration_ms":150}\n  WebRTC/LiveKit \u2014 publishData() or sendText() on topic "vent:tool-calls". Same JSON.\n</tool_call_capture>\n\n<component_timing>\nPlatform adapters (vapi/retell/elevenlabs/bland/livekit) get STT/LLM/TTS breakdown automatically.\nWebSocket agents can opt in by sending a JSON text frame after each agent turn:\n  {"type":"vent:timing","stt_ms":120,"llm_ms":450,"tts_ms":80}\nAll fields optional. Send one per agent response. Without this, component_latency is omitted from results.\nWhen modifying a WebSocket agent\'s code, add this text frame after TTS completes to enable component latency reporting.\n</component_timing>\n\n<metadata_capture>\nWebSocket and LiveKit/WebRTC agents can also emit richer observability metadata:\n  {"type":"vent:session","platform":"custom","provider_call_id":"call_123","provider_session_id":"session_abc"}\n  {"type":"vent:call-metadata","call_metadata":{"recording_url":"https://...","cost_usd":0.12,"provider_debug_urls":{"log":"https://..."}}}\n  {"type":"vent:debug-url","label":"trace","url":"https://..."}\n  {"type":"vent:session-report","report":{"room_name":"room-123","events":[...],"metrics":[...]}}\n  {"type":"vent:metrics","event":"metrics_collected","metric_type":"eou","metrics":{"speechId":"speech_123","endOfUtteranceDelayMs":420}}\n  {"type":"vent:function-tools-executed","event":"function_tools_executed","hasAgentHandoff":true,"tool_calls":[{"name":"lookup_customer","arguments":{"id":"123"}}]}\n  {"type":"vent:conversation-item","event":"conversation_item_added","item":{"type":"agent_handoff","newAgentId":"billing-agent"}}\n  {"type":"vent:session-usage","usage":{"llm":{"promptTokens":123,"completionTokens":45}}}\nTransport:\n  WebSocket \u2014 send JSON text frames with these payloads. WebSocket agents may also emit {"type":"vent:transcript","role":"caller","text":"I need to reschedule","turn_index":0} when they have native transcript text.\n  WebRTC/LiveKit \u2014 publishData() or sendText() on the matching "vent:*" topic, e.g. topic "vent:call-metadata" with the JSON body above.\nFor LiveKit, transcript and timing stay authoritative from native room signals (`lk.transcription`, `lk.agent.state`). Do not emit `vent:transcript` from LiveKit agents.\nFor LiveKit Node agents, prefer the first-party helper instead of manual forwarding:\n```ts\nimport { instrumentLiveKitAgent } from "@vent-hq/livekit";\n\nconst vent = instrumentLiveKitAgent({\n  ctx,\n  session,\n});\n```\nThis helper must run inside the LiveKit agent runtime with the existing Agents SDK `session` and `ctx` objects. It is the Vent integration layer on top of the Agents SDK, not a replacement for it.\nInstall it with `npm install @vent-hq/livekit` after the package is published to the `vent-hq` npm org. Until then, use the workspace package from this repo.\nThis automatically publishes only the in-agent-only LiveKit signals: `metrics_collected`, `function_tools_executed`, `conversation_item_added`, and a session report on close/shutdown.\nDo not use it to mirror room-visible signals like transcript, agent state timing, or room/session ID \u2014 Vent already gets those from LiveKit itself.\nFor LiveKit inside-agent forwarding, prefer sending the raw LiveKit event payloads on:\n  `vent:metrics`\n  `vent:function-tools-executed`\n  `vent:conversation-item`\n  `vent:session-usage`\nUse these metadata events when the agent runtime already knows native IDs, recordings, warnings, debug links, session reports, metrics events, or handoff artifacts. This gives custom and LiveKit agents parity with hosted adapters without needing a LiveKit Cloud connector.\n</metadata_capture>\n\n<config_call>\nEach call in the `calls` map. The key is the call name (e.g. `"reschedule-appointment"`, not `"call-1"`).\n{\n      "caller_prompt": "required \u2014 caller persona and behavior (name -> goal -> emotion -> conditional behavior)",\n      "max_turns": "required \u2014 default 6",\n      "silence_threshold_ms": "optional \u2014 end-of-turn threshold ms (default 800, 200-10000). 800-1200 FAQ, 2000-3000 tool calls, 3000-5000 complex reasoning.",\n      "persona": "optional \u2014 caller behavior controls",\n      {\n        "pace": "slow | normal | fast",\n        "clarity": "clear | vague | rambling",\n        "disfluencies": "true | false",\n        "cooperation": "cooperative | reluctant | hostile",\n        "emotion": "neutral | cheerful | confused | frustrated | skeptical | rushed",\n        "interruption_style": "optional preplanned interrupt tendency: low | high. If set, Vent may pre-plan a caller cut-in before the agent turn starts. It does NOT make a mid-turn interrupt LLM call.",\n        "memory": "reliable | unreliable",\n        "intent_clarity": "clear | indirect | vague",\n        "confirmation_style": "explicit | vague"\n      },\n      "audio_actions": "optional \u2014 per-turn audio stress calls",\n      [\n        { "action": "interrupt", "at_turn": "N", "prompt": "what caller says" },\n        { "action": "inject_noise", "at_turn": "N", "noise_type": "babble | white | pink", "snr_db": "0-40" },\n        { "action": "split_sentence", "at_turn": "N", "split": { "part_a": "...", "part_b": "...", "pause_ms": "500-5000" } },\n        { "action": "noise_on_caller", "at_turn": "N" }\n      ],\n      "prosody": "optional \u2014 Hume emotion analysis (default false)",\n      "caller_audio": "optional \u2014 omit for clean audio",\n      {\n        "noise": { "type": "babble | white | pink", "snr_db": "0-40" },\n        "speed": "0.5-2.0 (1.0 = normal)",\n        "speakerphone": "true | false",\n        "mic_distance": "close | normal | far",\n        "clarity": "0.0-1.0 (1.0 = perfect)",\n        "accent": "american | british | australian | filipino | spanish_mexican | spanish_peninsular | spanish_colombian | spanish_argentine | german | french | italian | dutch | japanese",\n        "packet_loss": "0.0-0.3",\n        "jitter_ms": "0-100"\n      },\n      "language": "optional \u2014 ISO 639-1: en, es, fr, de, it, nl, ja"\n}\n\nInterruption rules:\n- `audio_actions: [{ "action": "interrupt", ... }]` is the deterministic per-turn interrupt test. Prefer this for evaluation.\n- `persona.interruption_style` is only a preplanned caller tendency. If used, Vent decides before the agent response starts whether this turn may cut in.\n- Vent no longer pauses mid-turn to ask a second LLM whether to interrupt.\n- For production-faithful testing, prefer explicit `audio_actions.interrupt` over persona interruption.\n\n<examples_call>\n<simple_suite_example>\n{\n  "connection": {\n    "adapter": "vapi",\n    "platform": { "provider": "vapi" }\n  },\n  "calls": {\n    "reschedule-appointment": {\n      "caller_prompt": "You are Maria, calling to reschedule her dentist appointment from Thursday to next Tuesday. She\'s in a hurry and wants this done quickly.",\n      "max_turns": 8\n    },\n    "cancel-appointment": {\n      "caller_prompt": "You are Tom, calling to cancel his appointment for Friday. He\'s calm and just wants confirmation.",\n      "max_turns": 6\n    }\n  }\n}\n</simple_suite_example>\n\n<advanced_call_example>\nA call entry with advanced options (persona, audio actions, prosody):\n{\n  "noisy-interruption-booking": {\n    "caller_prompt": "You are James, an impatient customer calling from a loud coffee shop to book a plumber for tomorrow morning. You interrupt the agent mid-sentence when they start listing availability \u2014 you just want the earliest slot.",\n    "max_turns": 12,\n    "persona": { "pace": "fast", "cooperation": "reluctant", "emotion": "rushed", "interruption_style": "high" },\n    "audio_actions": [\n      { "action": "interrupt", "at_turn": 3, "prompt": "Just give me the earliest one!" },\n      { "action": "inject_noise", "at_turn": 1, "noise_type": "babble", "snr_db": 15 }\n    ],\n    "caller_audio": { "noise": { "type": "babble", "snr_db": 20 }, "speed": 1.3 },\n    "prosody": true\n  }\n}\n</advanced_call_example>\n\n</examples_call>\n</config_call>\n\n<output_conversation_test>\n{\n  "name": "sarah-hotel-booking",\n  "status": "completed",\n  "caller_prompt": "You are Sarah, calling to book...",\n  "duration_ms": 45200,\n  "error": null,\n  "transcript": [\n    { "role": "caller", "text": "Hi, I\'d like to book..." },\n    { "role": "agent", "text": "Sure! What date?", "ttfb_ms": 650, "ttfw_ms": 780, "audio_duration_ms": 2400 },\n    { "role": "agent", "text": "Let me check avail\u2014", "ttfb_ms": 540, "ttfw_ms": 620, "audio_duration_ms": 1400, "interrupted": true },\n    { "role": "caller", "text": "Just the earliest slot please", "audio_duration_ms": 900, "is_interruption": true },\n    { "role": "agent", "text": "Sure, the earliest is 9 AM tomorrow.", "ttfb_ms": 220, "ttfw_ms": 260, "audio_duration_ms": 2100 }\n  ],\n  "latency": {\n    "response_time_ms": 890, "response_time_source": "ttfw",\n    "p50_response_time_ms": 850, "p90_response_time_ms": 1100, "p95_response_time_ms": 1400, "p99_response_time_ms": 1550,\n    "first_response_time_ms": 1950,\n    "mean_ttfw_ms": 890, "p50_ttfw_ms": 850, "p95_ttfw_ms": 1400, "p99_ttfw_ms": 1550,\n    "first_turn_ttfw_ms": 1950, "total_silence_ms": 4200, "mean_turn_gap_ms": 380,\n    "drift_slope_ms_per_turn": -45.2, "mean_silence_pad_ms": 128, "mouth_to_ear_est_ms": 1020\n  },\n  "transcript_quality": {\n    "wer": 0.04,\n    "hallucination_events": [\n      { "error_count": 5, "reference_text": "triple five one two", "hypothesis_text": "five five five nine two" }\n    ],\n    "repetition_score": 0.05,\n    "reprompt_count": 0,\n    "filler_word_rate": 0.8,\n    "words_per_minute": 148\n  },\n  "audio_analysis": {\n    "caller_talk_time_ms": 12400,\n    "agent_talk_time_ms": 28500,\n    "agent_speech_ratio": 0.72,\n    "talk_ratio_vad": 0.69,\n    "interruption_rate": 0.25,\n    "interruption_count": 1,\n    "agent_overtalk_after_barge_in_ms": 280,\n    "agent_interrupting_user_rate": 0.0,\n    "agent_interrupting_user_count": 0,\n    "missed_response_windows": 0,\n    "longest_monologue_ms": 5800,\n    "silence_gaps_over_2s": 1,\n    "total_internal_silence_ms": 2400,\n    "mean_agent_speech_segment_ms": 3450\n  },\n  "tool_calls": {\n    "total": 2, "successful": 2, "failed": 0, "mean_latency_ms": 340,\n    "names": ["check_availability", "book_appointment"],\n    "observed": [{ "name": "check_availability", "arguments": { "date": "2026-03-12" }, "result": { "slots": ["09:00", "10:00"] }, "successful": true, "latency_ms": 280, "turn_index": 3 }]\n  },\n  "component_latency": {\n    "mean_stt_ms": 120, "mean_llm_ms": 450, "mean_tts_ms": 80,\n    "p95_stt_ms": 180, "p95_llm_ms": 620, "p95_tts_ms": 110,\n    "mean_speech_duration_ms": 2100,\n    "bottleneck": "llm"\n  },\n  "call_metadata": {\n    "platform": "vapi",\n    "cost_usd": 0.08,\n    "recording_url": "https://example.com/recording",\n    "ended_reason": "customer_ended_call",\n    "transfers": []\n  },\n  "warnings": [],\n  "audio_actions": [],\n  "emotion": {\n    "naturalness": 0.72, "mean_calmness": 0.65, "mean_confidence": 0.58, "peak_frustration": 0.08, "emotion_trajectory": "stable"\n  }\n}\n\nAlways present: name, status, caller_prompt, duration_ms, error, transcript, tool_calls, warnings, audio_actions. Nullable when analysis didn\'t run: latency, transcript_quality, audio_analysis, component_latency, call_metadata, emotion (requires prosody: true), debug (requires --verbose).\n\n### Result presentation\n\nWhen you report a conversation result to the user, always include:\n\n1. **Summary** \u2014 the overall verdict and the 1-3 most important findings.\n2. **Transcript summary** \u2014 a short narrative of what happened in the call.\n3. **Recording URL** \u2014 include `call_metadata.recording_url` when present; explicitly say when it is unavailable.\n4. **Next steps** \u2014 concrete fixes, follow-up tests, or why no change is needed.\n\nUse metrics to support the summary, not as the whole answer. Do not dump raw numbers without interpretation.\n\nWhen `call_metadata.transfer_attempted` is present, explicitly say whether the transfer only appeared attempted or was mechanically verified as completed (`call_metadata.transfer_completed`). Use `call_metadata.transfers[]` to report transfer type, destination, status, and sources.\n\n### Judging guidance\n\nUse the transcript, metrics, test scenario, and relevant agent instructions/system prompt to judge:\n\n| Dimension | What to check |\n|--------|----------------|\n| **Hallucination detection** | Check whether the agent stated anything not grounded in its instructions, tools, or the conversation itself. Treat `transcript_quality.hallucination_events` only as a speech-recognition warning signal, not proof of agent hallucination. |\n| **Instruction following** | Compare the agent\'s behavior against its system prompt and the test\'s expected constraints. |\n| **Context retention** | Check whether the agent forgot or contradicted information established earlier in the call. |\n| **Semantic accuracy** | Check whether the agent correctly understood the caller\'s intent and responded to the real request. |\n| **Goal completion** | Decide whether the agent achieved what the test scenario was designed to verify. |\n| **Transfer correctness** | For transfer scenarios, judge whether transfer was appropriate, whether it completed, whether it went to the expected destination, and whether enough context was passed during the handoff. |\n\n### Interruption evaluation\n\nWhen the transcript contains `interrupted: true` / `is_interruption: true` turns, evaluate these metrics by reading the transcript:\n\n| Metric | How to evaluate | Target |\n|--------|----------------|--------|\n| **Recovery rate** | For each interrupted turn: does the post-interrupt agent response acknowledge or address the interruption? | >90% |\n| **Context retention** | After the interruption, does the agent remember pre-interrupt conversation state? | >95% |\n| **Agent overtalk after barge-in** | Use `audio_analysis.agent_overtalk_after_barge_in_ms` when available. Lower is better because it measures how long the agent kept speaking after the caller cut in. | <500ms acceptable |\n| **Agent interrupting user rate** | Use `audio_analysis.agent_interrupting_user_rate` and the transcript to see whether the agent starts speaking before the caller finished. | 0 ideal |\n\nReport these alongside standard metrics when interruption calls run.\n</output_conversation_test>\n</call_config>\n\n\n## Exit Codes\n\n0=pass, 1=fail, 2=error\n';
 // src/lib/setup.ts
 var SUITE_SCAFFOLD = JSON.stringify(
@@ -6675,7 +7000,8 @@ var RUN_USAGE = `Usage: vent-hq run -f <suite.json> [options]
 Options:
   --file, -f     Path to suite JSON file (required)
   --call         Name of the call to run (required if suite has multiple calls)
-  --session, -s  Reuse an existing local agent session`;
+  --session, -s  Reuse an existing local agent session
+  --verbose, -v  Include verbose fields in the result JSON`;
 var AGENT_USAGE = `Usage: vent-hq agent <command> [options]
 Commands:
@@ -6688,7 +7014,7 @@ Start options:
 Stop options:
   vent-hq agent stop <session-id>`;
-var STATUS_USAGE = `Usage: vent-hq status <run-id>`;
+var STATUS_USAGE = `Usage: vent-hq status <run-id> [--verbose]`;
 async function main() {
   loadDotenv();
   const args = process.argv.slice(2);
@@ -6698,7 +7024,7 @@ async function main() {
     return 0;
   }
   if (command === "--version" || command === "-v") {
-    const pkg = await import("./package-YOCP6D2K.mjs");
+    const pkg = await import("./package-767KASWC.mjs");
     console.log(`vent-hq ${pkg.default.version}`);
     return 0;
   }
@@ -6717,7 +7043,8 @@ async function main() {
         options: {
           file: { type: "string", short: "f" },
           call: { type: "string" },
-          session: { type: "string", short: "s" }
+          session: { type: "string", short: "s" },
+          verbose: { type: "boolean", short: "v", default: false }
         },
         strict: true
       });
@@ -6729,7 +7056,8 @@ async function main() {
       return runCommand({
         file: values.file,
         call: values.call,
-        session: values.session
+        session: values.session,
+        verbose: values.verbose
       });
     }
     case "agent": {
@@ -6769,8 +7097,20 @@ async function main() {
         console.log(STATUS_USAGE);
         return 0;
       }
-      const runId = commandArgs[0];
-      return statusCommand({ runId });
+      const { values, positionals } = parseArgs({
+        args: commandArgs,
+        options: {
+          verbose: { type: "boolean", short: "v", default: false }
+        },
+        allowPositionals: true,
+        strict: true
+      });
+      const runId = positionals[0];
+      if (!runId) {
+        console.log(STATUS_USAGE);
+        return 2;
+      }
+      return statusCommand({ runId, verbose: values.verbose });
     }
     case "stop": {
       const runId = commandArgs[0];