xo-harness 0.1.2 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. package/README.md +17 -0
  2. package/dist/internal/harness/index.d.ts +3 -0
  3. package/dist/internal/harness/index.js +3 -0
  4. package/dist/internal/harness/message.d.ts +1 -2
  5. package/dist/internal/harness/message.js +45 -19
  6. package/dist/internal/harness/report-diff.d.ts +46 -0
  7. package/dist/internal/harness/report-diff.js +62 -0
  8. package/dist/internal/harness/report.d.ts +2 -0
  9. package/dist/internal/harness/report.js +10 -0
  10. package/dist/internal/harness/shadow.d.ts +9 -2
  11. package/dist/internal/harness/shadow.js +5 -1
  12. package/dist/internal/harness/task-supervisor.js +21 -6
  13. package/dist/internal/harness/tool-delivery.d.ts +17 -0
  14. package/dist/internal/harness/tool-delivery.js +53 -0
  15. package/dist/internal/harness/tool-policy.d.ts +52 -0
  16. package/dist/internal/harness/tool-policy.js +22 -0
  17. package/dist/internal/harness/tool-runtime.d.ts +4 -0
  18. package/dist/internal/harness/tool-runtime.js +82 -1
  19. package/dist/internal/harness/tools.d.ts +22 -1
  20. package/dist/internal/harness/voice-session.d.ts +9 -1
  21. package/dist/internal/harness/voice-session.js +22 -2
  22. package/dist/internal/harness/xo.d.ts +13 -0
  23. package/dist/internal/harness/xo.js +8 -0
  24. package/dist/internal/protocol/events.d.ts +72 -0
  25. package/dist/internal/protocol/events.js +14 -1
  26. package/dist/internal/protocol/provider.d.ts +84 -2
  27. package/dist/internal/protocol/provider.js +51 -3
  28. package/dist/internal/provider/grok-voice.d.ts +1 -0
  29. package/dist/internal/provider/grok-voice.js +4 -0
  30. package/dist/internal/provider/openai-realtime.d.ts +10 -1
  31. package/dist/internal/provider/openai-realtime.js +26 -1
  32. package/dist/internal/provider/realtime-session.d.ts +6 -0
  33. package/dist/internal/provider/realtime-session.js +267 -32
  34. package/dist/internal/provider-fake/replay-voice-provider.d.ts +6 -2
  35. package/dist/internal/provider-fake/replay-voice-provider.js +13 -3
  36. package/dist/internal/skills/index.d.ts +2 -0
  37. package/dist/internal/skills/index.js +2 -0
  38. package/dist/internal/skills/node.d.ts +7 -0
  39. package/dist/internal/skills/node.js +99 -0
  40. package/dist/internal/skills/skill.d.ts +17 -0
  41. package/dist/internal/skills/skill.js +42 -0
  42. package/dist/internal/skills/tools.d.ts +7 -0
  43. package/dist/internal/skills/tools.js +82 -0
  44. package/dist/internal/storage/memory.d.ts +2 -0
  45. package/dist/internal/storage/memory.js +1 -0
  46. package/dist/internal/tools-openai/delegate-conversation.d.ts +12 -0
  47. package/dist/internal/tools-openai/delegate-conversation.js +66 -0
  48. package/dist/internal/tools-openai/index.d.ts +24 -0
  49. package/dist/internal/tools-openai/index.js +119 -0
  50. package/dist/internal/tools-openai/responses.d.ts +31 -0
  51. package/dist/internal/tools-openai/responses.js +146 -0
  52. package/dist/skills-node.d.ts +1 -0
  53. package/dist/skills-node.js +1 -0
  54. package/dist/skills.d.ts +1 -0
  55. package/dist/skills.js +1 -0
  56. package/dist/storage-memory.d.ts +1 -0
  57. package/dist/storage-memory.js +1 -0
  58. package/dist/tools-openai.d.ts +1 -0
  59. package/dist/tools-openai.js +1 -0
  60. package/package.json +18 -1
@@ -15,6 +15,8 @@ export const ProviderCapabilitiesSchema = z.object({
15
15
  supportsAsyncContext: z.boolean(),
16
16
  /** Honors playout acknowledgements for truncation-accurate barge-in. */
17
17
  supportsPlayoutAcknowledgements: z.boolean(),
18
+ /** Emits response.state for pending generation and queued continuations. Optional for continuous models. */
19
+ reportsResponseState: z.boolean().optional(),
18
20
  });
19
21
  export const PlayoutProgressSchema = z.object({
20
22
  streamId: z.string().min(1),
@@ -50,25 +52,71 @@ export const TranscriptPayloadSchema = z.object({
50
52
  text: z.string().min(1),
51
53
  streamId: z.string().min(1).optional(),
52
54
  });
53
- /** Incremental assistant text as the model streams it; finalized by a `transcript`. */
55
+ /** Incremental text from either speaker; finalized by a matching `transcript`. */
54
56
  export const TranscriptDeltaPayloadSchema = z.object({
55
57
  role: z.enum(["user", "assistant"]),
56
58
  delta: z.string().min(1),
57
59
  streamId: z.string().min(1).optional(),
58
60
  });
61
+ /** Generation state only: idle says nothing about buffered audio playback or background tasks. */
62
+ export const ResponseStateSchema = z.enum(["pending", "idle"]);
63
+ /** Limits optional provider diagnostics so inspection artifacts stay bounded. */
64
+ export const PROVIDER_DIAGNOSTIC_STRING_MAX_LENGTH = 256;
65
+ export const PROVIDER_DIAGNOSTIC_MAX_OUTPUTS = 32;
66
+ export const PROVIDER_DIAGNOSTIC_MAX_CONTENT_TYPES = 16;
67
+ const ProviderDiagnosticStringSchema = z.string().min(1).max(PROVIDER_DIAGNOSTIC_STRING_MAX_LENGTH);
68
+ /**
69
+ * Bounded, allow-listed provider observations for diagnosing a silent turn. These
70
+ * facts are never used to schedule media or model work, and intentionally exclude
71
+ * vendor payloads, reasoning, credentials, and content text.
72
+ */
73
+ export const ProviderDiagnosticSchema = z.discriminatedUnion("kind", [
74
+ z.object({
75
+ kind: z.literal("session"),
76
+ model: ProviderDiagnosticStringSchema.optional(),
77
+ }),
78
+ z.object({
79
+ kind: z.literal("speech"),
80
+ phase: z.enum(["started", "stopped"]),
81
+ streamId: ProviderDiagnosticStringSchema.optional(),
82
+ audioOffsetMs: z.number().nonnegative().optional(),
83
+ }),
84
+ z.object({
85
+ kind: z.literal("response"),
86
+ phase: z.enum(["started", "completed"]),
87
+ responseId: ProviderDiagnosticStringSchema.optional(),
88
+ status: ProviderDiagnosticStringSchema.optional(),
89
+ reason: ProviderDiagnosticStringSchema.optional(),
90
+ /** A bounded provider error code, when response status reports one. */
91
+ code: ProviderDiagnosticStringSchema.optional(),
92
+ outputs: z
93
+ .array(z.object({
94
+ streamId: ProviderDiagnosticStringSchema.optional(),
95
+ type: ProviderDiagnosticStringSchema,
96
+ contentTypes: z.array(ProviderDiagnosticStringSchema).max(PROVIDER_DIAGNOSTIC_MAX_CONTENT_TYPES),
97
+ }))
98
+ .max(PROVIDER_DIAGNOSTIC_MAX_OUTPUTS)
99
+ .optional(),
100
+ }),
101
+ ]);
59
102
  /**
60
103
  * What a provider adapter yields to the harness. Deliberately minimal: adapters
61
- * translate vendor wire dialects into these eight shapes, and the harness turns them
104
+ * translate vendor wire dialects into these shapes, and the harness turns them
62
105
  * into HarnessEvents. Anything not expressible here does not exist to the harness.
63
106
  */
64
107
  export const ProviderEventSchema = z.discriminatedUnion("type", [
65
108
  /** Session configuration acknowledged; the model is listening. */
66
109
  z.object({ type: z.literal("ready") }),
110
+ z.object({ type: z.literal("response.state"), state: ResponseStateSchema }),
111
+ /** Optional allow-listed diagnostics for explaining provider behavior after the fact. */
112
+ z.object({ type: z.literal("diagnostic"), diagnostic: ProviderDiagnosticSchema }),
67
113
  /** Model speech as sample-addressed PCM. */
68
114
  z.object({ type: z.literal("audio.output"), chunk: AudioChunkSchema }),
115
+ /** Clear queued playback for this stream and ignore its subsequent audio chunks. */
116
+ z.object({ type: z.literal("audio.interrupted"), streamId: z.string().min(1) }),
69
117
  /** Final transcript text for either role. */
70
118
  TranscriptPayloadSchema.extend({ type: z.literal("transcript") }),
71
- /** Incremental assistant text; assembled into a streaming text part. */
119
+ /** Incremental text from either speaker; assembled into a streaming text part. */
72
120
  TranscriptDeltaPayloadSchema.extend({ type: z.literal("transcript.delta") }),
73
121
  /** The model requested a tool invocation. */
74
122
  z.object({ type: z.literal("tool.call"), call: ToolCallSchema }),
@@ -39,6 +39,7 @@ export declare class GrokVoiceProvider implements VoiceProvider {
39
39
  readonly supportsToolCalls: true;
40
40
  readonly supportsAsyncContext: true;
41
41
  readonly supportsPlayoutAcknowledgements: true;
42
+ readonly reportsResponseState: true;
42
43
  };
43
44
  constructor(options: GrokVoiceProviderOptions);
44
45
  createSession(options: CreateProviderSessionOptions): Promise<VoiceProviderSession>;
@@ -22,6 +22,7 @@ export class GrokVoiceProvider {
22
22
  supportsToolCalls: true,
23
23
  supportsAsyncContext: true,
24
24
  supportsPlayoutAcknowledgements: true,
25
+ reportsResponseState: true,
25
26
  };
26
27
  #options;
27
28
  constructor(options) {
@@ -41,10 +42,13 @@ export class GrokVoiceProvider {
41
42
  });
42
43
  return new RealtimeVoiceSession(socket, {
43
44
  label: "Grok voice",
45
+ inputTranscriptSettleMs: 300,
44
46
  sessionUpdate: this.#sessionUpdate(options, inputSampleRate),
45
47
  inputSampleRate,
46
48
  outputSampleRate: DEFAULT_SAMPLE_RATE,
47
49
  announceTaskSettlement: this.#options.announceTaskSettlement ?? true,
50
+ autoRespondToAudio: this.#options.turnDetection !== null && this.#options.turnDetection?.create_response !== false,
51
+ interruptOnSpeech: this.#options.turnDetection !== null && this.#options.turnDetection?.interrupt_response !== false,
48
52
  }, options);
49
53
  }
50
54
  #sessionUpdate(options, inputSampleRate) {
@@ -12,10 +12,18 @@ export interface OpenAIRealtimeProviderOptions {
12
12
  voice?: string;
13
13
  /** Sample rate of the PCM16 mono input the harness will send. */
14
14
  inputSampleRate?: number;
15
+ /** Input filtering before VAD/model processing. Omitted keeps the API default; null disables it. */
16
+ noiseReduction?: "near_field" | "far_field" | null;
15
17
  /** Server-side turn detection. Defaults to semantic VAD; null disables it for push-to-talk. */
16
18
  turnDetection?: Record<string, unknown> | null;
17
- /** Input transcription model; final transcripts surface as transcript events. Null disables. */
19
+ /** Input transcription model; partial/final text surfaces as transcript.delta/transcript. Null disables. */
18
20
  transcriptionModel?: string | null;
21
+ /** Optional ISO-639-1 input language hint, e.g. cs or en. Omitted means automatic detection. */
22
+ transcriptionLanguage?: string;
23
+ /** Latency/accuracy tradeoff for gpt-live-transcribe. */
24
+ transcriptionDelay?: "minimal" | "low" | "medium" | "high" | "xhigh";
25
+ /** Realtime 2 reasoning/latency tradeoff. Omitted leaves the model default intact. */
26
+ reasoningEffort?: "minimal" | "low" | "medium" | "high" | "xhigh";
19
27
  /** Request a spoken response when a background task settles. */
20
28
  announceTaskSettlement?: boolean;
21
29
  handshakeTimeoutMs?: number;
@@ -37,6 +45,7 @@ export declare class OpenAIRealtimeVoiceProvider implements VoiceProvider {
37
45
  readonly supportsToolCalls: true;
38
46
  readonly supportsAsyncContext: true;
39
47
  readonly supportsPlayoutAcknowledgements: true;
48
+ readonly reportsResponseState: true;
40
49
  };
41
50
  constructor(options: OpenAIRealtimeProviderOptions);
42
51
  createSession(options: CreateProviderSessionOptions): Promise<VoiceProviderSession>;
@@ -21,6 +21,7 @@ export class OpenAIRealtimeVoiceProvider {
21
21
  supportsToolCalls: true,
22
22
  supportsAsyncContext: true,
23
23
  supportsPlayoutAcknowledgements: true,
24
+ reportsResponseState: true,
24
25
  };
25
26
  #options;
26
27
  constructor(options) {
@@ -45,6 +46,8 @@ export class OpenAIRealtimeVoiceProvider {
45
46
  inputSampleRate,
46
47
  outputSampleRate: DEFAULT_SAMPLE_RATE,
47
48
  announceTaskSettlement: this.#options.announceTaskSettlement ?? true,
49
+ autoRespondToAudio: this.#options.turnDetection !== null && this.#options.turnDetection?.create_response !== false,
50
+ interruptOnSpeech: this.#options.turnDetection !== null && this.#options.turnDetection?.interrupt_response !== false,
48
51
  }, options);
49
52
  }
50
53
  #sessionUpdate(options, inputSampleRate) {
@@ -60,7 +63,26 @@ export class OpenAIRealtimeVoiceProvider {
60
63
  input: {
61
64
  format: { type: "audio/pcm", rate: inputSampleRate },
62
65
  turn_detection: turnDetection,
63
- ...(transcriptionModel === null ? {} : { transcription: { model: transcriptionModel } }),
66
+ ...(this.#options.noiseReduction === undefined
67
+ ? {}
68
+ : {
69
+ noise_reduction: this.#options.noiseReduction === null ? null : { type: this.#options.noiseReduction },
70
+ }),
71
+ ...(transcriptionModel === null
72
+ ? {}
73
+ : {
74
+ transcription: {
75
+ model: transcriptionModel,
76
+ ...(this.#options.transcriptionLanguage === undefined
77
+ ? {}
78
+ : transcriptionModel.startsWith("gpt-live-transcribe")
79
+ ? { languages: [this.#options.transcriptionLanguage] }
80
+ : { language: this.#options.transcriptionLanguage }),
81
+ ...(this.#options.transcriptionDelay === undefined
82
+ ? {}
83
+ : { delay: this.#options.transcriptionDelay }),
84
+ },
85
+ }),
64
86
  },
65
87
  output: {
66
88
  format: { type: "audio/pcm", rate: DEFAULT_SAMPLE_RATE },
@@ -71,6 +93,9 @@ export class OpenAIRealtimeVoiceProvider {
71
93
  };
72
94
  if (options.instructions !== undefined)
73
95
  session.instructions = options.instructions;
96
+ if (this.#options.reasoningEffort !== undefined) {
97
+ session.reasoning = { effort: this.#options.reasoningEffort };
98
+ }
74
99
  return session;
75
100
  }
76
101
  }
@@ -15,6 +15,12 @@ export interface RealtimeWireOptions {
15
15
  inputSampleRate: number;
16
16
  outputSampleRate: number;
17
17
  announceTaskSettlement: boolean;
18
+ /** VAD will create a response after the user's speech; coalesce tool continuations into it. */
19
+ autoRespondToAudio: boolean;
20
+ /** Follow the provider's VAD interruption setting. Defaults to true. */
21
+ interruptOnSpeech?: boolean;
22
+ /** xAI sends revisions through its completed event; other providers settle immediately. */
23
+ inputTranscriptSettleMs?: number;
18
24
  }
19
25
  export declare class RealtimeVoiceSession implements VoiceProviderSession {
20
26
  #private;