xo-harness 0.1.2 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +17 -0
- package/dist/internal/harness/index.d.ts +3 -0
- package/dist/internal/harness/index.js +3 -0
- package/dist/internal/harness/message.d.ts +1 -2
- package/dist/internal/harness/message.js +45 -19
- package/dist/internal/harness/report-diff.d.ts +46 -0
- package/dist/internal/harness/report-diff.js +62 -0
- package/dist/internal/harness/report.d.ts +2 -0
- package/dist/internal/harness/report.js +10 -0
- package/dist/internal/harness/shadow.d.ts +9 -2
- package/dist/internal/harness/shadow.js +5 -1
- package/dist/internal/harness/task-supervisor.js +21 -6
- package/dist/internal/harness/tool-delivery.d.ts +17 -0
- package/dist/internal/harness/tool-delivery.js +53 -0
- package/dist/internal/harness/tool-policy.d.ts +52 -0
- package/dist/internal/harness/tool-policy.js +22 -0
- package/dist/internal/harness/tool-runtime.d.ts +4 -0
- package/dist/internal/harness/tool-runtime.js +82 -1
- package/dist/internal/harness/tools.d.ts +22 -1
- package/dist/internal/harness/voice-session.d.ts +9 -1
- package/dist/internal/harness/voice-session.js +22 -2
- package/dist/internal/harness/xo.d.ts +13 -0
- package/dist/internal/harness/xo.js +8 -0
- package/dist/internal/protocol/events.d.ts +72 -0
- package/dist/internal/protocol/events.js +14 -1
- package/dist/internal/protocol/provider.d.ts +84 -2
- package/dist/internal/protocol/provider.js +51 -3
- package/dist/internal/provider/grok-voice.d.ts +1 -0
- package/dist/internal/provider/grok-voice.js +4 -0
- package/dist/internal/provider/openai-realtime.d.ts +10 -1
- package/dist/internal/provider/openai-realtime.js +26 -1
- package/dist/internal/provider/realtime-session.d.ts +6 -0
- package/dist/internal/provider/realtime-session.js +267 -32
- package/dist/internal/provider-fake/replay-voice-provider.d.ts +6 -2
- package/dist/internal/provider-fake/replay-voice-provider.js +13 -3
- package/dist/internal/skills/index.d.ts +2 -0
- package/dist/internal/skills/index.js +2 -0
- package/dist/internal/skills/node.d.ts +7 -0
- package/dist/internal/skills/node.js +99 -0
- package/dist/internal/skills/skill.d.ts +17 -0
- package/dist/internal/skills/skill.js +42 -0
- package/dist/internal/skills/tools.d.ts +7 -0
- package/dist/internal/skills/tools.js +82 -0
- package/dist/internal/storage/memory.d.ts +2 -0
- package/dist/internal/storage/memory.js +1 -0
- package/dist/internal/tools-openai/delegate-conversation.d.ts +12 -0
- package/dist/internal/tools-openai/delegate-conversation.js +66 -0
- package/dist/internal/tools-openai/index.d.ts +24 -0
- package/dist/internal/tools-openai/index.js +119 -0
- package/dist/internal/tools-openai/responses.d.ts +31 -0
- package/dist/internal/tools-openai/responses.js +146 -0
- package/dist/skills-node.d.ts +1 -0
- package/dist/skills-node.js +1 -0
- package/dist/skills.d.ts +1 -0
- package/dist/skills.js +1 -0
- package/dist/storage-memory.d.ts +1 -0
- package/dist/storage-memory.js +1 -0
- package/dist/tools-openai.d.ts +1 -0
- package/dist/tools-openai.js +1 -0
- package/package.json +18 -1
|
@@ -15,6 +15,8 @@ export const ProviderCapabilitiesSchema = z.object({
|
|
|
15
15
|
supportsAsyncContext: z.boolean(),
|
|
16
16
|
/** Honors playout acknowledgements for truncation-accurate barge-in. */
|
|
17
17
|
supportsPlayoutAcknowledgements: z.boolean(),
|
|
18
|
+
/** Emits response.state for pending generation and queued continuations. Optional for continuous models. */
|
|
19
|
+
reportsResponseState: z.boolean().optional(),
|
|
18
20
|
});
|
|
19
21
|
export const PlayoutProgressSchema = z.object({
|
|
20
22
|
streamId: z.string().min(1),
|
|
@@ -50,25 +52,71 @@ export const TranscriptPayloadSchema = z.object({
|
|
|
50
52
|
text: z.string().min(1),
|
|
51
53
|
streamId: z.string().min(1).optional(),
|
|
52
54
|
});
|
|
53
|
-
/** Incremental
|
|
55
|
+
/** Incremental text from either speaker; finalized by a matching `transcript`. */
|
|
54
56
|
export const TranscriptDeltaPayloadSchema = z.object({
|
|
55
57
|
role: z.enum(["user", "assistant"]),
|
|
56
58
|
delta: z.string().min(1),
|
|
57
59
|
streamId: z.string().min(1).optional(),
|
|
58
60
|
});
|
|
61
|
+
/** Generation state only: idle says nothing about buffered audio playback or background tasks. */
|
|
62
|
+
export const ResponseStateSchema = z.enum(["pending", "idle"]);
|
|
63
|
+
/** Limits optional provider diagnostics so inspection artifacts stay bounded. */
|
|
64
|
+
export const PROVIDER_DIAGNOSTIC_STRING_MAX_LENGTH = 256;
|
|
65
|
+
export const PROVIDER_DIAGNOSTIC_MAX_OUTPUTS = 32;
|
|
66
|
+
export const PROVIDER_DIAGNOSTIC_MAX_CONTENT_TYPES = 16;
|
|
67
|
+
const ProviderDiagnosticStringSchema = z.string().min(1).max(PROVIDER_DIAGNOSTIC_STRING_MAX_LENGTH);
|
|
68
|
+
/**
|
|
69
|
+
* Bounded, allow-listed provider observations for diagnosing a silent turn. These
|
|
70
|
+
* facts are never used to schedule media or model work, and intentionally exclude
|
|
71
|
+
* vendor payloads, reasoning, credentials, and content text.
|
|
72
|
+
*/
|
|
73
|
+
export const ProviderDiagnosticSchema = z.discriminatedUnion("kind", [
|
|
74
|
+
z.object({
|
|
75
|
+
kind: z.literal("session"),
|
|
76
|
+
model: ProviderDiagnosticStringSchema.optional(),
|
|
77
|
+
}),
|
|
78
|
+
z.object({
|
|
79
|
+
kind: z.literal("speech"),
|
|
80
|
+
phase: z.enum(["started", "stopped"]),
|
|
81
|
+
streamId: ProviderDiagnosticStringSchema.optional(),
|
|
82
|
+
audioOffsetMs: z.number().nonnegative().optional(),
|
|
83
|
+
}),
|
|
84
|
+
z.object({
|
|
85
|
+
kind: z.literal("response"),
|
|
86
|
+
phase: z.enum(["started", "completed"]),
|
|
87
|
+
responseId: ProviderDiagnosticStringSchema.optional(),
|
|
88
|
+
status: ProviderDiagnosticStringSchema.optional(),
|
|
89
|
+
reason: ProviderDiagnosticStringSchema.optional(),
|
|
90
|
+
/** A bounded provider error code, when response status reports one. */
|
|
91
|
+
code: ProviderDiagnosticStringSchema.optional(),
|
|
92
|
+
outputs: z
|
|
93
|
+
.array(z.object({
|
|
94
|
+
streamId: ProviderDiagnosticStringSchema.optional(),
|
|
95
|
+
type: ProviderDiagnosticStringSchema,
|
|
96
|
+
contentTypes: z.array(ProviderDiagnosticStringSchema).max(PROVIDER_DIAGNOSTIC_MAX_CONTENT_TYPES),
|
|
97
|
+
}))
|
|
98
|
+
.max(PROVIDER_DIAGNOSTIC_MAX_OUTPUTS)
|
|
99
|
+
.optional(),
|
|
100
|
+
}),
|
|
101
|
+
]);
|
|
59
102
|
/**
|
|
60
103
|
* What a provider adapter yields to the harness. Deliberately minimal: adapters
|
|
61
|
-
* translate vendor wire dialects into these
|
|
104
|
+
* translate vendor wire dialects into these shapes, and the harness turns them
|
|
62
105
|
* into HarnessEvents. Anything not expressible here does not exist to the harness.
|
|
63
106
|
*/
|
|
64
107
|
export const ProviderEventSchema = z.discriminatedUnion("type", [
|
|
65
108
|
/** Session configuration acknowledged; the model is listening. */
|
|
66
109
|
z.object({ type: z.literal("ready") }),
|
|
110
|
+
z.object({ type: z.literal("response.state"), state: ResponseStateSchema }),
|
|
111
|
+
/** Optional allow-listed diagnostics for explaining provider behavior after the fact. */
|
|
112
|
+
z.object({ type: z.literal("diagnostic"), diagnostic: ProviderDiagnosticSchema }),
|
|
67
113
|
/** Model speech as sample-addressed PCM. */
|
|
68
114
|
z.object({ type: z.literal("audio.output"), chunk: AudioChunkSchema }),
|
|
115
|
+
/** Clear queued playback for this stream and ignore its subsequent audio chunks. */
|
|
116
|
+
z.object({ type: z.literal("audio.interrupted"), streamId: z.string().min(1) }),
|
|
69
117
|
/** Final transcript text for either role. */
|
|
70
118
|
TranscriptPayloadSchema.extend({ type: z.literal("transcript") }),
|
|
71
|
-
/** Incremental
|
|
119
|
+
/** Incremental text from either speaker; assembled into a streaming text part. */
|
|
72
120
|
TranscriptDeltaPayloadSchema.extend({ type: z.literal("transcript.delta") }),
|
|
73
121
|
/** The model requested a tool invocation. */
|
|
74
122
|
z.object({ type: z.literal("tool.call"), call: ToolCallSchema }),
|
|
@@ -39,6 +39,7 @@ export declare class GrokVoiceProvider implements VoiceProvider {
|
|
|
39
39
|
readonly supportsToolCalls: true;
|
|
40
40
|
readonly supportsAsyncContext: true;
|
|
41
41
|
readonly supportsPlayoutAcknowledgements: true;
|
|
42
|
+
readonly reportsResponseState: true;
|
|
42
43
|
};
|
|
43
44
|
constructor(options: GrokVoiceProviderOptions);
|
|
44
45
|
createSession(options: CreateProviderSessionOptions): Promise<VoiceProviderSession>;
|
|
@@ -22,6 +22,7 @@ export class GrokVoiceProvider {
|
|
|
22
22
|
supportsToolCalls: true,
|
|
23
23
|
supportsAsyncContext: true,
|
|
24
24
|
supportsPlayoutAcknowledgements: true,
|
|
25
|
+
reportsResponseState: true,
|
|
25
26
|
};
|
|
26
27
|
#options;
|
|
27
28
|
constructor(options) {
|
|
@@ -41,10 +42,13 @@ export class GrokVoiceProvider {
|
|
|
41
42
|
});
|
|
42
43
|
return new RealtimeVoiceSession(socket, {
|
|
43
44
|
label: "Grok voice",
|
|
45
|
+
inputTranscriptSettleMs: 300,
|
|
44
46
|
sessionUpdate: this.#sessionUpdate(options, inputSampleRate),
|
|
45
47
|
inputSampleRate,
|
|
46
48
|
outputSampleRate: DEFAULT_SAMPLE_RATE,
|
|
47
49
|
announceTaskSettlement: this.#options.announceTaskSettlement ?? true,
|
|
50
|
+
autoRespondToAudio: this.#options.turnDetection !== null && this.#options.turnDetection?.create_response !== false,
|
|
51
|
+
interruptOnSpeech: this.#options.turnDetection !== null && this.#options.turnDetection?.interrupt_response !== false,
|
|
48
52
|
}, options);
|
|
49
53
|
}
|
|
50
54
|
#sessionUpdate(options, inputSampleRate) {
|
|
@@ -12,10 +12,18 @@ export interface OpenAIRealtimeProviderOptions {
|
|
|
12
12
|
voice?: string;
|
|
13
13
|
/** Sample rate of the PCM16 mono input the harness will send. */
|
|
14
14
|
inputSampleRate?: number;
|
|
15
|
+
/** Input filtering before VAD/model processing. Omitted keeps the API default; null disables it. */
|
|
16
|
+
noiseReduction?: "near_field" | "far_field" | null;
|
|
15
17
|
/** Server-side turn detection. Defaults to semantic VAD; null disables it for push-to-talk. */
|
|
16
18
|
turnDetection?: Record<string, unknown> | null;
|
|
17
|
-
/** Input transcription model; final
|
|
19
|
+
/** Input transcription model; partial/final text surfaces as transcript.delta/transcript. Null disables. */
|
|
18
20
|
transcriptionModel?: string | null;
|
|
21
|
+
/** Optional ISO-639-1 input language hint, e.g. cs or en. Omitted means automatic detection. */
|
|
22
|
+
transcriptionLanguage?: string;
|
|
23
|
+
/** Latency/accuracy tradeoff for gpt-live-transcribe. */
|
|
24
|
+
transcriptionDelay?: "minimal" | "low" | "medium" | "high" | "xhigh";
|
|
25
|
+
/** Realtime 2 reasoning/latency tradeoff. Omitted leaves the model default intact. */
|
|
26
|
+
reasoningEffort?: "minimal" | "low" | "medium" | "high" | "xhigh";
|
|
19
27
|
/** Request a spoken response when a background task settles. */
|
|
20
28
|
announceTaskSettlement?: boolean;
|
|
21
29
|
handshakeTimeoutMs?: number;
|
|
@@ -37,6 +45,7 @@ export declare class OpenAIRealtimeVoiceProvider implements VoiceProvider {
|
|
|
37
45
|
readonly supportsToolCalls: true;
|
|
38
46
|
readonly supportsAsyncContext: true;
|
|
39
47
|
readonly supportsPlayoutAcknowledgements: true;
|
|
48
|
+
readonly reportsResponseState: true;
|
|
40
49
|
};
|
|
41
50
|
constructor(options: OpenAIRealtimeProviderOptions);
|
|
42
51
|
createSession(options: CreateProviderSessionOptions): Promise<VoiceProviderSession>;
|
|
@@ -21,6 +21,7 @@ export class OpenAIRealtimeVoiceProvider {
|
|
|
21
21
|
supportsToolCalls: true,
|
|
22
22
|
supportsAsyncContext: true,
|
|
23
23
|
supportsPlayoutAcknowledgements: true,
|
|
24
|
+
reportsResponseState: true,
|
|
24
25
|
};
|
|
25
26
|
#options;
|
|
26
27
|
constructor(options) {
|
|
@@ -45,6 +46,8 @@ export class OpenAIRealtimeVoiceProvider {
|
|
|
45
46
|
inputSampleRate,
|
|
46
47
|
outputSampleRate: DEFAULT_SAMPLE_RATE,
|
|
47
48
|
announceTaskSettlement: this.#options.announceTaskSettlement ?? true,
|
|
49
|
+
autoRespondToAudio: this.#options.turnDetection !== null && this.#options.turnDetection?.create_response !== false,
|
|
50
|
+
interruptOnSpeech: this.#options.turnDetection !== null && this.#options.turnDetection?.interrupt_response !== false,
|
|
48
51
|
}, options);
|
|
49
52
|
}
|
|
50
53
|
#sessionUpdate(options, inputSampleRate) {
|
|
@@ -60,7 +63,26 @@ export class OpenAIRealtimeVoiceProvider {
|
|
|
60
63
|
input: {
|
|
61
64
|
format: { type: "audio/pcm", rate: inputSampleRate },
|
|
62
65
|
turn_detection: turnDetection,
|
|
63
|
-
...(
|
|
66
|
+
...(this.#options.noiseReduction === undefined
|
|
67
|
+
? {}
|
|
68
|
+
: {
|
|
69
|
+
noise_reduction: this.#options.noiseReduction === null ? null : { type: this.#options.noiseReduction },
|
|
70
|
+
}),
|
|
71
|
+
...(transcriptionModel === null
|
|
72
|
+
? {}
|
|
73
|
+
: {
|
|
74
|
+
transcription: {
|
|
75
|
+
model: transcriptionModel,
|
|
76
|
+
...(this.#options.transcriptionLanguage === undefined
|
|
77
|
+
? {}
|
|
78
|
+
: transcriptionModel.startsWith("gpt-live-transcribe")
|
|
79
|
+
? { languages: [this.#options.transcriptionLanguage] }
|
|
80
|
+
: { language: this.#options.transcriptionLanguage }),
|
|
81
|
+
...(this.#options.transcriptionDelay === undefined
|
|
82
|
+
? {}
|
|
83
|
+
: { delay: this.#options.transcriptionDelay }),
|
|
84
|
+
},
|
|
85
|
+
}),
|
|
64
86
|
},
|
|
65
87
|
output: {
|
|
66
88
|
format: { type: "audio/pcm", rate: DEFAULT_SAMPLE_RATE },
|
|
@@ -71,6 +93,9 @@ export class OpenAIRealtimeVoiceProvider {
|
|
|
71
93
|
};
|
|
72
94
|
if (options.instructions !== undefined)
|
|
73
95
|
session.instructions = options.instructions;
|
|
96
|
+
if (this.#options.reasoningEffort !== undefined) {
|
|
97
|
+
session.reasoning = { effort: this.#options.reasoningEffort };
|
|
98
|
+
}
|
|
74
99
|
return session;
|
|
75
100
|
}
|
|
76
101
|
}
|
|
@@ -15,6 +15,12 @@ export interface RealtimeWireOptions {
|
|
|
15
15
|
inputSampleRate: number;
|
|
16
16
|
outputSampleRate: number;
|
|
17
17
|
announceTaskSettlement: boolean;
|
|
18
|
+
/** VAD will create a response after the user's speech; coalesce tool continuations into it. */
|
|
19
|
+
autoRespondToAudio: boolean;
|
|
20
|
+
/** Follow the provider's VAD interruption setting. Defaults to true. */
|
|
21
|
+
interruptOnSpeech?: boolean;
|
|
22
|
+
/** xAI sends revisions through its completed event; other providers settle immediately. */
|
|
23
|
+
inputTranscriptSettleMs?: number;
|
|
18
24
|
}
|
|
19
25
|
export declare class RealtimeVoiceSession implements VoiceProviderSession {
|
|
20
26
|
#private;
|