@zerowidth/workbench-sdk 2.2.0 → 2.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -48,7 +48,7 @@ import Workbench from '@zerowidth/workbench-sdk';
48
48
  // Create engine instance by passing the location of your configured flow
49
49
  const engine = await Workbench.create('./path/to/myflow.zwf', {
50
50
  keys: {
51
- openrouter: process.env.OPENROUTER_API_KEY
51
+ inference: process.env.INFERENCE_API_KEY // your LLM endpoint's key
52
52
  }
53
53
  });
54
54
 
@@ -281,14 +281,36 @@ The engine supports secure API key management for nodes that require external se
281
281
  ```javascript
282
282
  const engine = await Workbench.create(flow, {
283
283
  keys: {
284
- openrouter: "sk-...", // OpenRouter API key
284
+ inference: "sk-...", // key for the LLM endpoint (default: OpenRouter)
285
285
  }
286
286
  });
287
287
  ```
288
288
 
289
+ ### Pointing at your own LLM
290
+
291
+ Every OpenAI-compatible LLM node runs against a single inference endpoint.
292
+ By default that's OpenRouter, but you can point it at **your own**
293
+ OpenAI-compatible endpoint (vLLM, Ollama, TGI, a LiteLLM proxy, or any
294
+ gateway) — nothing leaves your infrastructure:
295
+
296
+ ```javascript
297
+ const engine = await Workbench.create(flow, {
298
+ keys: { inference: process.env.MY_LLM_KEY },
299
+ inferenceBaseURL: "https://llm.my-internal-host/v1",
300
+ });
301
+ ```
302
+
303
+ The model ids in the flow (e.g. `openai/gpt-4o`) must be ones your endpoint
304
+ serves — a LiteLLM `model_list` mapping is the usual way to line these up.
305
+ A custom `inferenceBaseURL` sends plain OpenAI-shaped requests (the
306
+ OpenRouter-only payload fields are dropped automatically).
307
+
308
+ > **Aliases:** `keys.openrouter` and `openrouterBaseURL` are still accepted
309
+ > as backward-compatible aliases for `keys.inference` and `inferenceBaseURL`.
310
+
289
311
  ### Node Key Requirements
290
312
 
291
- Nodes specify their key requirements in their configuration. All LLMs are configured by default to use OpenRouter, but this can be overridden.
313
+ Nodes specify their key requirements in their configuration. All LLMs are configured by default to use the inference endpoint (OpenRouter unless overridden), but this can be overridden.
292
314
 
293
315
  ```json
294
316
  {
@@ -0,0 +1,158 @@
1
+ {
2
+ "display_name": "Custom Inference",
3
+ "tagline": "Generate text on a bring-your-own endpoint",
4
+ "description": "Runs a chat completion against a host-registered custom inference provider (an OpenAI-compatible or Azure OpenAI endpoint). The host resolves `settings.provider` to a `custom:<provider>` integration and sends `settings.model` as the model. Used by ZeroWidth's BYO inference endpoints (ADR 0048 Phase 3); the host rewrites a `byo:<provider>:<model>` node into this one at run time.",
5
+ "category": "llm",
6
+ "accepts_plugins": true,
7
+ "settings": [
8
+ {
9
+ "name": "provider",
10
+ "display_name": "Provider",
11
+ "type": "string",
12
+ "description": "Name of the host-registered custom inference provider.",
13
+ "default": null
14
+ },
15
+ {
16
+ "name": "model",
17
+ "display_name": "Model",
18
+ "type": "string",
19
+ "description": "Model / deployment id the endpoint serves.",
20
+ "default": null
21
+ }
22
+ ],
23
+ "inputs": [
24
+ {
25
+ "name": "messages",
26
+ "display_name": "Conversation",
27
+ "type": "conversation or message or string",
28
+ "description": "Array of chat messages that make up the conversation",
29
+ "required": true
30
+ },
31
+ {
32
+ "name": "system_prompt",
33
+ "display_name": "System Message",
34
+ "type": "message or string",
35
+ "description": "System prompt to instruct the model",
36
+ "default": null
37
+ },
38
+ {
39
+ "name": "tools",
40
+ "display_name": "Tools",
41
+ "type": "tool or array of tools",
42
+ "description": "Array of tools to use",
43
+ "default": null,
44
+ "allow_multiple": true
45
+ },
46
+ {
47
+ "name": "tool_choice",
48
+ "display_name": "Tool Choice",
49
+ "type": "string",
50
+ "description": "Tool selection control",
51
+ "default": null
52
+ },
53
+ {
54
+ "name": "response_format",
55
+ "display_name": "Response Format",
56
+ "type": "object",
57
+ "description": "Output format specification",
58
+ "default": null
59
+ },
60
+ {
61
+ "name": "stop",
62
+ "display_name": "Stop",
63
+ "type": "string or array",
64
+ "description": "Custom stop sequences",
65
+ "default": null
66
+ },
67
+ {
68
+ "name": "temperature",
69
+ "display_name": "Temperature",
70
+ "type": "number",
71
+ "description": "Controls randomness (0-2)",
72
+ "default": null
73
+ },
74
+ {
75
+ "name": "top_p",
76
+ "display_name": "Top P",
77
+ "type": "number",
78
+ "description": "Controls diversity via nucleus sampling",
79
+ "default": null
80
+ },
81
+ {
82
+ "name": "max_tokens",
83
+ "display_name": "Max Tokens",
84
+ "type": "number",
85
+ "description": "Maximum tokens to generate",
86
+ "default": null
87
+ },
88
+ {
89
+ "name": "frequency_penalty",
90
+ "display_name": "Frequency Penalty",
91
+ "type": "number",
92
+ "description": "Reduces repetition (-2 to 2)",
93
+ "default": null
94
+ },
95
+ {
96
+ "name": "presence_penalty",
97
+ "display_name": "Presence Penalty",
98
+ "type": "number",
99
+ "description": "Encourages new topics (-2 to 2)",
100
+ "default": null
101
+ },
102
+ {
103
+ "name": "seed",
104
+ "display_name": "Seed",
105
+ "type": "number",
106
+ "description": "Deterministic outputs",
107
+ "default": null
108
+ }
109
+ ],
110
+ "outputs": [
111
+ {
112
+ "name": "conversation",
113
+ "display_name": "Conversation",
114
+ "type": "conversation",
115
+ "can_stream": true,
116
+ "description": "An array of messages including any tool call & response messages as well as the final generated output."
117
+ },
118
+ {
119
+ "name": "content",
120
+ "display_name": "Content",
121
+ "can_stream": true,
122
+ "type": "string",
123
+ "description": "The content portion of the final generated response message."
124
+ },
125
+ {
126
+ "name": "message",
127
+ "display_name": "Final Message",
128
+ "type": "message",
129
+ "can_stream": true,
130
+ "description": "The final generated response message."
131
+ },
132
+ {
133
+ "name": "role",
134
+ "display_name": "Role",
135
+ "can_stream": true,
136
+ "type": "string",
137
+ "description": "Role of the response (usually 'assistant')"
138
+ },
139
+ {
140
+ "name": "tool_calls",
141
+ "display_name": "Tool Calls",
142
+ "type": "array of tools",
143
+ "description": "Tool calls made by the model"
144
+ },
145
+ {
146
+ "name": "usage",
147
+ "display_name": "Token Usage",
148
+ "type": "object",
149
+ "description": "Token usage statistics"
150
+ },
151
+ {
152
+ "name": "finish_reason",
153
+ "display_name": "Finish Reason",
154
+ "type": "string",
155
+ "description": "Why the completion finished"
156
+ }
157
+ ]
158
+ }
@@ -0,0 +1,117 @@
1
+ export default async ({inputs, settings, config, nodeConfig}) => {
2
+ try {
3
+ // Resolve the host-registered custom inference provider (ADR 0048
4
+ // Phase 3). `settings.provider` names it; the host built a
5
+ // `custom:<provider>` integration in loadIntegrations from
6
+ // config.customInferenceProviders. `settings.model` is the model /
7
+ // Azure deployment the endpoint serves.
8
+ const providerName = settings?.provider;
9
+ const model = settings?.model;
10
+ if (!providerName || !model) {
11
+ throw new Error("Custom Inference node requires provider + model settings");
12
+ }
13
+ const integration = config.integrations?.['custom:' + providerName];
14
+ if (!integration) {
15
+ throw new Error(`Custom inference provider "${providerName}" is not configured`);
16
+ }
17
+
18
+ let messages = inputs.messages;
19
+
20
+ if(typeof messages === 'string') {
21
+ messages = [{ role: 'user', content: messages }];
22
+ }
23
+
24
+ if(typeof messages === 'object' && !Array.isArray(messages)) {
25
+ messages = [messages];
26
+ }
27
+
28
+ if(inputs.system_prompt) {
29
+ let systemPrompt = inputs.system_prompt;
30
+ if(typeof systemPrompt === 'string') {
31
+ systemPrompt = { role: 'system', content: systemPrompt };
32
+ }
33
+ messages = [systemPrompt, ...messages];
34
+ }
35
+
36
+ // Build parameters object from the wired inputs (everything but the
37
+ // conversation itself), mirroring the platform LLM nodes.
38
+ const params = {};
39
+ for (const input of (nodeConfig?.inputs || [])) {
40
+ if (input.name === 'messages') continue;
41
+ const value = inputs[input.name];
42
+ if (value !== null && value !== undefined) {
43
+ if (input.name === 'tools' && Array.isArray(value)) {
44
+ params.tools = value.flat();
45
+ } else {
46
+ params[input.name] = value;
47
+ }
48
+ }
49
+ }
50
+
51
+ const response = await integration.chatCompletion({
52
+ model,
53
+ messages,
54
+ ...params
55
+ }, nodeConfig, config);
56
+
57
+ // Conversation output: keep only internal-tool history + the fresh
58
+ // response — identical to the platform chat nodes.
59
+ const hasInternalToolTracking = config.internal_tool_names !== undefined;
60
+ const internalToolNames = new Set(config.internal_tool_names || []);
61
+
62
+ let conversationMessages = [];
63
+ if (Array.isArray(messages) && messages.length > 0) {
64
+ for (let i = messages.length - 1; i >= 0; i--) {
65
+ const msg = messages[i];
66
+ if (!msg || typeof msg !== 'object') continue;
67
+ const isTool = msg.role === 'tool';
68
+ const hasToolCalls = msg.tool_calls && Array.isArray(msg.tool_calls) && msg.tool_calls.length > 0;
69
+ if (isTool) {
70
+ const toolName = msg.name;
71
+ if (!hasInternalToolTracking || internalToolNames.has(toolName)) {
72
+ conversationMessages.unshift(msg);
73
+ }
74
+ } else if (hasToolCalls) {
75
+ const internalCalls = !hasInternalToolTracking
76
+ ? msg.tool_calls
77
+ : msg.tool_calls.filter(tc => internalToolNames.has(tc.function?.name));
78
+ if (internalCalls.length > 0) {
79
+ conversationMessages.unshift({ ...msg, tool_calls: internalCalls });
80
+ }
81
+ } else {
82
+ break;
83
+ }
84
+ }
85
+ }
86
+
87
+ const finalMessage = { content: response.content, role: response.role };
88
+ if (response.tool_calls && Array.isArray(response.tool_calls) && response.tool_calls.length > 0) {
89
+ finalMessage.tool_calls = response.tool_calls;
90
+ }
91
+ if (response.images) {
92
+ finalMessage.images = response.images;
93
+ }
94
+ conversationMessages.push(finalMessage);
95
+
96
+ return {
97
+ conversation: conversationMessages,
98
+ message: {
99
+ content: response.content,
100
+ role: response.role,
101
+ tool_calls: response.tool_calls
102
+ },
103
+ content: response.content,
104
+ role: response.role,
105
+ tool_calls: response.tool_calls,
106
+ annotations: response.annotations,
107
+ citations: response.citations,
108
+ logprobs: response.logprobs,
109
+ finish_reason: response.finish_reason,
110
+ usage: response.usage,
111
+ cost_total: response.cost_total,
112
+ cost_itemized: response.cost_itemized
113
+ };
114
+ } catch (error) {
115
+ throw new Error(`Custom Inference node error: ${error.message}`);
116
+ }
117
+ };
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@zerowidth/workbench-sdk",
3
- "version": "2.2.0",
3
+ "version": "2.3.0",
4
4
  "dependencies": {
5
5
  "adm-zip": "^0.5.16",
6
6
  "ajv": "^8.17.1",
@@ -11,7 +11,11 @@
11
11
  "tiktoken": "^1.0.22",
12
12
  "uuid": "^11.1.0"
13
13
  },
14
- "repository": "https://github.com/zerowidth-ai/workbench-sdk",
14
+ "repository": {
15
+ "type": "git",
16
+ "url": "https://github.com/zerowidth-ai/workbench-sdk.git",
17
+ "directory": "sdks/nodejs"
18
+ },
15
19
  "description": "Workbench SDK \u2014 execute Workbench-authored flows in production. Successor to the zv1 package; see https://zerowidth.ai for the design surface.",
16
20
  "main": "./src/index.js",
17
21
  "files": [
@@ -34,7 +38,8 @@
34
38
  "test-nodes": "node tests/test.all-nodes.js",
35
39
  "test-flows": "node tests/test.flows.js",
36
40
  "test-flow": "node tests/test.flows.js",
37
- "test-kb": "node tests/test.kb-search.js && node tests/test.kb-graph.js"
41
+ "test-kb": "node tests/test.kb-search.js && node tests/test.kb-graph.js",
42
+ "test-custom-inference": "node tests/test.custom-inference.js"
38
43
  },
39
44
  "author": "Peter Binggeser @ ZeroWidth, LLC",
40
45
  "license": "Apache-2.0",
@@ -1,18 +1,38 @@
1
- import OpenAI from 'openai';
1
+ import OpenAI, { AzureOpenAI } from 'openai';
2
2
  import { emitAPICallEvent } from '../utilities/sanitizeAPICall.js';
3
3
 
4
4
  export default class OpenRouterIntegration {
5
5
  constructor(apiKey, options = {}) {
6
-
7
- this.client = new OpenAI({
8
- baseURL: options.baseURL || 'https://openrouter.ai/api/v1',
9
- apiKey: apiKey,
10
- defaultHeaders: {
11
- 'Content-Type': 'application/json',
12
- 'HTTP-Referer': options.referer || 'https://workbench.zerowidth.ai',
13
- 'X-Title': options.title || 'Workbench by ZeroWidth'
14
- }
15
- });
6
+ // dialect: 'openrouter' (default — the platform endpoint, which
7
+ // accepts OpenRouter-specific payload extensions), or a custom
8
+ // OpenAI-compatible endpoint: 'openai' (vLLM/Ollama/TGI/gateways)
9
+ // or 'azure' (Azure OpenAI — deployment routing + api-version).
10
+ // See ADR 0048 Phase 3 in the zerowidth monorepo.
11
+ this.dialect = options.dialect || 'openrouter';
12
+
13
+ const defaultHeaders = {
14
+ 'Content-Type': 'application/json',
15
+ 'HTTP-Referer': options.referer || 'https://workbench.zerowidth.ai',
16
+ 'X-Title': options.title || 'Workbench by ZeroWidth'
17
+ };
18
+
19
+ if (this.dialect === 'azure') {
20
+ // Azure OpenAI: the model name is the deployment; auth is the
21
+ // `api-key` header + a required `api-version`. AzureOpenAI wires
22
+ // all three from these options.
23
+ this.client = new AzureOpenAI({
24
+ endpoint: options.baseURL,
25
+ apiKey: apiKey,
26
+ apiVersion: options.apiVersion || '2024-10-21',
27
+ defaultHeaders
28
+ });
29
+ } else {
30
+ this.client = new OpenAI({
31
+ baseURL: options.baseURL || 'https://openrouter.ai/api/v1',
32
+ apiKey: apiKey,
33
+ defaultHeaders
34
+ });
35
+ }
16
36
  }
17
37
 
18
38
  async chatCompletion(params, nodeConfig = null, engineConfig = null) {
@@ -24,16 +44,20 @@ export default class OpenRouterIntegration {
24
44
  } = params;
25
45
 
26
46
  // Base payload with required fields
27
- const payload = {
28
- model,
29
- provider: {
47
+ const payload = { model };
48
+ // OpenRouter-only extensions: the `provider` routing directive and
49
+ // usage accounting. A custom OpenAI-compatible / Azure endpoint
50
+ // doesn't understand these and strict servers (Azure) reject
51
+ // unknown fields — so only send them to the platform endpoint.
52
+ if (this.dialect === 'openrouter') {
53
+ payload.provider = {
30
54
  data_collection: "deny",
31
55
  require_parameters: true,
32
- },
56
+ };
33
57
  // Request OpenRouter usage accounting so the response carries the
34
58
  // authoritative cost (usage.cost). buildCostData prefers it.
35
- usage: { include: true }
36
- };
59
+ payload.usage = { include: true };
60
+ }
37
61
 
38
62
  // Add messages or prompt (required)
39
63
  if (messages) {
@@ -112,13 +112,27 @@ export async function loadNodes(flow) {
112
112
  export async function loadIntegrations(config, flow = null) {
113
113
  const integrations = {};
114
114
 
115
- // Load OpenRouter integration if API key is provided
116
- if (config.keys?.openrouter) {
115
+ // Load the primary LLM integration (OpenAI-compatible chat/completions).
116
+ // Public config is provider-agnostic — `inferenceBaseURL` + `keys.inference`
117
+ // — so a self-hosted host can point the engine at its own endpoint without
118
+ // any "openrouter" branding. `openrouterBaseURL` + `keys.openrouter` remain
119
+ // as backward-compatible aliases (the platform default endpoint stays
120
+ // OpenRouter). The integration is still registered under `openrouter` for
121
+ // the LLM nodes that look it up by that name.
122
+ const inferenceKey = config.keys?.inference ?? config.keys?.openrouter;
123
+ if (inferenceKey) {
117
124
  const openrouterPath = path.join(getDirname(import.meta.url), '../integrations', 'openrouter.js');
125
+ const inferenceBaseURL = config.inferenceBaseURL || config.openrouterBaseURL || 'https://openrouter.ai/api/v1';
126
+ // Only the real OpenRouter endpoint understands the OpenRouter-only
127
+ // payload extensions (provider routing, usage accounting). A custom
128
+ // base URL is a plain OpenAI-compatible endpoint, so use the 'openai'
129
+ // dialect there and don't send those fields (see openrouter.js).
130
+ const isOpenRouterEndpoint = /^https?:\/\/openrouter\.ai(\/|$)/i.test(inferenceBaseURL);
118
131
  try {
119
132
  const OpenRouterIntegration = await import(openrouterPath).then(module => module.default);
120
- integrations.openrouter = new OpenRouterIntegration(config.keys.openrouter, {
121
- baseURL: config.openrouterBaseURL || 'https://openrouter.ai/api/v1',
133
+ integrations.openrouter = new OpenRouterIntegration(inferenceKey, {
134
+ baseURL: inferenceBaseURL,
135
+ dialect: isOpenRouterEndpoint ? 'openrouter' : 'openai',
122
136
  referer: 'https://workbench.zerowidth.ai',
123
137
  title: 'Workbench by ZeroWidth'
124
138
  });
@@ -126,7 +140,28 @@ export async function loadIntegrations(config, flow = null) {
126
140
  throw error;
127
141
  }
128
142
  }
129
-
143
+
144
+ // Bring-your-own inference endpoints (ADR 0048 Phase 3). The host passes
145
+ // `customInferenceProviders: { <name>: { baseURL, apiKey, dialect,
146
+ // apiVersion?, models? } }`. Each becomes an integration keyed
147
+ // `custom:<name>`, reusing the OpenRouter integration's completion logic
148
+ // with a custom destination + dialect. The `custom-inference` node
149
+ // resolves `custom:<provider>` by its settings.
150
+ if (config.customInferenceProviders && typeof config.customInferenceProviders === 'object') {
151
+ const openrouterPath = path.join(getDirname(import.meta.url), '../integrations', 'openrouter.js');
152
+ const OpenRouterIntegration = await import(openrouterPath).then(module => module.default);
153
+ for (const [name, provider] of Object.entries(config.customInferenceProviders)) {
154
+ if (!provider?.baseURL || !provider?.apiKey) continue;
155
+ integrations['custom:' + name] = new OpenRouterIntegration(provider.apiKey, {
156
+ baseURL: provider.baseURL,
157
+ dialect: provider.dialect === 'azure' ? 'azure' : 'openai',
158
+ apiVersion: provider.apiVersion,
159
+ referer: 'https://workbench.zerowidth.ai',
160
+ title: 'Workbench by ZeroWidth'
161
+ });
162
+ }
163
+ }
164
+
130
165
  // Load knowledge base integration if available
131
166
  const knowledgeBaseType = config.knowledgeBase?.type || 'sqlite';
132
167
  const knowledgeBaseConfig = config.knowledgeBase || {};