@zerowidth/workbench-sdk 2.5.0 → 2.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/nodes/system-prompt/system-prompt.process.js +28 -19
- package/nodes/truncate-old-tool-responses/truncate-old-tool-responses.config.json +8 -0
- package/nodes/truncate-old-tool-responses/truncate-old-tool-responses.process.js +4 -1
- package/package.json +1 -1
- package/src/integrations/openrouter.js +47 -4
- package/src/utilities/promptCache.js +142 -0
|
@@ -56,30 +56,39 @@ export default async ({inputs, settings, config}) => {
|
|
|
56
56
|
// Combine chained content with base content
|
|
57
57
|
let fullContent = chainedContent ? `${chainedContent}\n\n${baseContent}` : baseContent;
|
|
58
58
|
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
{
|
|
64
|
-
|
|
65
|
-
text: fullContent
|
|
59
|
+
const fill = (text) =>
|
|
60
|
+
text.replace(/\{\{(.*?)\}\}/g, (match, p1) => {
|
|
61
|
+
// look for a variable with the key p1
|
|
62
|
+
let variable = variables.find(variable => Object.keys(variable).find(key => key === p1));
|
|
63
|
+
if(variable) {
|
|
64
|
+
return renderVariable(variable[p1]);
|
|
66
65
|
}
|
|
67
|
-
|
|
68
|
-
|
|
66
|
+
return match;
|
|
67
|
+
});
|
|
69
68
|
|
|
70
|
-
//
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
69
|
+
// Values filled in can change from run to run (the time, memory, search
|
|
70
|
+
// results); the text before the first of them doesn't. The block records
|
|
71
|
+
// that length as `cache_prefix_length`, so the model client can cache
|
|
72
|
+
// the prompt up to there for a model that caches on request. The client
|
|
73
|
+
// removes the hint before any model sees it.
|
|
74
|
+
let firstFilled = -1;
|
|
75
|
+
for (const m of fullContent.matchAll(/\{\{(.*?)\}\}/g)) {
|
|
76
|
+
if (variables.some(variable => Object.keys(variable).includes(m[1]))) {
|
|
77
|
+
firstFilled = m.index;
|
|
78
|
+
break;
|
|
76
79
|
}
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
const text = firstFilled < 0 ? fullContent : fullContent.slice(0, firstFilled) + fill(fullContent.slice(firstFilled));
|
|
83
|
+
const prefix = firstFilled < 0 ? text.length : firstFilled;
|
|
84
|
+
const message = {
|
|
85
|
+
role: "system",
|
|
86
|
+
content: [{ type: "text", text, ...(text.slice(0, prefix).trim() ? { cache_prefix_length: prefix } : {}) }],
|
|
87
|
+
};
|
|
88
|
+
|
|
80
89
|
// Return the message and string prompt
|
|
81
90
|
return {
|
|
82
91
|
message: message,
|
|
83
|
-
prompt:
|
|
92
|
+
prompt: text
|
|
84
93
|
};
|
|
85
94
|
};
|
|
@@ -20,6 +20,14 @@
|
|
|
20
20
|
"required": false,
|
|
21
21
|
"default": 20
|
|
22
22
|
},
|
|
23
|
+
{
|
|
24
|
+
"name": "step",
|
|
25
|
+
"display_name": "Step",
|
|
26
|
+
"type": "number",
|
|
27
|
+
"description": "Move the cutoff only in steps of this many messages, so earlier messages stay the same from turn to turn and a model's prompt cache keeps matching them. 1 moves it every message.",
|
|
28
|
+
"required": false,
|
|
29
|
+
"default": 1
|
|
30
|
+
},
|
|
23
31
|
{
|
|
24
32
|
"name": "placeholder",
|
|
25
33
|
"display_name": "Placeholder",
|
|
@@ -1,14 +1,17 @@
|
|
|
1
1
|
export default async ({ inputs, settings, config }) => {
|
|
2
2
|
const messages = inputs.messages;
|
|
3
3
|
const keepRecent = Math.max(0, Math.floor(Number(inputs.keep_recent ?? 20)));
|
|
4
|
+
const step = Math.max(1, Math.floor(Number(inputs.step ?? 1)) || 1);
|
|
4
5
|
const placeholder = inputs.placeholder ?? "[Truncated]";
|
|
5
6
|
|
|
6
7
|
if (!Array.isArray(messages)) {
|
|
7
8
|
throw new Error("Messages input must be an array");
|
|
8
9
|
}
|
|
9
10
|
|
|
11
|
+
// The cutoff moves in whole steps, so between steps the earlier
|
|
12
|
+
// messages are byte-for-byte what they were last turn (cacheable).
|
|
10
13
|
const total = messages.length;
|
|
11
|
-
const cutoffIndex = total - keepRecent;
|
|
14
|
+
const cutoffIndex = Math.floor(Math.max(0, total - keepRecent) / step) * step;
|
|
12
15
|
|
|
13
16
|
const result = [];
|
|
14
17
|
let truncatedCount = 0;
|
package/package.json
CHANGED
|
@@ -1,5 +1,19 @@
|
|
|
1
1
|
import OpenAI, { AzureOpenAI } from 'openai';
|
|
2
2
|
import { emitAPICallEvent } from '../utilities/sanitizeAPICall.js';
|
|
3
|
+
import { applyPromptCache, stripCacheHints } from '../utilities/promptCache.js';
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* Prompt-cache counts from a usage block, in OpenAI's shape
|
|
7
|
+
* (`prompt_tokens_details`), which OpenRouter normalizes every provider
|
|
8
|
+
* to. Both are part of `prompt_tokens`, not on top of it.
|
|
9
|
+
*/
|
|
10
|
+
export function readCacheUsage(usage) {
|
|
11
|
+
const details = usage?.prompt_tokens_details || {};
|
|
12
|
+
return {
|
|
13
|
+
cached_tokens: Number(details.cached_tokens) || 0,
|
|
14
|
+
cache_write_tokens: Number(details.cache_write_tokens) || 0,
|
|
15
|
+
};
|
|
16
|
+
}
|
|
3
17
|
|
|
4
18
|
export default class OpenRouterIntegration {
|
|
5
19
|
constructor(apiKey, options = {}) {
|
|
@@ -72,6 +86,15 @@ export default class OpenRouterIntegration {
|
|
|
72
86
|
return clean;
|
|
73
87
|
});
|
|
74
88
|
|
|
89
|
+
// Prompt-cache marks for models that need them (promptCache.js).
|
|
90
|
+
// Only the platform endpoint understands them; the SDK's own
|
|
91
|
+
// cache flags are removed for every endpoint.
|
|
92
|
+
const cacheConfig = engineConfig || this._engineConfig;
|
|
93
|
+
payload.messages = applyPromptCache(payload.messages, {
|
|
94
|
+
model,
|
|
95
|
+
enabled: this.dialect === 'openrouter' && cacheConfig?.promptCache !== false,
|
|
96
|
+
});
|
|
97
|
+
|
|
75
98
|
} else if (prompt) {
|
|
76
99
|
payload.prompt = prompt;
|
|
77
100
|
} else {
|
|
@@ -151,7 +174,11 @@ export default class OpenRouterIntegration {
|
|
|
151
174
|
let usage = {
|
|
152
175
|
prompt_tokens: 0,
|
|
153
176
|
completion_tokens: 0,
|
|
154
|
-
total_tokens: 0
|
|
177
|
+
total_tokens: 0,
|
|
178
|
+
// Of prompt_tokens: read from the provider's prompt cache, and
|
|
179
|
+
// written to it. Both 0 when the provider reports neither.
|
|
180
|
+
cached_tokens: 0,
|
|
181
|
+
cache_write_tokens: 0
|
|
155
182
|
}
|
|
156
183
|
|
|
157
184
|
// Authoritative cost from OpenRouter usage accounting (final usage chunk)
|
|
@@ -240,6 +267,9 @@ export default class OpenRouterIntegration {
|
|
|
240
267
|
usage.prompt_tokens += chunk.usage.prompt_tokens || 0;
|
|
241
268
|
usage.completion_tokens += chunk.usage.completion_tokens || 0;
|
|
242
269
|
usage.total_tokens += chunk.usage.total_tokens || 0;
|
|
270
|
+
const cache = readCacheUsage(chunk.usage);
|
|
271
|
+
usage.cached_tokens += cache.cached_tokens;
|
|
272
|
+
usage.cache_write_tokens += cache.cache_write_tokens;
|
|
243
273
|
if (typeof chunk.usage.cost === 'number') apiCost = chunk.usage.cost;
|
|
244
274
|
if (chunk.usage.cost_details) apiCostDetails = chunk.usage.cost_details;
|
|
245
275
|
}
|
|
@@ -413,7 +443,9 @@ export default class OpenRouterIntegration {
|
|
|
413
443
|
const data = await res.json();
|
|
414
444
|
const choice = data.choices?.[0];
|
|
415
445
|
const message = choice?.message || {};
|
|
416
|
-
const usage = data.usage
|
|
446
|
+
const usage = data.usage
|
|
447
|
+
? { ...data.usage, ...readCacheUsage(data.usage) }
|
|
448
|
+
: { prompt_tokens: 0, completion_tokens: 0, total_tokens: 0, cached_tokens: 0, cache_write_tokens: 0 };
|
|
417
449
|
|
|
418
450
|
let costData = null;
|
|
419
451
|
if (nodeConfig) {
|
|
@@ -547,10 +579,20 @@ export default class OpenRouterIntegration {
|
|
|
547
579
|
outputCost = 0;
|
|
548
580
|
}
|
|
549
581
|
|
|
582
|
+
// How much of the input came from (or went into) the provider's
|
|
583
|
+
// prompt cache. The total above already reflects what that cost.
|
|
584
|
+
const cachedTokens = usage.cached_tokens || 0;
|
|
585
|
+
const cacheWriteTokens = usage.cache_write_tokens || 0;
|
|
550
586
|
return {
|
|
551
587
|
totalCost: Number(totalCost.toFixed(8)),
|
|
552
588
|
itemizedCosts: [
|
|
553
|
-
{
|
|
589
|
+
{
|
|
590
|
+
label: "Input Tokens",
|
|
591
|
+
cost: Number(inputCost.toFixed(8)),
|
|
592
|
+
tokens: promptTokens,
|
|
593
|
+
...(cachedTokens > 0 && { cached_tokens: cachedTokens }),
|
|
594
|
+
...(cacheWriteTokens > 0 && { cache_write_tokens: cacheWriteTokens })
|
|
595
|
+
},
|
|
554
596
|
{ label: "Output Tokens", cost: Number(outputCost.toFixed(8)), tokens: completionTokens }
|
|
555
597
|
]
|
|
556
598
|
};
|
|
@@ -803,7 +845,8 @@ export default class OpenRouterIntegration {
|
|
|
803
845
|
throw new Error('questions must be a non-empty object mapping question ids to { type, instructions, criteria }');
|
|
804
846
|
}
|
|
805
847
|
|
|
806
|
-
|
|
848
|
+
// A conversation's cache hints are for chat models only.
|
|
849
|
+
const payload = { model, state: stripCacheHints(state), questions };
|
|
807
850
|
|
|
808
851
|
const url = `${this.client.baseURL}/systemone`;
|
|
809
852
|
const headers = {
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Prompt caching for models that need to be told where to cache.
|
|
3
|
+
*
|
|
4
|
+
* Anthropic models cache a request's prefix only up to blocks marked
|
|
5
|
+
* `cache_control: { type: "ephemeral" }`, at most four per request.
|
|
6
|
+
* Everything before a mark (tools, then the system prompt, then the
|
|
7
|
+
* messages) is reused by the next call that starts the same way, read
|
|
8
|
+
* at a fraction of the input price. Other providers OpenRouter serves
|
|
9
|
+
* cache a repeated prefix on their own.
|
|
10
|
+
*
|
|
11
|
+
* Where marks go, for a model that takes them:
|
|
12
|
+
* - the end of the system prompt's fixed part. The System Prompt node
|
|
13
|
+
* records it as `cache_prefix_length` on its text block (the text
|
|
14
|
+
* before the first variable it fills in); the block is split there
|
|
15
|
+
* and the first part marked, so a value that changes per run (the
|
|
16
|
+
* time, memory, search results) doesn't spoil the cache;
|
|
17
|
+
* - marks the caller placed itself (`cache_control` on a block);
|
|
18
|
+
* - the last message, whatever its role, so everything so far
|
|
19
|
+
* (earlier turns, and this turn's tool calls and results) is reused
|
|
20
|
+
* by the next call.
|
|
21
|
+
* At most four: the earliest three are kept, then the last message.
|
|
22
|
+
* A mark that comes before a one-hour mark gets the one-hour lifetime
|
|
23
|
+
* too, since Anthropic needs longer-lived marks to come first.
|
|
24
|
+
*
|
|
25
|
+
* For any other model, or with caching off, the hint and every mark are
|
|
26
|
+
* removed, and the system prompt goes out exactly as it was written.
|
|
27
|
+
*/
|
|
28
|
+
|
|
29
|
+
const MAX_MARKS = 4;
|
|
30
|
+
const MARK = { type: "ephemeral" };
|
|
31
|
+
|
|
32
|
+
/** Models that cache only where they're told to. */
|
|
33
|
+
export function takesCacheMarks(model) {
|
|
34
|
+
return typeof model === "string" && model.startsWith("anthropic/");
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
const isText = (b) => b && typeof b === "object" && b.type === "text" && typeof b.text === "string";
|
|
38
|
+
|
|
39
|
+
function strip(block) {
|
|
40
|
+
if (!block || typeof block !== "object") return block;
|
|
41
|
+
if (!("cache_control" in block) && !("cache_prefix_length" in block)) return block;
|
|
42
|
+
const { cache_control, cache_prefix_length, ...rest } = block;
|
|
43
|
+
return rest;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
/** One message's blocks, with the System Prompt node's hint turned into
|
|
47
|
+
* a split and a mark. */
|
|
48
|
+
function splitAtPrefix(blocks) {
|
|
49
|
+
return blocks.flatMap((b) => {
|
|
50
|
+
if (!isText(b) || typeof b.cache_prefix_length !== "number") return [b];
|
|
51
|
+
const { cache_prefix_length: at, ...block } = b;
|
|
52
|
+
if (at <= 0 || !b.text.slice(0, at).trim()) return [block];
|
|
53
|
+
if (at >= b.text.length) return [{ ...block, cache_control: block.cache_control ?? MARK }];
|
|
54
|
+
return [
|
|
55
|
+
{ type: "text", text: b.text.slice(0, at), cache_control: MARK },
|
|
56
|
+
{ ...block, text: b.text.slice(at) },
|
|
57
|
+
];
|
|
58
|
+
});
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
const stripMessage = (m) =>
|
|
62
|
+
m && typeof m === "object" && Array.isArray(m.content) ? { ...m, content: m.content.map(strip) } : m;
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* A message or conversation with every cache mark and hint removed, for
|
|
66
|
+
* anything that sends one somewhere other than a chat model (a decision
|
|
67
|
+
* model's `state`). Anything else passes through. Never mutates the input.
|
|
68
|
+
*/
|
|
69
|
+
export function stripCacheHints(value) {
|
|
70
|
+
return Array.isArray(value) ? value.map(stripMessage) : stripMessage(value);
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/**
|
|
74
|
+
* The messages to send, with cache marks placed for a model that takes
|
|
75
|
+
* them and removed for any other. Never mutates the input.
|
|
76
|
+
*/
|
|
77
|
+
export function applyPromptCache(messages, { model, enabled = true } = {}) {
|
|
78
|
+
if (!Array.isArray(messages)) return messages;
|
|
79
|
+
const marking = enabled && takesCacheMarks(model);
|
|
80
|
+
|
|
81
|
+
if (!marking) {
|
|
82
|
+
return messages.map(stripMessage);
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
// Split the hinted blocks, then keep the earliest three marks.
|
|
86
|
+
let kept = 0;
|
|
87
|
+
const out = messages.map((m) => {
|
|
88
|
+
if (!m || !Array.isArray(m.content)) return m;
|
|
89
|
+
const blocks = splitAtPrefix(m.content).map((b) => {
|
|
90
|
+
if (!b || !b.cache_control) return b;
|
|
91
|
+
// A mark on an empty block is refused by the provider.
|
|
92
|
+
if (kept < MAX_MARKS - 1 && !(isText(b) && !b.text)) {
|
|
93
|
+
kept++;
|
|
94
|
+
return b;
|
|
95
|
+
}
|
|
96
|
+
return strip(b);
|
|
97
|
+
});
|
|
98
|
+
return { ...m, content: blocks };
|
|
99
|
+
});
|
|
100
|
+
|
|
101
|
+
// The last message's last text block.
|
|
102
|
+
const last = out.length - 1;
|
|
103
|
+
const m = out[last];
|
|
104
|
+
if (m && m.role !== "system") {
|
|
105
|
+
const blocks =
|
|
106
|
+
typeof m.content === "string"
|
|
107
|
+
? [{ type: "text", text: m.content }]
|
|
108
|
+
: Array.isArray(m.content)
|
|
109
|
+
? [...m.content]
|
|
110
|
+
: null;
|
|
111
|
+
if (blocks && !blocks.some((b) => b && b.cache_control)) {
|
|
112
|
+
for (let j = blocks.length - 1; j >= 0; j--) {
|
|
113
|
+
if (isText(blocks[j]) && blocks[j].text) {
|
|
114
|
+
blocks[j] = { ...blocks[j], cache_control: MARK };
|
|
115
|
+
out[last] = { ...m, content: blocks };
|
|
116
|
+
break;
|
|
117
|
+
}
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
// Longer-lived marks must come first: anything before a one-hour mark
|
|
123
|
+
// lives an hour too.
|
|
124
|
+
let lastHourAt = -1;
|
|
125
|
+
out.forEach((msg, i) => {
|
|
126
|
+
if (Array.isArray(msg?.content) && msg.content.some((b) => b?.cache_control?.ttl === "1h")) lastHourAt = i;
|
|
127
|
+
});
|
|
128
|
+
if (lastHourAt > 0) {
|
|
129
|
+
for (let i = 0; i < lastHourAt; i++) {
|
|
130
|
+
const msg = out[i];
|
|
131
|
+
if (!Array.isArray(msg?.content)) continue;
|
|
132
|
+
if (!msg.content.some((b) => b?.cache_control && b.cache_control.ttl !== "1h")) continue;
|
|
133
|
+
out[i] = {
|
|
134
|
+
...msg,
|
|
135
|
+
content: msg.content.map((b) =>
|
|
136
|
+
b?.cache_control && b.cache_control.ttl !== "1h" ? { ...b, cache_control: { ...b.cache_control, ttl: "1h" } } : b,
|
|
137
|
+
),
|
|
138
|
+
};
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
return out;
|
|
142
|
+
}
|