@cap-js/agents 0.0.0 → 0.9.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/README.md +154 -0
- package/_i18n/messages.properties +31 -0
- package/cds-plugin.js +140 -0
- package/index.cds +116 -0
- package/index.js +0 -0
- package/lib/agents/markdown/backends/mime-utils.js +37 -0
- package/lib/agents/markdown/backends/outputs-backend.js +156 -0
- package/lib/agents/markdown/backends/readonly-backend.js +48 -0
- package/lib/agents/markdown/backends/uploads-backend.js +160 -0
- package/lib/agents/markdown/deep-agent.js +82 -0
- package/lib/agents/middleware/agent-actions.js +18 -0
- package/lib/agents/middleware/content-filter.js +191 -0
- package/lib/agents/middleware/hitl-edit-note-injector.js +18 -0
- package/lib/agents/middleware/hitl.js +20 -0
- package/lib/agents/middleware/index.js +19 -0
- package/lib/agents/middleware/patch-tool-calls.js +51 -0
- package/lib/agents/middleware/quota-enforcer.js +94 -0
- package/lib/agents/middleware/status-update.js +153 -0
- package/lib/agents/middleware/tool-selection.js +22 -0
- package/lib/agents/quota-enforcer-at-start.js +198 -0
- package/lib/agents/summarize-on-timeout.js +91 -0
- package/lib/compile.js +55 -0
- package/lib/index.cjs +1 -0
- package/lib/index.js +422 -0
- package/lib/models/aicore.js +441 -0
- package/lib/models/anthropic.js +77 -0
- package/lib/models/mock.js +88 -0
- package/lib/preview/chat.html +897 -0
- package/lib/preview/preview.js +46 -0
- package/lib/protocol/agent-card.js +297 -0
- package/lib/protocol/persistence/checkpoint-saver.js +320 -0
- package/lib/protocol/persistence/cleanup.js +84 -0
- package/lib/protocol/persistence/file-store.js +209 -0
- package/lib/protocol/persistence/push-notification-store.js +59 -0
- package/lib/protocol/persistence/task-store.js +47 -0
- package/lib/protocol/push-notification-sender.js +57 -0
- package/lib/sidecar.js +162 -0
- package/lib/telemetry/active-users.js +106 -0
- package/lib/telemetry/chat-tracing.js +364 -0
- package/lib/telemetry/metrics.js +85 -0
- package/lib/telemetry/mlflow.js +290 -0
- package/lib/telemetry/tool-tracing.js +164 -0
- package/lib/telemetry/tracing.js +150 -0
- package/lib/utils/inner-auth.js +33 -0
- package/lib/utils/markdown.js +199 -0
- package/lib/utils/message-handling.js +155 -0
- package/lib/utils/utils.js +168 -0
- package/package.json +225 -2
- package/srv/graph-cache.js +82 -0
- package/srv/handlers/graph-executor.js +1374 -0
- package/srv/handlers/index.js +178 -0
- package/srv/handlers/mcp-tools.js +161 -0
- package/srv/handlers/sub-agent-tools.js +316 -0
- package/srv/handlers/system-prompt.js +25 -0
- package/srv/handlers/tools.js +366 -0
- package/srv/langgraph-executor-srv.js +70 -0
- package/srv/push-notification-srv.js +149 -0
|
@@ -0,0 +1,441 @@
|
|
|
1
|
+
import cds from "@sap/cds"
|
|
2
|
+
import { OrchestrationClient } from "@sap-ai-sdk/langchain"
|
|
3
|
+
import { circuitBreaker, timeout } from "@sap-cloud-sdk/resilience"
|
|
4
|
+
|
|
5
|
+
import { SystemMessage, ToolMessage, HumanMessage, AIMessage } from "@langchain/core/messages"
|
|
6
|
+
import { ms4 } from "../utils/utils.js"
|
|
7
|
+
|
|
8
|
+
const LOG = cds.log("agents")
|
|
9
|
+
|
|
10
|
+
class _InstrumentedOrchestrationClient extends OrchestrationClient {
|
|
11
|
+
constructor(name, options) {
|
|
12
|
+
const { model, deepAgent, streaming, destinationName, resourceGroup } = options
|
|
13
|
+
const flatten = options.flatten ?? (deepAgent ? true : false)
|
|
14
|
+
|
|
15
|
+
// Restore pre-72ef6fa params resolution: caller > deep-agent default > root
|
|
16
|
+
// cds.env.agents.params. Without this, AI Core silently applies its own
|
|
17
|
+
// defaults (temperature ≈ 1, verbose model → repeated max_tokens truncation
|
|
18
|
+
// and extra ReAct iterations — see PR #188 review).
|
|
19
|
+
const params =
|
|
20
|
+
options.params || (deepAgent ? { max_tokens: 4096, temperature: 0 } : cds.env.agents?.params)
|
|
21
|
+
|
|
22
|
+
LOG.debug("Initializing LLM", { model, deepAgent: !!deepAgent })
|
|
23
|
+
|
|
24
|
+
let { contentFilter } = options
|
|
25
|
+
contentFilter =
|
|
26
|
+
contentFilter === true ? buildContentFilter() : (contentFilter ?? buildContentFilter())
|
|
27
|
+
|
|
28
|
+
// only output filters (input handled by contentFilterMiddleware)
|
|
29
|
+
const filtering = toSdkFilterFormat({ output: contentFilter?.output })
|
|
30
|
+
|
|
31
|
+
// When AICore is behind a BTP destination (not bound as service instance),
|
|
32
|
+
// pass destinationName + resourceGroup to the SDK. The SDK resolves the
|
|
33
|
+
// destination via BTP Destination Service and uses it for all API calls.
|
|
34
|
+
const deploymentConfig = destinationName
|
|
35
|
+
? { resourceGroup: resourceGroup || "default" }
|
|
36
|
+
: resourceGroup
|
|
37
|
+
? { resourceGroup }
|
|
38
|
+
: undefined
|
|
39
|
+
const destination = destinationName ? { destinationName } : undefined
|
|
40
|
+
|
|
41
|
+
super(
|
|
42
|
+
{
|
|
43
|
+
promptTemplating: { model: { name: model, params } },
|
|
44
|
+
...(filtering && { filtering }),
|
|
45
|
+
// `streaming` controls the SDK's auto-stream-and-concatenate in _generate()
|
|
46
|
+
// (i.e. direct model.invoke() calls). It does NOT gate LangGraph token
|
|
47
|
+
// streaming — that is driven by graph.stream(streamMode:["messages"]) via a
|
|
48
|
+
// streaming callback handler + our overridden _streamResponseChunks, and
|
|
49
|
+
// works for every agent regardless of this flag (deep or managed alike).
|
|
50
|
+
// Default on; `streaming: false` opts out.
|
|
51
|
+
...(streaming !== false ? { streaming: true } : {}),
|
|
52
|
+
},
|
|
53
|
+
{
|
|
54
|
+
onFailedAttempt: (err) => {
|
|
55
|
+
// Abort retries when circuit breaker is open (otherwise pRetry delays ~30-60s)
|
|
56
|
+
if (err.code === "EOPENBREAKER" || err.message === "Breaker is open") {
|
|
57
|
+
throw err
|
|
58
|
+
}
|
|
59
|
+
},
|
|
60
|
+
},
|
|
61
|
+
deploymentConfig,
|
|
62
|
+
destination,
|
|
63
|
+
)
|
|
64
|
+
this.name = name
|
|
65
|
+
this.options = { ...options, params, contentFilter, flatten }
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
async _generate(messages, opts, runManager) {
|
|
69
|
+
const { model, flatten } = this.options
|
|
70
|
+
opts = _withMiddleware(this, opts)
|
|
71
|
+
const prepared = _prepareMessages(messages, { flatten, model, opts })
|
|
72
|
+
return super._generate(prepared.inputMessages, prepared.opts, runManager)
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/**
|
|
76
|
+
* Override _streamResponseChunks for three reasons:
|
|
77
|
+
*
|
|
78
|
+
* 1. Message flattening / normalization — same as _generate: content arrays
|
|
79
|
+
* must be reduced to text before reaching AI Core's assistant template.
|
|
80
|
+
* 2. Claude prompt caching — injectCacheControl must run here too, or
|
|
81
|
+
* cache_control is silently skipped on the streaming path.
|
|
82
|
+
* 3. Content extraction — @sap-ai-sdk/langchain getDeltaContent() only handles
|
|
83
|
+
* string deltas; Anthropic streams content-block arrays (and reasoning in a
|
|
84
|
+
* sibling field), so we re-extract and re-emit them via buildStreamBlocks().
|
|
85
|
+
*/
|
|
86
|
+
async *_streamResponseChunks(messages, opts, runManager) {
|
|
87
|
+
const { model, flatten } = this.options
|
|
88
|
+
opts = _withMiddleware(this, opts)
|
|
89
|
+
const prepared = _prepareMessages(messages, { flatten, model, opts })
|
|
90
|
+
const inputMessages = prepared.inputMessages
|
|
91
|
+
opts = prepared.opts
|
|
92
|
+
|
|
93
|
+
let turnHasToolCall = false
|
|
94
|
+
for await (const chunk of super._streamResponseChunks(inputMessages, opts, runManager)) {
|
|
95
|
+
if (chunk.message?.tool_call_chunks?.length > 0) turnHasToolCall = true
|
|
96
|
+
const text = chunk.text || extractTextFromContentBlocks(chunk)
|
|
97
|
+
const reasoning = extractReasoningFromChunk(chunk)
|
|
98
|
+
const blocks = buildStreamBlocks(text, reasoning, turnHasToolCall)
|
|
99
|
+
|
|
100
|
+
if (!blocks.length && !(text && turnHasToolCall)) {
|
|
101
|
+
yield chunk
|
|
102
|
+
continue
|
|
103
|
+
}
|
|
104
|
+
yield patchChunkContent(chunk, blocks)
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
export default _InstrumentedOrchestrationClient
|
|
110
|
+
|
|
111
|
+
// ─── SDK helpers ─────────────────────────────────────────────────────────────
|
|
112
|
+
|
|
113
|
+
function isClaude(model) {
|
|
114
|
+
return /anthropic|claude/i.test(model || "")
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
/**
|
|
118
|
+
* Inject timeout and circuit breaker middleware into opts.
|
|
119
|
+
*/
|
|
120
|
+
function _withMiddleware(instance, opts) {
|
|
121
|
+
const llmTimeout = ms4(cds.env.agents?.pool?.maxLLMCallTimeout || "120s")
|
|
122
|
+
const middleware = [timeout(llmTimeout), circuitBreaker()]
|
|
123
|
+
return {
|
|
124
|
+
...opts,
|
|
125
|
+
customRequestConfig: { ...opts?.customRequestConfig, middleware },
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
/**
|
|
130
|
+
* Prepare input messages: flatten, normalize assistant content, inject cache_control.
|
|
131
|
+
* Returns { inputMessages, opts } (opts may be mutated with cache_control on last tool).
|
|
132
|
+
*/
|
|
133
|
+
function _prepareMessages(messages, { flatten, model, opts }) {
|
|
134
|
+
let inputMessages = flatten ? flattenMessages(messages) : messages
|
|
135
|
+
inputMessages = normalizeAssistantContent(inputMessages)
|
|
136
|
+
if (isClaude(model)) {
|
|
137
|
+
inputMessages = injectCacheControl(inputMessages)
|
|
138
|
+
if (opts?.tools?.length > 0) {
|
|
139
|
+
const tools = [...opts.tools]
|
|
140
|
+
tools[tools.length - 1] = {
|
|
141
|
+
...tools[tools.length - 1],
|
|
142
|
+
cache_control: CACHE_CONTROL_EPHEMERAL,
|
|
143
|
+
}
|
|
144
|
+
opts = { ...opts, tools }
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
return { inputMessages, opts }
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
// Claude currently only supports caching of type ephemeral. TTL can differ between 5min or 1h but
|
|
151
|
+
// we use the 5min default
|
|
152
|
+
const CACHE_CONTROL_EPHEMERAL = { type: "ephemeral" }
|
|
153
|
+
|
|
154
|
+
/**
|
|
155
|
+
* Marks: all system messages, the last AI message (with text content), and the last human message.
|
|
156
|
+
* Converts string content to content-block arrays where needed so cache_control
|
|
157
|
+
* can be attached per the Anthropic/SAP AI Core API format.
|
|
158
|
+
*/
|
|
159
|
+
function injectCacheControl(messages) {
|
|
160
|
+
if (!messages || messages.length === 0) return messages
|
|
161
|
+
|
|
162
|
+
const result = messages.map((m) => {
|
|
163
|
+
// Clone to avoid mutating original
|
|
164
|
+
if (m._getType?.() === "system" || m.type === "system") {
|
|
165
|
+
return _withCacheControl(m)
|
|
166
|
+
}
|
|
167
|
+
return m
|
|
168
|
+
})
|
|
169
|
+
// Mark last AI message with non-empty text content (stable breakpoint for multi-turn)
|
|
170
|
+
for (let i = result.length - 1; i >= 0; i--) {
|
|
171
|
+
const type = result[i]._getType?.() || result[i].type
|
|
172
|
+
if (type === "ai" && _hasTextContent(result[i])) {
|
|
173
|
+
result[i] = _withCacheControl(result[i])
|
|
174
|
+
break
|
|
175
|
+
}
|
|
176
|
+
}
|
|
177
|
+
// Mark last human message
|
|
178
|
+
for (let i = result.length - 1; i >= 0; i--) {
|
|
179
|
+
const type = result[i]._getType?.() || result[i].type
|
|
180
|
+
if (type === "human") {
|
|
181
|
+
result[i] = _withCacheControl(result[i])
|
|
182
|
+
break
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
return result
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/**
|
|
189
|
+
* Check if a message has non-empty text content (not just tool_calls).
|
|
190
|
+
*/
|
|
191
|
+
function _hasTextContent(msg) {
|
|
192
|
+
const content = msg.content
|
|
193
|
+
if (typeof content === "string") return content.length > 0
|
|
194
|
+
if (Array.isArray(content)) return content.some((b) => b.type === "text" && b.text?.length > 0)
|
|
195
|
+
return false
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
/**
|
|
199
|
+
* If content is a string, convert to [{type:"text", text, cache_control}].
|
|
200
|
+
* If content is an array, add cache_control to the last text block.
|
|
201
|
+
*/
|
|
202
|
+
function _withCacheControl(msg) {
|
|
203
|
+
const content = msg.content
|
|
204
|
+
if (typeof content === "string") {
|
|
205
|
+
// Convert to content blocks with cache_control on the block
|
|
206
|
+
const newContent = [{ type: "text", text: content, cache_control: CACHE_CONTROL_EPHEMERAL }]
|
|
207
|
+
return _cloneMessageWithContent(msg, newContent)
|
|
208
|
+
}
|
|
209
|
+
if (Array.isArray(content) && content.length > 0) {
|
|
210
|
+
const newContent = [...content]
|
|
211
|
+
// Find last text block and add cache_control
|
|
212
|
+
for (let i = newContent.length - 1; i >= 0; i--) {
|
|
213
|
+
if (newContent[i].type === "text") {
|
|
214
|
+
newContent[i] = { ...newContent[i], cache_control: CACHE_CONTROL_EPHEMERAL }
|
|
215
|
+
break
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
return _cloneMessageWithContent(msg, newContent)
|
|
219
|
+
}
|
|
220
|
+
return msg
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
function _cloneMessageWithContent(msg, newContent) {
|
|
224
|
+
const type = msg._getType?.() || msg.type
|
|
225
|
+
if (type === "system") {
|
|
226
|
+
return new SystemMessage({ content: newContent })
|
|
227
|
+
}
|
|
228
|
+
if (type === "ai") {
|
|
229
|
+
return new AIMessage({ content: newContent, tool_calls: msg.tool_calls })
|
|
230
|
+
}
|
|
231
|
+
if (type === "human") {
|
|
232
|
+
return new HumanMessage({ content: newContent })
|
|
233
|
+
}
|
|
234
|
+
if (type === "tool") {
|
|
235
|
+
return new ToolMessage({ content: newContent, tool_call_id: msg.tool_call_id, name: msg.name })
|
|
236
|
+
}
|
|
237
|
+
// Fallback: shallow clone with new content
|
|
238
|
+
return { ...msg, content: newContent }
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
export function flattenMessages(messages) {
|
|
242
|
+
return messages.map((m) => {
|
|
243
|
+
if (!Array.isArray(m.content)) return m
|
|
244
|
+
const isSystem = SystemMessage.isInstance?.(m) || m._getType?.() === "system"
|
|
245
|
+
const isTool = ToolMessage.isInstance?.(m) || m._getType?.() === "tool"
|
|
246
|
+
if (!isSystem && !isTool) return m
|
|
247
|
+
|
|
248
|
+
const parts = m.content.map((b) => {
|
|
249
|
+
if (typeof b === "string") return b
|
|
250
|
+
if (!b || typeof b !== "object") return ""
|
|
251
|
+
if (b.type === "text") return b.text || ""
|
|
252
|
+
if (b.type === "image" || b.type === "audio" || b.type === "video" || b.type === "file") {
|
|
253
|
+
const mime = b.mimeType || b.mime_type || "application/octet-stream"
|
|
254
|
+
const data = b.data || b.source?.data || ""
|
|
255
|
+
const bytes = typeof data === "string" ? Buffer.byteLength(data, "base64") : 0
|
|
256
|
+
return `[binary ${mime}, ${bytes} bytes]`
|
|
257
|
+
}
|
|
258
|
+
return JSON.stringify(b).slice(0, 200)
|
|
259
|
+
})
|
|
260
|
+
const text = parts.join("\n")
|
|
261
|
+
|
|
262
|
+
if (isTool) {
|
|
263
|
+
return new ToolMessage({
|
|
264
|
+
content: text,
|
|
265
|
+
tool_call_id: m.tool_call_id,
|
|
266
|
+
name: m.name,
|
|
267
|
+
status: m.status,
|
|
268
|
+
additional_kwargs: m.additional_kwargs,
|
|
269
|
+
})
|
|
270
|
+
}
|
|
271
|
+
return new SystemMessage({
|
|
272
|
+
content: text,
|
|
273
|
+
additional_kwargs: m.additional_kwargs,
|
|
274
|
+
response_metadata: m.response_metadata,
|
|
275
|
+
})
|
|
276
|
+
})
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
/**
|
|
280
|
+
* AI Core's assistant template only accepts `text` content blocks, but after a
|
|
281
|
+
* tool round-trip LangChain puts a `tool_call` block in the content array (the
|
|
282
|
+
* call itself already travels in top-level `tool_calls`). We join the text
|
|
283
|
+
* blocks and drop the rest. Clone-and-overwrite rather than `new AIMessage(...)`:
|
|
284
|
+
* that constructor re-materializes the tool_call block from `tool_calls` when
|
|
285
|
+
* content is empty, undoing the strip.
|
|
286
|
+
*/
|
|
287
|
+
export function normalizeAssistantContent(messages) {
|
|
288
|
+
return messages.map((m) => {
|
|
289
|
+
if (!Array.isArray(m.content)) return m
|
|
290
|
+
const isAI = AIMessage.isInstance?.(m) || m._getType?.() === "ai" || m.type === "ai"
|
|
291
|
+
if (!isAI) return m
|
|
292
|
+
const text = m.content
|
|
293
|
+
.filter((b) => (typeof b === "string" ? true : b?.type === "text"))
|
|
294
|
+
.map((b) => (typeof b === "string" ? b : (b.text ?? "")))
|
|
295
|
+
.join("")
|
|
296
|
+
const clone = Object.assign(Object.create(Object.getPrototypeOf(m)), m)
|
|
297
|
+
clone.content = text
|
|
298
|
+
if (clone.lc_kwargs) clone.lc_kwargs = { ...clone.lc_kwargs, content: text }
|
|
299
|
+
return clone
|
|
300
|
+
})
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
/** Content filter thresholds */
|
|
304
|
+
const AZURE_THRESHOLDS = {
|
|
305
|
+
ALLOW_SAFE: 0,
|
|
306
|
+
ALLOW_SAFE_LOW: 2,
|
|
307
|
+
ALLOW_SAFE_LOW_MEDIUM: 4,
|
|
308
|
+
ALLOW_ALL: 6,
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
export function buildContentFilter() {
|
|
312
|
+
return {
|
|
313
|
+
input: {
|
|
314
|
+
azure_content_safety: {
|
|
315
|
+
hate: "ALLOW_SAFE_LOW",
|
|
316
|
+
violence: "ALLOW_SAFE_LOW_MEDIUM",
|
|
317
|
+
prompt_shield: true,
|
|
318
|
+
},
|
|
319
|
+
},
|
|
320
|
+
output: {
|
|
321
|
+
azure_content_safety: {
|
|
322
|
+
hate: "ALLOW_SAFE",
|
|
323
|
+
violence: "ALLOW_SAFE_LOW_MEDIUM",
|
|
324
|
+
},
|
|
325
|
+
},
|
|
326
|
+
}
|
|
327
|
+
}
|
|
328
|
+
|
|
329
|
+
/**
|
|
330
|
+
* Convert simplified dictionary to SDK array format.
|
|
331
|
+
* Azure threshold strings are converted to numeric values.
|
|
332
|
+
*/
|
|
333
|
+
export function toSdkFilterFormat(filter) {
|
|
334
|
+
const result = {}
|
|
335
|
+
if (filter?.input) {
|
|
336
|
+
result.input = convertFilter(filter.input)
|
|
337
|
+
}
|
|
338
|
+
if (filter?.output) {
|
|
339
|
+
result.output = convertFilter(filter.output)
|
|
340
|
+
}
|
|
341
|
+
return result
|
|
342
|
+
}
|
|
343
|
+
|
|
344
|
+
function convertFilter(c) {
|
|
345
|
+
const contentSafety = { ...c.azure_content_safety }
|
|
346
|
+
for (const [key, value] of Object.entries(contentSafety)) {
|
|
347
|
+
contentSafety[key] = AZURE_THRESHOLDS[value] ?? value
|
|
348
|
+
}
|
|
349
|
+
const converted = { ...c, azure_content_safety: contentSafety }
|
|
350
|
+
return {
|
|
351
|
+
filters: Object.entries(converted).map(([type, config]) => ({ type, config })),
|
|
352
|
+
}
|
|
353
|
+
}
|
|
354
|
+
|
|
355
|
+
/**
|
|
356
|
+
* Reads the per-chunk LLM `delta` for choice 0. On the streaming path the SDK
|
|
357
|
+
* yields a ChatGenerationChunk with no `_data` — the raw payload lives on
|
|
358
|
+
* `message.additional_kwargs.intermediate_results` — so we check both.
|
|
359
|
+
*/
|
|
360
|
+
function getChunkDelta(chunk) {
|
|
361
|
+
const ir =
|
|
362
|
+
chunk.message?.additional_kwargs?.intermediate_results ?? chunk._data?.intermediate_results
|
|
363
|
+
return ir?.llm?.choices?.[0]?.delta ?? chunk._data?.final_result?.choices?.[0]?.delta
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
/**
|
|
367
|
+
* Extracts text from Anthropic content-block arrays in a streaming chunk.
|
|
368
|
+
*
|
|
369
|
+
* @sap-ai-sdk/langchain getDeltaContent() only handles string deltas
|
|
370
|
+
* (ChatDelta.content typed as string in the API spec). Anthropic returns
|
|
371
|
+
* content as Array<{type,text}> — getDeltaContent() returns "" for every
|
|
372
|
+
* chunk, silencing handleLLMNewToken. We extract the text manually.
|
|
373
|
+
*/
|
|
374
|
+
export function extractTextFromContentBlocks(chunk) {
|
|
375
|
+
const rawContent = getChunkDelta(chunk)?.content
|
|
376
|
+
if (!Array.isArray(rawContent)) return ""
|
|
377
|
+
return rawContent
|
|
378
|
+
.filter((b) => b && b.type === "text")
|
|
379
|
+
.map((b) => b.text ?? "")
|
|
380
|
+
.join("")
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
/**
|
|
384
|
+
* Extracts reasoning ("thinking") text from a streaming chunk. AI Core streams
|
|
385
|
+
* Claude reasoning tokens in a sibling `delta.reasoning_content` field (array of
|
|
386
|
+
* `{ content, signature }`), NOT in `delta.content`, and the SDK never surfaces it.
|
|
387
|
+
*/
|
|
388
|
+
export function extractReasoningFromChunk(chunk) {
|
|
389
|
+
const reasoning = getChunkDelta(chunk)?.reasoning_content
|
|
390
|
+
if (!Array.isArray(reasoning)) return ""
|
|
391
|
+
return reasoning.map((b) => (typeof b === "string" ? b : (b?.content ?? ""))).join("")
|
|
392
|
+
}
|
|
393
|
+
|
|
394
|
+
// High indices keep text/reasoning off the tool-call block indices (0..N).
|
|
395
|
+
const TEXT_BLOCK_INDEX = 100
|
|
396
|
+
const REASONING_BLOCK_INDEX = 101
|
|
397
|
+
|
|
398
|
+
/**
|
|
399
|
+
* Builds the content-block parts for a streaming chunk.
|
|
400
|
+
*
|
|
401
|
+
* NOTE ON PRIMING: an earlier version pushed an empty `{ text: "", index: 100 }`
|
|
402
|
+
* placeholder alongside the first real text delta of a turn — a workaround for
|
|
403
|
+
* langchain's `TextContentStream` (via `convertChunksToEvents`), which reads
|
|
404
|
+
* only `content-block-delta` events and would otherwise drop the first text.
|
|
405
|
+
* We removed it because langchain-core's `_mergeLists` (`messages/base.js`)
|
|
406
|
+
* matches array items by `index`: the empty priming block became the merge
|
|
407
|
+
* target for every subsequent delta while the real first delta was orphaned
|
|
408
|
+
* at position 1 in the aggregated content array. Downstream readers that join
|
|
409
|
+
* the array in order (e.g. `messageText` in graph-executor.js) then produce
|
|
410
|
+
* scrambled text with the first delta stranded at the end.
|
|
411
|
+
*
|
|
412
|
+
* No code path in this repo consumes chunks via `convertChunksToEvents`/
|
|
413
|
+
* `TextContentStream`, so priming is unnecessary here. If such a consumer is
|
|
414
|
+
* ever introduced, add priming inside that consumer rather than on the wire.
|
|
415
|
+
*
|
|
416
|
+
* `suppressText` drops the text block for chunks in a tool-calling turn — the
|
|
417
|
+
* planning/reasoning text emitted alongside a tool_use block is not the
|
|
418
|
+
* user-facing answer and must not leak into the final response.
|
|
419
|
+
*/
|
|
420
|
+
export function buildStreamBlocks(text, reasoning, suppressText) {
|
|
421
|
+
const blocks = []
|
|
422
|
+
if (reasoning) blocks.push({ type: "reasoning", reasoning, index: REASONING_BLOCK_INDEX })
|
|
423
|
+
if (text && !suppressText) blocks.push({ type: "text", text, index: TEXT_BLOCK_INDEX })
|
|
424
|
+
return blocks
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
/**
|
|
428
|
+
* Shallow-clones the chunk with new content blocks, preserving tool_call_chunks
|
|
429
|
+
* so tool calls survive alongside text/reasoning on the same chunk.
|
|
430
|
+
*/
|
|
431
|
+
function patchChunkContent(chunk, blocks) {
|
|
432
|
+
const patchedMessage = chunk.message
|
|
433
|
+
? Object.assign(Object.create(Object.getPrototypeOf(chunk.message)), chunk.message, {
|
|
434
|
+
content: blocks,
|
|
435
|
+
})
|
|
436
|
+
: chunk.message
|
|
437
|
+
return Object.assign(Object.create(Object.getPrototypeOf(chunk)), chunk, {
|
|
438
|
+
text: blocks.filter((b) => b.type === "text").at(-1)?.text ?? "",
|
|
439
|
+
...(patchedMessage !== undefined && { message: patchedMessage }),
|
|
440
|
+
})
|
|
441
|
+
}
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
import { ChatAnthropic } from '@langchain/anthropic'
|
|
2
|
+
import path from 'node:path'
|
|
3
|
+
import fs from 'node:fs'
|
|
4
|
+
import os from 'node:os'
|
|
5
|
+
import cds from '@sap/cds'
|
|
6
|
+
|
|
7
|
+
const HOME = os.homedir() || process.env.HOME || process.env.USERPROFILE
|
|
8
|
+
const LOG = cds.log('agents')
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* `cds.connect.to` compliant langchain model
|
|
12
|
+
* for connecting to an Anthropic compatible API,
|
|
13
|
+
* with autoconfiguration based on env, options,
|
|
14
|
+
* ~/.claude/settings.json and ~/.config/opencode/opencode.json
|
|
15
|
+
*/
|
|
16
|
+
export default class ChatAnthropicService extends ChatAnthropic {
|
|
17
|
+
constructor (name, options) {
|
|
18
|
+
// REVISIT: may be better handled via options.credentials?
|
|
19
|
+
let config = { ...options, ...fromEnv() }
|
|
20
|
+
if (!config.anthropicApiUrl) config = {
|
|
21
|
+
...fromClaude() || fromOpencode(),
|
|
22
|
+
...config
|
|
23
|
+
}
|
|
24
|
+
LOG.debug (`Using effective config for ChatAnthropic:`, config)
|
|
25
|
+
super (config)
|
|
26
|
+
this.name = name
|
|
27
|
+
this.options = config
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
function fromEnv (env = process.env) {
|
|
32
|
+
let any, config = {}
|
|
33
|
+
if ((any = env.ANTHROPIC_BASE_URL)) config.anthropicApiUrl = any
|
|
34
|
+
if ((any = env.ANTHROPIC_AUTH_TOKEN)) config.apiKey = any
|
|
35
|
+
if ((any = env.ANTHROPIC_API_KEY)) config.apiKey = any
|
|
36
|
+
if ((any = env.ANTHROPIC_MODEL)) config.model = any
|
|
37
|
+
if (!Object.keys(config).length) return null
|
|
38
|
+
LOG.debug (`Loaded Anthropic settings from env:`, config)
|
|
39
|
+
return config
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
function fromClaude() {
|
|
43
|
+
if ('cached' in fromClaude) return fromClaude.cached
|
|
44
|
+
const settings_json = path.join (HOME,'.claude/settings.json')
|
|
45
|
+
try {
|
|
46
|
+
let conf = JSON.parse (fs.readFileSync (settings_json,'utf8'))
|
|
47
|
+
// https://www.schemastore.org/claude-code-settings.json
|
|
48
|
+
fromClaude.cached = conf = fromEnv (conf?.env)
|
|
49
|
+
LOG.debug(`Loaded Claude settings from`, settings_json, ':', conf)
|
|
50
|
+
} catch {
|
|
51
|
+
LOG.debug(`Failed loading Claude settings from`, settings_json)
|
|
52
|
+
fromClaude.cached = null
|
|
53
|
+
}
|
|
54
|
+
return fromClaude.cached
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
function fromOpencode() {
|
|
58
|
+
if ('cached' in fromOpencode) return fromOpencode.cached
|
|
59
|
+
const opencode_json = path.join (HOME,'.config/opencode/opencode.json')
|
|
60
|
+
try {
|
|
61
|
+
let conf = JSON.parse (fs.readFileSync (opencode_json,'utf8'))
|
|
62
|
+
LOG.debug(`Loaded OpenCode settings from`, opencode_json)
|
|
63
|
+
// https://opencode.ai/config.json
|
|
64
|
+
let o = conf?.provider?.anthropic?.options
|
|
65
|
+
if (!o) return fromOpencode.cached = null
|
|
66
|
+
let any, config = {}
|
|
67
|
+
if ((any = o.anthropicApiUrl ?? o.apiUrl ?? o.baseURL)) config.anthropicApiUrl = any.replace(/\/v1$/,'') // opencode expects the versioned baseUrl, others do not
|
|
68
|
+
if ((any = o.anthropicApiKey ?? o.apiKey)) config.apiKey = any
|
|
69
|
+
if ((any = conf?.model)) config.model = any.replace('anthropic/','')
|
|
70
|
+
fromOpencode.cached = Object.keys(config).length ? config : null
|
|
71
|
+
LOG.debug(`Loaded OpenCode settings from`, opencode_json, ':', conf)
|
|
72
|
+
} catch {
|
|
73
|
+
LOG.debug(`Failed loading OpenCode settings from`, opencode_json)
|
|
74
|
+
fromOpencode.cached = null
|
|
75
|
+
}
|
|
76
|
+
return fromOpencode.cached
|
|
77
|
+
}
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
import { BaseChatModel } from "@langchain/core/language_models/chat_models"
|
|
2
|
+
import { AIMessage } from "@langchain/core/messages"
|
|
3
|
+
|
|
4
|
+
const DEFAULT_MESSAGE =
|
|
5
|
+
"[Mock LLM] This is a mocked response from @cap-js/agents development mode. No real LLM was invoked."
|
|
6
|
+
|
|
7
|
+
export default class MockChatModel extends BaseChatModel {
|
|
8
|
+
constructor(name, options = {}) {
|
|
9
|
+
super({})
|
|
10
|
+
this.name = name
|
|
11
|
+
this.options = options
|
|
12
|
+
this._tools = []
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
_llmType() {
|
|
16
|
+
return "cap-mock-llm"
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
bindTools(tools) {
|
|
20
|
+
const bound = Object.create(this)
|
|
21
|
+
bound._tools = tools ?? []
|
|
22
|
+
return bound
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
async _generate(messages) {
|
|
26
|
+
const message = this.options.message || DEFAULT_MESSAGE
|
|
27
|
+
|
|
28
|
+
if (this._tools?.length > 0) {
|
|
29
|
+
const lastMsg = messages[messages.length - 1]
|
|
30
|
+
const lastType = lastMsg?._getType?.()
|
|
31
|
+
|
|
32
|
+
if (lastType !== "tool") {
|
|
33
|
+
const queryTool = this._tools.find((t) => t.name === "query")
|
|
34
|
+
const args = queryTool && buildQueryArgs(queryTool, this._tools)
|
|
35
|
+
if (args) {
|
|
36
|
+
return {
|
|
37
|
+
generations: [
|
|
38
|
+
{
|
|
39
|
+
message: new AIMessage({
|
|
40
|
+
content: "",
|
|
41
|
+
tool_calls: [{ id: `mock_${Date.now()}`, name: "query", args }],
|
|
42
|
+
}),
|
|
43
|
+
},
|
|
44
|
+
],
|
|
45
|
+
llmOutput: { model: `mock-${this.name}`, mock: true },
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
} else {
|
|
49
|
+
const toolResult = lastMsg?.content ?? ""
|
|
50
|
+
return {
|
|
51
|
+
generations: [{ message: new AIMessage(`${message}\n\nTool result: ${toolResult}`) }],
|
|
52
|
+
llmOutput: { model: `mock-${this.name}`, mock: true },
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
return {
|
|
58
|
+
generations: [{ message: new AIMessage(message) }],
|
|
59
|
+
llmOutput: { model: `mock-${this.name}`, mock: true },
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* Build the args object for a mock `query` tool call
|
|
66
|
+
* (works with format: cqn and cql)
|
|
67
|
+
*/
|
|
68
|
+
function buildQueryArgs(queryTool, tools) {
|
|
69
|
+
// CQN mode
|
|
70
|
+
const cqnEntities = queryTool?.schema?.shape?.entity?.def?.entries
|
|
71
|
+
if (cqnEntities) {
|
|
72
|
+
const entity = Object.keys(cqnEntities)[0]
|
|
73
|
+
if (entity) return { entity, limit: 3 }
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
// CQL mode
|
|
77
|
+
if (queryTool?.schema?.shape?.cql) {
|
|
78
|
+
const describeTool = tools.find((t) => t.name === "describe")
|
|
79
|
+
const describeEntities =
|
|
80
|
+
describeTool?.schema?.shape?.entities?.def?.innerType?.def?.element?.def?.entries
|
|
81
|
+
const entity = describeEntities && Object.keys(describeEntities)[0]
|
|
82
|
+
if (entity) return { cql: `SELECT * FROM ${entity} LIMIT 3` }
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
return null
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
MockChatModel._is_service_class = true
|