@cap-js/agents 0.0.0 → 0.9.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/LICENSE +201 -0
  2. package/README.md +155 -0
  3. package/_i18n/messages.properties +31 -0
  4. package/cds-plugin.js +140 -0
  5. package/index.cds +105 -0
  6. package/index.js +0 -0
  7. package/lib/agents/markdown/backends/mime-utils.js +37 -0
  8. package/lib/agents/markdown/backends/outputs-backend.js +152 -0
  9. package/lib/agents/markdown/backends/uploads-backend.js +143 -0
  10. package/lib/agents/markdown/deep-agent.js +93 -0
  11. package/lib/agents/middleware/agent-actions.js +18 -0
  12. package/lib/agents/middleware/content-filter.js +191 -0
  13. package/lib/agents/middleware/hitl-edit-note-injector.js +18 -0
  14. package/lib/agents/middleware/hitl.js +20 -0
  15. package/lib/agents/middleware/index.js +19 -0
  16. package/lib/agents/middleware/patch-tool-calls.js +51 -0
  17. package/lib/agents/middleware/quota-enforcer.js +94 -0
  18. package/lib/agents/middleware/status-update.js +153 -0
  19. package/lib/agents/middleware/tool-selection.js +22 -0
  20. package/lib/agents/quota-enforcer-at-start.js +198 -0
  21. package/lib/agents/summarize-on-timeout.js +91 -0
  22. package/lib/compile.js +55 -0
  23. package/lib/index.cjs +1 -0
  24. package/lib/index.js +422 -0
  25. package/lib/models/aicore.js +441 -0
  26. package/lib/models/anthropic.js +77 -0
  27. package/lib/models/mock.js +88 -0
  28. package/lib/preview/chat.html +875 -0
  29. package/lib/preview/preview.js +46 -0
  30. package/lib/protocol/agent-card.js +297 -0
  31. package/lib/protocol/persistence/checkpoint-saver.js +317 -0
  32. package/lib/protocol/persistence/file-store.js +209 -0
  33. package/lib/protocol/persistence/push-notification-store.js +59 -0
  34. package/lib/protocol/persistence/task-store.js +47 -0
  35. package/lib/protocol/push-notification-sender.js +57 -0
  36. package/lib/sidecar.js +162 -0
  37. package/lib/telemetry/active-users.js +106 -0
  38. package/lib/telemetry/chat-tracing.js +342 -0
  39. package/lib/telemetry/metrics.js +85 -0
  40. package/lib/telemetry/mlflow.js +290 -0
  41. package/lib/telemetry/tool-tracing.js +164 -0
  42. package/lib/telemetry/tracing.js +150 -0
  43. package/lib/utils/inner-auth.js +33 -0
  44. package/lib/utils/markdown.js +199 -0
  45. package/lib/utils/message-handling.js +155 -0
  46. package/lib/utils/utils.js +168 -0
  47. package/package.json +225 -2
  48. package/srv/graph-cache.js +82 -0
  49. package/srv/handlers/graph-executor.js +1369 -0
  50. package/srv/handlers/index.js +173 -0
  51. package/srv/handlers/mcp-tools.js +159 -0
  52. package/srv/handlers/sub-agent-tools.js +314 -0
  53. package/srv/handlers/system-prompt.js +25 -0
  54. package/srv/handlers/tools.js +366 -0
  55. package/srv/langgraph-executor-srv.js +70 -0
  56. package/srv/push-notification-srv.js +149 -0
@@ -0,0 +1,342 @@
1
+ import cds from "@sap/cds"
2
+ import * as metrics from "./metrics.js"
3
+ import { mlflowAttrs, setSpanAttrs } from "./mlflow.js"
4
+ import { audit } from "../utils/utils.js"
5
+
6
+ const LOG = cds.log("agents")
7
+
8
+ const PATCHED = Symbol.for("@cap-js/agents:patched")
9
+ const SPAN_KIND_CLIENT = 3
10
+
11
+ // ─── Public API ──────────────────────────────────────────────────────────────
12
+
13
+ export async function patchChatModel() {
14
+ try {
15
+ const mod = await import("@langchain/core/language_models/chat_models")
16
+ if (mod.BaseChatModel?.prototype) _patchChatModelProto(mod.BaseChatModel.prototype)
17
+
18
+ // Also patch CJS prototype (different from ESM in dual-package modules)
19
+ try {
20
+ const { createRequire } = await import("node:module")
21
+ const req = createRequire(import.meta.url)
22
+ const cjs = req("@langchain/core/language_models/chat_models")
23
+ if (cjs.BaseChatModel?.prototype) _patchChatModelProto(cjs.BaseChatModel.prototype)
24
+ } catch {
25
+ /* CJS not available */
26
+ }
27
+ } catch (err) {
28
+ LOG.error("Failed to patch BaseChatModel for tracing", { error: err.message })
29
+ }
30
+ }
31
+
32
+ // ─── Prototype patch ─────────────────────────────────────────────────────────
33
+
34
+ export function _patchChatModelProto(proto) {
35
+ if (proto[PATCHED]) return
36
+ const original = proto.invoke
37
+ if (typeof original !== "function") {
38
+ LOG.warn("BaseChatModel.invoke not found — tracing patch skipped")
39
+ return
40
+ }
41
+
42
+ proto.invoke = async function (input, opts) {
43
+ // Skip instances explicitly opted out (e.g. content-filter probe model)
44
+ const INSTRUMENTED = Symbol.for("@cap-js/agents:instrumented")
45
+ if (this[INSTRUMENTED]) return original.call(this, input, opts)
46
+
47
+ const tracer = metrics.getTracer()
48
+ if (!tracer) return original.call(this, input, opts)
49
+
50
+ const model = this.options?.model || this.model || this.constructor.name
51
+ const provider = _detectProvider(this)
52
+ const node = opts?.runName || "agent"
53
+ const messages = Array.isArray(input) ? input : undefined
54
+ const cacheControl = _detectCacheControl(messages)
55
+ const streaming = this.streaming || this.orchestrationConfig?.streaming || false
56
+ const mAttrs = _metricAttrs(model, node)
57
+
58
+ const invoke = async (span) => {
59
+ if (span) {
60
+ setLLMSpanStartAttrs(span, {
61
+ model,
62
+ provider,
63
+ params: this.options?.params,
64
+ streaming,
65
+ cacheControl,
66
+ node,
67
+ messages,
68
+ })
69
+ }
70
+ const t0 = Date.now()
71
+ try {
72
+ const result = await original.call(this, input, opts)
73
+ const duration = Date.now() - t0
74
+ /** @type {import('@langchain/core/messages').UsageMetadata} */
75
+ const usage = result.usage_metadata
76
+ const finishReason =
77
+ result.response_metadata?.finish_reason ||
78
+ result.response_metadata?.stop_reason ||
79
+ (result.tool_calls?.length > 0 ? "tool_use" : "stop")
80
+
81
+ _handleSuccess(span, {
82
+ model,
83
+ provider,
84
+ mAttrs,
85
+ opts,
86
+ params: this.options?.params,
87
+ node,
88
+ tokenUsage: convertUsageData(usage),
89
+ outputContent: result.content ? extractText(result.content) : undefined,
90
+ response: {
91
+ finishReason,
92
+ model: result.response_metadata?.model_name || result.response_metadata?.model || model,
93
+ id: result.response_metadata?.id || result.id,
94
+ toolCalls: result.tool_calls?.map((tc) => ({ name: tc.name, args: tc.args })),
95
+ },
96
+ duration,
97
+ })
98
+ return result
99
+ } catch (err) {
100
+ _handleError(span, { err, mAttrs, model, node, messages })
101
+ throw err
102
+ } finally {
103
+ if (span) span.end()
104
+ }
105
+ }
106
+
107
+ return tracer.startActiveSpan(`chat ${model}`, { kind: SPAN_KIND_CLIENT }, (span) =>
108
+ invoke(span),
109
+ )
110
+ }
111
+ proto[PATCHED] = true
112
+ }
113
+
114
+ // ─── Internal helpers ────────────────────────────────────────────────────────
115
+
116
+ /** Detect LLM provider from model instance properties. */
117
+ function _detectProvider(instance) {
118
+ if (instance.orchestrationConfig) return "sap-ai-core"
119
+ return "langchain"
120
+ }
121
+
122
+ function _metricAttrs(model, node) {
123
+ return {
124
+ "sap.tenantId": cds.context?.tenant || "anonymous",
125
+ "agent.service": cds.context?.["agent.service"],
126
+ model,
127
+ node,
128
+ }
129
+ }
130
+
131
+ /** Detect cache_control presence in messages (set by injectCacheControl in aicore). */
132
+ function _detectCacheControl(messages) {
133
+ if (!Array.isArray(messages)) return false
134
+ return messages.some((m) => {
135
+ if (!Array.isArray(m.content)) return false
136
+ return m.content.some((b) => b.cache_control)
137
+ })
138
+ }
139
+
140
+ /** Handle successful LLM call: metrics, span end attrs, truncation warning, audit. */
141
+ function _handleSuccess(
142
+ span,
143
+ { model, provider, mAttrs, opts, params, node, tokenUsage, outputContent, response, duration },
144
+ ) {
145
+ metrics.llmInvocations.add(1, { ...mAttrs, outcome: "success" })
146
+ if (tokenUsage?.input_tokens) metrics.llmInputTokens.add(tokenUsage.input_tokens, mAttrs)
147
+ if (tokenUsage?.output_tokens) metrics.llmOutputTokens.add(tokenUsage.output_tokens, mAttrs)
148
+
149
+ if (span) {
150
+ setLLMSpanEndAttrs(span, { model, provider, tokenUsage, outputContent, response })
151
+ span.setStatus({ code: 1 })
152
+ }
153
+
154
+ if (response?.finishReason === "length" || response?.finishReason === "max_tokens") {
155
+ LOG.warn("LLM response truncated: output_tokens reached max_tokens limit", {
156
+ model,
157
+ node,
158
+ max_tokens: params?.max_tokens,
159
+ output_tokens: tokenUsage?.output_tokens,
160
+ })
161
+ }
162
+
163
+ const taskId = opts?.configurable?._taskId || cds.context?.["agent.task.id"]
164
+ if (taskId) {
165
+ audit("AgentDecision", {
166
+ data: {
167
+ taskId,
168
+ contextId:
169
+ opts?.configurable?.thread_id?.split(":")[1] || cds.context?.["agent.context.id"],
170
+ service: opts?.configurable?._service || cds.context?.["agent.service"],
171
+ model,
172
+ iteration: opts?.configurable?._iteration ?? cds.context?.["agent.iteration"],
173
+ toolCalls: response?.toolCalls,
174
+ tokenUsage,
175
+ duration,
176
+ },
177
+ })
178
+ }
179
+ }
180
+
181
+ /** Handle LLM error: content-filter warnings, metrics, span error attrs. */
182
+ function _handleError(span, { err, mAttrs, model, node, messages }) {
183
+ const status = err.rootCause?.status
184
+ const data = err.rootCause?.response?.data
185
+ const headers = err.rootCause?.response?.headers
186
+ const isFilterModule = /Filtering Module/i.test(data?.error?.location || "")
187
+ const isExternalFailure = headers?.["ai-external-failure"] === "true"
188
+
189
+ if (isFilterModule && status === 503 && isExternalFailure) {
190
+ LOG.warn(
191
+ "Content filter service rejected the request (likely payload too large for prompt_shield). " +
192
+ "Disable content filtering via the buildContentFilter event handler — see README → Content Filter → Limitations.",
193
+ { model, node, status, location: data?.error?.location, messageCount: messages?.length },
194
+ )
195
+ } else if (isFilterModule && status === 400) {
196
+ LOG.warn("Content filter blocked the request", {
197
+ model,
198
+ node,
199
+ status,
200
+ reason: data?.error?.message,
201
+ })
202
+ }
203
+
204
+ metrics.llmInvocations.add(1, { ...mAttrs, outcome: "error" })
205
+ if (span) {
206
+ span.setAttribute("error.type", err.constructor?.name || "Error")
207
+ span.setStatus({ code: 2, message: err.message })
208
+ span.recordException(err)
209
+ }
210
+ }
211
+
212
+ // ─── Shared LLM span helpers ─────────────────────────────────────────────────
213
+
214
+ /** Extract plain text from LangChain message content (string or content-block array). */
215
+ export function extractText(content) {
216
+ if (typeof content === "string") return content
217
+ if (Array.isArray(content)) {
218
+ return content
219
+ .filter((b) => b.type === "text")
220
+ .map((b) => b.text || "")
221
+ .join("")
222
+ }
223
+ return String(content ?? "")
224
+ }
225
+
226
+ /** Map LangChain message role to OpenAI role string. */
227
+ export function toRole(msg) {
228
+ const t = msg._getType?.()
229
+ if (t === "human") return "user"
230
+ if (t === "ai") return "assistant"
231
+ return t || "user"
232
+ }
233
+
234
+ /**
235
+ * Set common LLM span start attributes.
236
+ */
237
+ export function setLLMSpanStartAttrs(
238
+ span,
239
+ { model, provider, params, streaming, cacheControl, node, messages },
240
+ ) {
241
+ span.setAttribute("gen_ai.operation.name", "chat")
242
+ span.setAttribute("gen_ai.provider.name", provider)
243
+ span.setAttribute("gen_ai.request.model", model)
244
+ span.setAttribute("mlflow.message.format", "langchain-js")
245
+ if (params?.temperature != null)
246
+ span.setAttribute("gen_ai.request.temperature", params.temperature)
247
+ if (params?.max_tokens != null) span.setAttribute("gen_ai.request.max_tokens", params.max_tokens)
248
+ if (cds.context?.["agent.context.id"])
249
+ span.setAttribute("gen_ai.conversation.id", cds.context["agent.context.id"])
250
+ if (streaming) span.setAttribute("gen_ai.request.stream", true)
251
+ if (node) span.setAttribute("agent.llm.node", node)
252
+ if (cacheControl) span.setAttribute("gen_ai.request.cache_control", true)
253
+
254
+ // MLflow inputs: chat messages for Chat tab
255
+ if ((cds.env.agents?.mlflow || LOG._debug) && messages) {
256
+ const mlMessages = messages.map((m) => ({ role: toRole(m), content: extractText(m.content) }))
257
+ setSpanAttrs(span, mlflowAttrs("LLM", { model, provider, inputs: { messages: mlMessages } }))
258
+ if (LOG._debug) {
259
+ span.setAttribute("gen_ai.input.messages", JSON.stringify(messages.map((m) => m.content)))
260
+ }
261
+ }
262
+ }
263
+
264
+ /**
265
+ * Set gen_ai.response.*, gen_ai.usage.*, MLflow outputs on span end.
266
+ */
267
+ export function setLLMSpanEndAttrs(span, { model, provider, tokenUsage, outputContent, response }) {
268
+ if (LOG._debug && outputContent) {
269
+ span.setAttribute("gen_ai.output.messages", outputContent)
270
+ }
271
+
272
+ // gen_ai.response.* attributes
273
+ if (response) {
274
+ if (response.finishReason)
275
+ span.setAttribute("gen_ai.response.finish_reasons", [response.finishReason])
276
+ if (response.model) span.setAttribute("gen_ai.response.model", response.model)
277
+ if (response.id) span.setAttribute("gen_ai.response.id", response.id)
278
+ if (response.finishReason === "length" || response.finishReason === "max_tokens") {
279
+ span.setAttribute("gen_ai.response.truncated", true)
280
+ }
281
+ if (response.toolCalls?.length > 0) {
282
+ span.setAttribute("gen_ai.response.tool_calls", JSON.stringify(response.toolCalls))
283
+ }
284
+ }
285
+
286
+ if (tokenUsage) {
287
+ setTokenUsage(span, tokenUsage)
288
+ if (cds.env.agents?.mlflow) {
289
+ const mlOpts = { model, provider, tokenUsage: { ...tokenUsage, reasoning_tokens: undefined } }
290
+ if (outputContent) {
291
+ mlOpts.outputs = {
292
+ choices: [{ message: { role: "assistant", content: outputContent } }],
293
+ }
294
+ }
295
+ setSpanAttrs(span, mlflowAttrs("LLM", mlOpts))
296
+ return
297
+ }
298
+ }
299
+
300
+ // MLflow outputs without token usage
301
+ if (cds.env.agents?.mlflow && outputContent) {
302
+ setSpanAttrs(
303
+ span,
304
+ mlflowAttrs("LLM", {
305
+ model,
306
+ provider,
307
+ outputs: { choices: [{ message: { role: "assistant", content: outputContent } }] },
308
+ }),
309
+ )
310
+ }
311
+ }
312
+
313
+ export function setTokenUsage(span, tokenUsage) {
314
+ span.setAttribute("gen_ai.usage.input_tokens", tokenUsage.input_tokens)
315
+ span.setAttribute("gen_ai.usage.output_tokens", tokenUsage.output_tokens)
316
+ span.setAttribute("gen_ai.usage.total_tokens", tokenUsage.total_tokens)
317
+
318
+ if (tokenUsage.cache_read_input_tokens != null)
319
+ span.setAttribute("gen_ai.usage.cache_read.input_tokens", tokenUsage.cache_read_input_tokens)
320
+ if (tokenUsage.cache_creation_input_tokens != null)
321
+ span.setAttribute(
322
+ "gen_ai.usage.cache_creation.input_tokens",
323
+ tokenUsage.cache_creation_input_tokens,
324
+ )
325
+ if (tokenUsage.reasoning_tokens != null)
326
+ span.setAttribute("gen_ai.usage.reasoning.output_tokens", tokenUsage.reasoning_tokens)
327
+ }
328
+
329
+ /**
330
+ * Convert LangChain UsageMetadata to flat token usage format.
331
+ */
332
+ export function convertUsageData(usage) {
333
+ if (!usage) return undefined
334
+ return {
335
+ input_tokens: usage.input_tokens,
336
+ output_tokens: usage.output_tokens,
337
+ total_tokens: usage.total_tokens,
338
+ cache_creation_input_tokens: usage.input_token_details?.cache_creation,
339
+ cache_read_input_tokens: usage.input_token_details?.cache_read,
340
+ reasoning_tokens: usage.output_token_details?.reasoning,
341
+ }
342
+ }
@@ -0,0 +1,85 @@
1
+ import cds from "@sap/cds"
2
+ import { createRequire } from "node:module"
3
+
4
+ const NOOP_METER = (() => {
5
+ const noop = () => ({ add() {}, record() {} })
6
+ return { createCounter: noop, createHistogram: noop, createUpDownCounter: noop }
7
+ })()
8
+
9
+ let _otel
10
+ try {
11
+ // Synchronous require avoids top-level await which blocks module graph
12
+ // resolution and causes issues with CDS plugin loading order.
13
+ const require = createRequire(import.meta.url)
14
+ _otel = require("@opentelemetry/api")
15
+ } catch {
16
+ _otel = null
17
+ }
18
+
19
+ function getMeter() {
20
+ return _otel ? _otel.metrics.getMeter("@cap-js/agents") : NOOP_METER
21
+ }
22
+
23
+ const meter = getMeter()
24
+
25
+ export const requestDuration = meter.createHistogram("agent.request.duration", {
26
+ description: "End-to-end agent request duration",
27
+ unit: "ms",
28
+ })
29
+ export const requestsTotal = meter.createCounter("agent.requests.total", {
30
+ description: "Total inbound agent requests",
31
+ })
32
+ export const errorsTotal = meter.createCounter("agent.errors.total", {
33
+ description: "Agent requests resulting in error",
34
+ })
35
+ export const concurrentExecutions = meter.createUpDownCounter("agent.executions.concurrent", {
36
+ description: "Currently active workflow executions",
37
+ })
38
+ export const workflowsCompleted = meter.createCounter("agent.workflows.completed", {
39
+ description: "Number of completed agent workflows",
40
+ })
41
+ export const agentActions = meter.createCounter("agent_actions", {
42
+ description: "LLM invocations (agent node calls) per tenant",
43
+ })
44
+ export const llmInputTokens = meter.createCounter("agent.llm.input_tokens", {
45
+ description: "LLM input tokens consumed",
46
+ })
47
+ export const llmOutputTokens = meter.createCounter("agent.llm.output_tokens", {
48
+ description: "LLM output tokens generated",
49
+ })
50
+ export const llmInvocations = meter.createCounter("agent.llm.invocations", {
51
+ description: "LLM invocation count",
52
+ })
53
+ export const toolInvocations = meter.createCounter("agent.tool.invocations", {
54
+ description: "Tool invocation count",
55
+ })
56
+
57
+ /** Common attributes for all agent metrics */
58
+ export function attrs(srv) {
59
+ return {
60
+ "sap.tenantId": cds.context?.tenant || "anonymous",
61
+ "agent.service": typeof srv === "string" ? srv : srv.name,
62
+ }
63
+ }
64
+
65
+ /**
66
+ * Get active OTel span. Returns null if OTel is unavailable or no span is active.
67
+ */
68
+ export function getActiveSpan() {
69
+ return _otel?.trace.getActiveSpan() || null
70
+ }
71
+
72
+ /** Get OTel tracer for @cap-js/agents (or null if unavailable) */
73
+ export function getTracer() {
74
+ return _otel?.trace.getTracer("@cap-js/agents") || null
75
+ }
76
+
77
+ /** Create the active_users ObservableGauge with given observation callback */
78
+ export function createActiveUsersGauge(callback) {
79
+ if (!_otel) return null
80
+ const gauge = meter.createObservableGauge("active_users", {
81
+ description: "Active users per tenant and agent service (24h rolling window)",
82
+ })
83
+ gauge.addCallback(callback)
84
+ return gauge
85
+ }