@cap-js/agents 0.9.1 → 0.9.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -1
- package/cds-plugin.js +2 -2
- package/lib/agents/markdown/backends/outputs-backend.js +16 -12
- package/lib/agents/markdown/backends/readonly-backend.js +48 -0
- package/lib/agents/markdown/backends/uploads-backend.js +30 -13
- package/lib/agents/markdown/deep-agent.js +15 -26
- package/lib/agents/middleware/content-filter.js +3 -2
- package/lib/agents/middleware/index.js +11 -0
- package/lib/agents/middleware/remote-mcp.js +62 -0
- package/lib/agents/middleware/tool-wrap.js +52 -0
- package/lib/compile.js +6 -2
- package/lib/eval/Judge.js +239 -0
- package/lib/eval/eval-describe.js +69 -0
- package/lib/eval/eval-run.js +147 -0
- package/lib/eval/index.js +6 -0
- package/lib/eval/metrics.js +53 -0
- package/lib/eval/span-collector.js +50 -0
- package/lib/index.js +4 -3
- package/lib/models/aicore.js +23 -6
- package/lib/preview/chat.html +394 -69
- package/lib/protocol/agent-card.js +6 -2
- package/lib/protocol/persistence/checkpoint-saver.js +4 -1
- package/lib/protocol/persistence/cleanup.js +84 -0
- package/lib/sidecar.js +1 -1
- package/lib/telemetry/chat-tracing.js +61 -8
- package/lib/telemetry/mlflow/credentials.js +79 -0
- package/lib/telemetry/mlflow/evaluation.js +38 -0
- package/lib/telemetry/mlflow/exporter/DatabricksExporter.js +75 -0
- package/lib/telemetry/mlflow/exporter/MlflowExporter.js +115 -0
- package/lib/telemetry/mlflow/exporter/index.js +23 -0
- package/lib/telemetry/mlflow/index.js +16 -0
- package/lib/telemetry/mlflow/prompts.js +123 -0
- package/lib/telemetry/mlflow/tracing.js +264 -0
- package/lib/telemetry/tool-tracing.js +27 -54
- package/lib/telemetry/tracing.js +3 -1
- package/lib/utils/markdown.js +1 -11
- package/lib/utils/resilience.js +133 -0
- package/lib/utils/utils.js +17 -0
- package/package.json +24 -15
- package/{index.cds → srv/entities.cds} +11 -0
- package/srv/handlers/chat.js +278 -0
- package/srv/handlers/graph-executor.js +48 -37
- package/srv/handlers/index.js +8 -0
- package/srv/handlers/mcp-tools.js +11 -47
- package/srv/handlers/sub-agent-tools.js +12 -3
- package/srv/handlers/tools.js +9 -5
- package/index.js +0 -0
- package/lib/telemetry/mlflow.js +0 -290
|
@@ -0,0 +1,264 @@
|
|
|
1
|
+
import cds from "@sap/cds"
|
|
2
|
+
import { resolveMlflowCredentials, resolveExperimentId } from "./credentials.js"
|
|
3
|
+
import { resolvePromptName, linkedPromptsAttr } from "./prompts.js"
|
|
4
|
+
|
|
5
|
+
// Validated wrapper: throws when @Core.SchemaVersion is present but non-numeric,
|
|
6
|
+
// since MLflow experiment IDs must be int64.
|
|
7
|
+
function _resolveExperimentId() {
|
|
8
|
+
const srvName = cds.context?.["agent.service"]
|
|
9
|
+
if (srvName) {
|
|
10
|
+
const def = cds.context?.model?.definitions?.[srvName] || cds.services?.[srvName]?.definition
|
|
11
|
+
const annotated = def?.["@Core.SchemaVersion"]
|
|
12
|
+
if (annotated) {
|
|
13
|
+
const id = String(annotated)
|
|
14
|
+
if (!/^\d+$/.test(id))
|
|
15
|
+
throw new Error(
|
|
16
|
+
`@Core.SchemaVersion on "${srvName}" must be a numeric string (MLflow experiment ID requires int64). Got: "${id}"`,
|
|
17
|
+
)
|
|
18
|
+
return id
|
|
19
|
+
}
|
|
20
|
+
}
|
|
21
|
+
return resolveExperimentId()
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* Build mlflow.* span attributes for a given span type.
|
|
26
|
+
* @param {"LLM"|"AGENT"|"TOOL"|"CHAIN"|"RETRIEVER"} spanType
|
|
27
|
+
* @param {{ inputs?: any, outputs?: any, model?: string, provider?: string, functionName?: string, tokenUsage?: object }} [opts]
|
|
28
|
+
* @returns {Record<string, string>} Attributes to set on span, or {} when disabled
|
|
29
|
+
*/
|
|
30
|
+
export function mlflowAttrs(spanType, opts = {}) {
|
|
31
|
+
if (!cds.env.agents?.mlflow) return {}
|
|
32
|
+
const experimentId = _resolveExperimentId()
|
|
33
|
+
const attrs = {}
|
|
34
|
+
if (experimentId) attrs["mlflow.experimentId"] = String(experimentId)
|
|
35
|
+
attrs["mlflow.spanType"] = spanType
|
|
36
|
+
if (opts.model) attrs["mlflow.llm.model"] = String(opts.model)
|
|
37
|
+
if (opts.provider) attrs["mlflow.llm.provider"] = String(opts.provider)
|
|
38
|
+
if (opts.functionName) attrs["mlflow.spanFunctionName"] = String(opts.functionName)
|
|
39
|
+
if (opts.inputs !== undefined) attrs["mlflow.spanInputs"] = JSON.stringify(opts.inputs)
|
|
40
|
+
if (opts.outputs !== undefined) attrs["mlflow.spanOutputs"] = JSON.stringify(opts.outputs)
|
|
41
|
+
if (opts.tokenUsage) attrs["mlflow.chat.tokenUsage"] = JSON.stringify(opts.tokenUsage)
|
|
42
|
+
return attrs
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* Build mlflow trace-level tag attributes (set on root/workflow span).
|
|
47
|
+
* Uses mlflow.traceTag.* prefix so MLflow server extracts them as trace tags.
|
|
48
|
+
* Also sets OTel semconv user.id and session.id which MLflow reads natively.
|
|
49
|
+
* When a system prompt is registered, adds mlflow.traceTag.mlflow.linkedPrompts
|
|
50
|
+
* so MLflow links this trace to the prompt version — no extra REST call needed.
|
|
51
|
+
* @returns {Record<string, string>} Attributes to set on root span, or {} when disabled
|
|
52
|
+
*/
|
|
53
|
+
export function mlflowTraceAttrs() {
|
|
54
|
+
if (!cds.env.agents?.mlflow) return {}
|
|
55
|
+
const session = String(cds.context?.["agent.context.id"] || "")
|
|
56
|
+
const user = String(cds.context?.user?.id || "")
|
|
57
|
+
const tenant = String(cds.context?.tenant || "")
|
|
58
|
+
const attrs = {
|
|
59
|
+
// OTel semconv keys MLflow reads natively for user/session display
|
|
60
|
+
"session.id": session,
|
|
61
|
+
"user.id": user,
|
|
62
|
+
// mlflow.traceTag.* prefix for custom trace tags
|
|
63
|
+
"mlflow.traceTag.session": session,
|
|
64
|
+
"mlflow.traceTag.user": user,
|
|
65
|
+
"mlflow.traceTag.tenant": tenant,
|
|
66
|
+
}
|
|
67
|
+
return attrs
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
export function linkTraceToPrompt() {
|
|
71
|
+
if (!cds.env.agents?.mlflow) return {}
|
|
72
|
+
const srvName = cds.context?.["agent.service"]
|
|
73
|
+
const srv = cds.services[srvName]
|
|
74
|
+
const linked = srv ? linkedPromptsAttr(resolvePromptName(srv)) : null
|
|
75
|
+
if (linked) {
|
|
76
|
+
return { "mlflow.traceTag.mlflow.linkedPrompts": linked }
|
|
77
|
+
}
|
|
78
|
+
return {}
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* Spread attributes object onto an OTel span (null-safe).
|
|
83
|
+
* Coerces all values to strings to prevent [object Object] in downstream systems.
|
|
84
|
+
* @param {import("@opentelemetry/api").Span | null} span
|
|
85
|
+
* @param {Record<string, string>} attrs
|
|
86
|
+
*/
|
|
87
|
+
export function setSpanAttrs(span, attrs) {
|
|
88
|
+
if (!cds.env.agents?.mlflow) return
|
|
89
|
+
if (!span) return
|
|
90
|
+
for (const [k, v] of Object.entries(attrs)) {
|
|
91
|
+
if (v !== undefined && v !== null) span.setAttribute(k, v)
|
|
92
|
+
}
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
/**
|
|
96
|
+
* SpanProcessor that routes spans to per-experiment-ID BatchSpanProcessors.
|
|
97
|
+
* One OTLPTraceExporter per unique mlflow.experimentId
|
|
98
|
+
*/
|
|
99
|
+
export class RoutingSpanProcessor {
|
|
100
|
+
/**
|
|
101
|
+
* @param {{ url: string, token?: string, getAuthHeaders?: () => Promise<Record<string,string>>,
|
|
102
|
+
* ucTableName?: string, BatchSpanProcessor: any, OTLPTraceExporter: any }} opts
|
|
103
|
+
*
|
|
104
|
+
* Auth: `getAuthHeaders` (async, OAuth-capable) takes precedence over static `token`.
|
|
105
|
+
* UC: When `ucTableName` is provided it is added as `X-Databricks-UC-Table-Name` header.
|
|
106
|
+
*/
|
|
107
|
+
constructor({ url, token, getAuthHeaders, ucTableName, BatchSpanProcessor, OTLPTraceExporter }) {
|
|
108
|
+
this._url = url
|
|
109
|
+
// OAuth factory (async) takes precedence over static token.
|
|
110
|
+
// When only a static token is given, keep headers as a plain object so
|
|
111
|
+
// OTLPTraceExporter can use them synchronously (and tests can inspect them directly).
|
|
112
|
+
this._getAuthHeaders = getAuthHeaders ?? null
|
|
113
|
+
this._staticToken = token ?? null
|
|
114
|
+
this._ucTableName = ucTableName
|
|
115
|
+
this._BatchSpanProcessor = BatchSpanProcessor
|
|
116
|
+
this._OTLPTraceExporter = OTLPTraceExporter
|
|
117
|
+
this._processors = new Map()
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
onStart() {}
|
|
121
|
+
|
|
122
|
+
onEnd(span) {
|
|
123
|
+
const expId = span.attributes?.["mlflow.experimentId"]
|
|
124
|
+
if (!expId) return
|
|
125
|
+
// If this span carries a run ID, update the per-experiment map so the
|
|
126
|
+
// OTLP exporter for this experiment sends x-mlflow-run-id on the next flush.
|
|
127
|
+
const sourceRun = span.attributes?.["mlflow.sourceRun"]
|
|
128
|
+
if (sourceRun) _runIdByExperiment.set(expId, sourceRun)
|
|
129
|
+
else if (!_runIdByExperiment.has(expId)) _runIdByExperiment.delete(expId)
|
|
130
|
+
this._getOrCreate(expId).onEnd(span)
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
_getOrCreate(experimentId) {
|
|
134
|
+
if (this._processors.has(experimentId)) return this._processors.get(experimentId)
|
|
135
|
+
const ucTableName = this._ucTableName
|
|
136
|
+
const expHeader = { "x-mlflow-experiment-id": experimentId }
|
|
137
|
+
const ucHeader = ucTableName ? { "X-Databricks-UC-Table-Name": ucTableName } : {}
|
|
138
|
+
|
|
139
|
+
let headers
|
|
140
|
+
if (this._getAuthHeaders) {
|
|
141
|
+
// OAuth / async path — factory is called at export time, so run ID is resolved
|
|
142
|
+
// dynamically from _runIdByExperiment without needing to recreate the exporter.
|
|
143
|
+
const getAuthHeaders = this._getAuthHeaders
|
|
144
|
+
headers = async () => {
|
|
145
|
+
const runId = _runIdByExperiment.get(experimentId)
|
|
146
|
+
return {
|
|
147
|
+
...(await getAuthHeaders()),
|
|
148
|
+
...expHeader,
|
|
149
|
+
...ucHeader,
|
|
150
|
+
...(runId ? { "x-mlflow-run-id": runId } : {}),
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
} else {
|
|
154
|
+
// Static token path — plain object so OTLPTraceExporter and tests can read it directly.
|
|
155
|
+
const authHeader = this._staticToken ? { Authorization: `Bearer ${this._staticToken}` } : {}
|
|
156
|
+
headers = { ...authHeader, ...expHeader, ...ucHeader }
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
const exporter = new this._OTLPTraceExporter({ url: this._url, headers })
|
|
160
|
+
const proc = new this._BatchSpanProcessor(exporter)
|
|
161
|
+
this._processors.set(experimentId, proc)
|
|
162
|
+
return proc
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
async forceFlush() {
|
|
166
|
+
await Promise.all([...this._processors.values()].map((p) => p.forceFlush()))
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
async shutdown() {
|
|
170
|
+
await Promise.all([...this._processors.values()].map((p) => p.shutdown()))
|
|
171
|
+
this._processors.clear()
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
/** Module-level singleton — set by setupMlflowExporter(), used as fast path by flushMlflowTraces(). */
|
|
176
|
+
let _routing = null
|
|
177
|
+
|
|
178
|
+
/**
|
|
179
|
+
* Run IDs keyed by experiment ID — populated by RoutingSpanProcessor.onEnd()
|
|
180
|
+
* when a span carries mlflow.sourceRun. The OTLP header factory reads this
|
|
181
|
+
* at export time to include x-mlflow-run-id on the batch.
|
|
182
|
+
*/
|
|
183
|
+
const _runIdByExperiment = new Map()
|
|
184
|
+
|
|
185
|
+
/**
|
|
186
|
+
* Force-flush all pending MLflow OTLP spans to the remote server.
|
|
187
|
+
*/
|
|
188
|
+
export async function flushMlflowTraces() {
|
|
189
|
+
try {
|
|
190
|
+
const { trace } = await import("@opentelemetry/api")
|
|
191
|
+
const provider = trace.getTracerProvider()
|
|
192
|
+
const delegate = provider.getDelegate?.() || provider
|
|
193
|
+
if (typeof delegate.forceFlush === "function") {
|
|
194
|
+
await delegate.forceFlush()
|
|
195
|
+
return
|
|
196
|
+
}
|
|
197
|
+
} catch {
|
|
198
|
+
// @opentelemetry/api not available — fall through
|
|
199
|
+
}
|
|
200
|
+
// Fallback: flush via the singleton registered in this module instance
|
|
201
|
+
await _routing?.forceFlush()
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
/**
|
|
205
|
+
* Register MLflow OTLP RoutingSpanProcessor.
|
|
206
|
+
* No-op when credentials are missing.
|
|
207
|
+
*/
|
|
208
|
+
export async function setupMlflowExporter() {
|
|
209
|
+
if (!cds.env.agents?.mlflow) return
|
|
210
|
+
|
|
211
|
+
const LOG = cds.log("agent")
|
|
212
|
+
try {
|
|
213
|
+
const mlflowCreds = resolveMlflowCredentials()
|
|
214
|
+
const creds = cds.env.requires?.mlflow?.credentials || {}
|
|
215
|
+
const mlflowEndpoint =
|
|
216
|
+
creds.MLFLOW_OTLP_ENDPOINT || (mlflowCreds?.host && `${mlflowCreds.host}/v1/traces`)
|
|
217
|
+
|
|
218
|
+
if (!mlflowEndpoint) {
|
|
219
|
+
LOG.warn(
|
|
220
|
+
"MLflow: credentials missing (MLFLOW_HOST or MLFLOW_OTLP_ENDPOINT) — export disabled",
|
|
221
|
+
)
|
|
222
|
+
return
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
const ucTableName = mlflowCreds?.uc
|
|
226
|
+
? `${mlflowCreds.uc.catalog}.${mlflowCreds.uc.schema}.${mlflowCreds.uc.tablePrefix}_otel_spans`
|
|
227
|
+
: undefined
|
|
228
|
+
|
|
229
|
+
const { trace } = await import("@opentelemetry/api")
|
|
230
|
+
const provider = trace.getTracerProvider()
|
|
231
|
+
const delegate = provider.getDelegate?.() || provider
|
|
232
|
+
|
|
233
|
+
const { BatchSpanProcessor } = await import("@opentelemetry/sdk-trace-base")
|
|
234
|
+
const { OTLPTraceExporter } = await import("@opentelemetry/exporter-trace-otlp-proto")
|
|
235
|
+
|
|
236
|
+
const routing = new RoutingSpanProcessor({
|
|
237
|
+
url: mlflowEndpoint,
|
|
238
|
+
getAuthHeaders: mlflowCreds?.getAuthHeaders ?? (async () => ({})),
|
|
239
|
+
ucTableName,
|
|
240
|
+
BatchSpanProcessor,
|
|
241
|
+
OTLPTraceExporter,
|
|
242
|
+
})
|
|
243
|
+
|
|
244
|
+
if (delegate.addSpanProcessor) {
|
|
245
|
+
// OTEL SDK v1
|
|
246
|
+
delegate.addSpanProcessor(routing)
|
|
247
|
+
} else if (delegate._activeSpanProcessor?._spanProcessors) {
|
|
248
|
+
// OTEL SDK v2: no addSpanProcessor — push into MultiSpanProcessor
|
|
249
|
+
delegate._activeSpanProcessor._spanProcessors.push(routing)
|
|
250
|
+
} else {
|
|
251
|
+
LOG.warn(
|
|
252
|
+
"MLflow: no TracerProvider with addSpanProcessor — ensure @cap-js/telemetry is loaded",
|
|
253
|
+
)
|
|
254
|
+
return
|
|
255
|
+
}
|
|
256
|
+
_routing = routing
|
|
257
|
+
LOG.info("MLflow: routing span processor added (per-experiment export)", {
|
|
258
|
+
endpoint: mlflowEndpoint,
|
|
259
|
+
...(ucTableName && { ucTableName }),
|
|
260
|
+
})
|
|
261
|
+
} catch (err) {
|
|
262
|
+
LOG.error("MLflow: failed to configure OTLP export", { error: err.message })
|
|
263
|
+
}
|
|
264
|
+
}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import cds from "@sap/cds"
|
|
2
2
|
import * as metrics from "./metrics.js"
|
|
3
|
-
import { mlflowAttrs, setSpanAttrs } from "./mlflow.js"
|
|
3
|
+
import { mlflowAttrs, setSpanAttrs } from "./mlflow/index.js"
|
|
4
4
|
import { audit } from "../utils/utils.js"
|
|
5
5
|
|
|
6
6
|
const LOG = cds.log("agents")
|
|
@@ -45,6 +45,7 @@ export function _patchToolsProto(proto) {
|
|
|
45
45
|
const toolName = this.name || this.constructor.name
|
|
46
46
|
|
|
47
47
|
const invoke = async (span) => {
|
|
48
|
+
const t0 = Date.now()
|
|
48
49
|
if (span) {
|
|
49
50
|
span.setAttribute("gen_ai.operation.name", "execute_tool")
|
|
50
51
|
span.setAttribute("gen_ai.provider.name", "langchain")
|
|
@@ -53,13 +54,16 @@ export function _patchToolsProto(proto) {
|
|
|
53
54
|
setSpanAttrs(span, mlflowAttrs("TOOL", { inputs, functionName: toolName }))
|
|
54
55
|
if (LOG._debug) span.setAttribute("gen_ai.tool.call.arguments", JSON.stringify(inputs))
|
|
55
56
|
}
|
|
56
|
-
const
|
|
57
|
+
const taskId = config?.configurable?._taskId || cds.context?.["agent.task.id"]
|
|
58
|
+
let outcome
|
|
59
|
+
let errorMessage
|
|
60
|
+
let duration
|
|
57
61
|
try {
|
|
58
62
|
const result = await original.call(this, args, config)
|
|
59
|
-
|
|
63
|
+
duration = Date.now() - t0
|
|
60
64
|
|
|
61
65
|
const semanticError = result?.artifact?.isError === true
|
|
62
|
-
|
|
66
|
+
outcome = semanticError ? "error" : "success"
|
|
63
67
|
const outputs = result?.content ?? result
|
|
64
68
|
|
|
65
69
|
if (span) {
|
|
@@ -81,36 +85,16 @@ export function _patchToolsProto(proto) {
|
|
|
81
85
|
span.setAttribute("gen_ai.tool.call.result", outputs)
|
|
82
86
|
}
|
|
83
87
|
}
|
|
84
|
-
|
|
85
|
-
metrics.toolInvocations.add(1, {
|
|
86
|
-
"sap.tenantId": cds.context?.tenant || "anonymous",
|
|
87
|
-
"agent.service": config?.configurable?._service || cds.context?.["agent.service"],
|
|
88
|
-
tool: toolName,
|
|
89
|
-
outcome,
|
|
90
|
-
})
|
|
91
|
-
|
|
92
|
-
// Audit: record tool invocation
|
|
93
|
-
const taskId = config?.configurable?._taskId || cds.context?.["agent.task.id"]
|
|
94
|
-
if (taskId) {
|
|
88
|
+
if (semanticError) {
|
|
95
89
|
const resultStr = typeof outputs === "string" ? outputs : JSON.stringify(outputs)
|
|
96
|
-
|
|
97
|
-
data: {
|
|
98
|
-
taskId,
|
|
99
|
-
service: config?.configurable?._service || cds.context?.["agent.service"],
|
|
100
|
-
tool: toolName,
|
|
101
|
-
args,
|
|
102
|
-
outcome,
|
|
103
|
-
...(semanticError
|
|
104
|
-
? { error: resultStr?.slice(0, 2000) }
|
|
105
|
-
: { result: resultStr?.slice(0, 2000) }),
|
|
106
|
-
duration,
|
|
107
|
-
},
|
|
108
|
-
})
|
|
90
|
+
errorMessage = resultStr
|
|
109
91
|
}
|
|
110
92
|
|
|
111
93
|
return result
|
|
112
94
|
} catch (err) {
|
|
113
|
-
|
|
95
|
+
duration = Date.now() - t0
|
|
96
|
+
outcome = "error"
|
|
97
|
+
errorMessage = err.message
|
|
114
98
|
if (span) {
|
|
115
99
|
span.setAttribute("gen_ai.tool.call.outcome", "error")
|
|
116
100
|
span.setAttribute("error.type", err.constructor?.name || "Error")
|
|
@@ -119,36 +103,25 @@ export function _patchToolsProto(proto) {
|
|
|
119
103
|
setSpanAttrs(span, mlflowAttrs("TOOL", { outputs: { error: err.message } }))
|
|
120
104
|
}
|
|
121
105
|
|
|
106
|
+
throw err
|
|
107
|
+
} finally {
|
|
122
108
|
metrics.toolInvocations.add(1, {
|
|
123
109
|
"sap.tenantId": cds.context?.tenant || "anonymous",
|
|
124
110
|
"agent.service": config?.configurable?._service || cds.context?.["agent.service"],
|
|
125
111
|
tool: toolName,
|
|
126
|
-
outcome
|
|
112
|
+
outcome,
|
|
113
|
+
})
|
|
114
|
+
audit("ToolInvocation", {
|
|
115
|
+
data: {
|
|
116
|
+
taskId,
|
|
117
|
+
service: config?.configurable?._service || cds.context?.["agent.service"],
|
|
118
|
+
tool: toolName,
|
|
119
|
+
args,
|
|
120
|
+
outcome,
|
|
121
|
+
...(outcome === "error" ? { error: errorMessage } : {}),
|
|
122
|
+
duration,
|
|
123
|
+
},
|
|
127
124
|
})
|
|
128
|
-
|
|
129
|
-
// Audit: record failed tool invocation
|
|
130
|
-
const taskId = config?.configurable?._taskId || cds.context?.["agent.task.id"]
|
|
131
|
-
if (taskId) {
|
|
132
|
-
audit("ToolInvocation", {
|
|
133
|
-
data: {
|
|
134
|
-
taskId,
|
|
135
|
-
service: config?.configurable?._service || cds.context?.["agent.service"],
|
|
136
|
-
tool: toolName,
|
|
137
|
-
args,
|
|
138
|
-
outcome: "error",
|
|
139
|
-
error: err.message,
|
|
140
|
-
duration,
|
|
141
|
-
},
|
|
142
|
-
})
|
|
143
|
-
}
|
|
144
|
-
|
|
145
|
-
// Swallow error: return as string so LLM can see it and retry.
|
|
146
|
-
// Deep agents use ToolNode with handleToolErrors (default: true) which
|
|
147
|
-
// already wraps errors as ToolMessages. Custom graphs that invoke tools
|
|
148
|
-
// directly also benefit from error swallowing — the graph continues and
|
|
149
|
-
// the LLM can retry with a corrected query.
|
|
150
|
-
return `Error: ${err.message}`
|
|
151
|
-
} finally {
|
|
152
125
|
if (span) span.end()
|
|
153
126
|
}
|
|
154
127
|
}
|
package/lib/telemetry/tracing.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import cds from "@sap/cds"
|
|
2
2
|
import * as metrics from "./metrics.js"
|
|
3
|
-
import { mlflowAttrs, setSpanAttrs } from "./mlflow.js"
|
|
3
|
+
import { mlflowAttrs, setSpanAttrs } from "./mlflow/index.js"
|
|
4
4
|
|
|
5
5
|
const LOG = cds.log("agents")
|
|
6
6
|
|
|
@@ -121,6 +121,8 @@ function _patchRunnableSequenceProto(proto) {
|
|
|
121
121
|
proto.invoke = async function (input, config) {
|
|
122
122
|
const tracer = metrics.getTracer()
|
|
123
123
|
if (!tracer) return original.call(this, input, config)
|
|
124
|
+
// Skip invokes fired during graph compilation, which would create orphan spans.
|
|
125
|
+
if (!cds.context?.["agent.service"]) return original.call(this, input, config)
|
|
124
126
|
|
|
125
127
|
const name = this.name || config?.runName || this.first?.name || "RunnableSequence"
|
|
126
128
|
return tracer.startActiveSpan(`workflow RunnableSequence ${name}`, async (span) => {
|
package/lib/utils/markdown.js
CHANGED
|
@@ -1,18 +1,8 @@
|
|
|
1
1
|
import cds from "@sap/cds"
|
|
2
|
+
import { slugified } from "./utils.js"
|
|
2
3
|
const { path, fs } = cds.utils
|
|
3
4
|
const LOG = cds.log("agents")
|
|
4
5
|
|
|
5
|
-
/**
|
|
6
|
-
* Mirrors CAP's internal slug rules used for service path generation.
|
|
7
|
-
*/
|
|
8
|
-
export const slugified = (name) =>
|
|
9
|
-
/[^.]+$/
|
|
10
|
-
.exec(name)[0]
|
|
11
|
-
.replace(/Service$/, "")
|
|
12
|
-
.replace(/_/g, "-")
|
|
13
|
-
.replace(/([a-z0-9])([A-Z])/g, (_m, c, C) => c + "-" + C)
|
|
14
|
-
.toLowerCase()
|
|
15
|
-
|
|
16
6
|
/**
|
|
17
7
|
* Absolute filesystem directory of the `.cds` source file that defines `srv`.
|
|
18
8
|
*/
|
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
import cds from "@sap/cds"
|
|
2
|
+
|
|
3
|
+
const LOG = cds.log("agents")
|
|
4
|
+
|
|
5
|
+
const DEFAULT_TIMEOUT_MS = 10_000
|
|
6
|
+
const DEFAULT_RETRIES = 3
|
|
7
|
+
const BACKOFF_BASE_MS = 1_000
|
|
8
|
+
const BACKOFF_MAX_MS = 32_000
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* Resilience middleware for the SAP AI SDK HTTP layer.
|
|
12
|
+
* Contract: (options: { fn, context: { uri, tenantId } }) => (arg) => Promise
|
|
13
|
+
* Compose right-to-left: [timeout(t), circuitBreaker()] wraps timeout around
|
|
14
|
+
* circuit breaker around the request.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
export function timeout(ms = DEFAULT_TIMEOUT_MS) {
|
|
18
|
+
if (ms < 0) throw new Error("Timeout must be greater or equal to 0.")
|
|
19
|
+
if (ms > 0 && ms < 10)
|
|
20
|
+
LOG.warn(`The timeout of ${ms} ms is too low. Make sure this is not intentional.`)
|
|
21
|
+
|
|
22
|
+
return ({ fn, context }) =>
|
|
23
|
+
(arg) => {
|
|
24
|
+
if (ms === 0) return fn(arg)
|
|
25
|
+
const controller = new AbortController()
|
|
26
|
+
const signal = arg?.signal
|
|
27
|
+
? AbortSignal.any([arg.signal, controller.signal])
|
|
28
|
+
: controller.signal
|
|
29
|
+
const request = arg && typeof arg === "object" ? { ...arg, signal } : arg
|
|
30
|
+
let timer
|
|
31
|
+
const call = fn(request).finally(() => clearTimeout(timer))
|
|
32
|
+
const race = new Promise((_, reject) => {
|
|
33
|
+
timer = setTimeout(() => {
|
|
34
|
+
controller.abort()
|
|
35
|
+
reject(new Error(`Request to URL: ${context?.uri} ran into a timeout after ${ms}ms.`))
|
|
36
|
+
}, ms)
|
|
37
|
+
})
|
|
38
|
+
return Promise.race([call, race])
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
// Retries on 5xx / network errors; bails immediately on 4xx.
|
|
43
|
+
export function retry(retries = DEFAULT_RETRIES) {
|
|
44
|
+
if (retries < 0) throw new Error("Number of retries must be greater or equal to 0.")
|
|
45
|
+
|
|
46
|
+
return ({ fn }) =>
|
|
47
|
+
async (arg) => {
|
|
48
|
+
for (let attempt = 0; ; attempt++) {
|
|
49
|
+
try {
|
|
50
|
+
return await fn(arg) // eslint-disable-line no-await-in-loop
|
|
51
|
+
} catch (error) {
|
|
52
|
+
const status = error?.response?.status
|
|
53
|
+
if (status == null)
|
|
54
|
+
LOG.debug("HTTP request failed without a response status. Rethrowing.")
|
|
55
|
+
else if (`${status}`.startsWith("4"))
|
|
56
|
+
throw Object.assign(new Error(`Request failed with status code ${status}`), {
|
|
57
|
+
cause: error,
|
|
58
|
+
})
|
|
59
|
+
if (attempt >= retries) throw error
|
|
60
|
+
const delay =
|
|
61
|
+
Math.min(BACKOFF_MAX_MS, BACKOFF_BASE_MS * 2 ** attempt) * (1 + Math.random())
|
|
62
|
+
await new Promise((r) => setTimeout(r, delay)) // eslint-disable-line no-await-in-loop
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
// One breaker per uri. Opens after volumeThreshold calls exceed the
|
|
69
|
+
// error rate; fails fast with EOPENBREAKER while open. 4xx never trips it.
|
|
70
|
+
export const circuitBreakers = {}
|
|
71
|
+
|
|
72
|
+
export function circuitBreaker() {
|
|
73
|
+
return ({ fn, context }) =>
|
|
74
|
+
(arg) => {
|
|
75
|
+
const key = context?.uri ?? "default"
|
|
76
|
+
circuitBreakers[key] ??= new CircuitBreaker()
|
|
77
|
+
return circuitBreakers[key].fire(fn, arg)
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
class CircuitBreaker {
|
|
82
|
+
#state = "closed"
|
|
83
|
+
#openedAt = 0
|
|
84
|
+
#window = []
|
|
85
|
+
|
|
86
|
+
async fire(fn, arg) {
|
|
87
|
+
const now = Date.now()
|
|
88
|
+
const opts = cds.env.agents.circuitBreaker
|
|
89
|
+
if (this.#state === "open") {
|
|
90
|
+
if (now - this.#openedAt < opts.resetTimeout)
|
|
91
|
+
throw Object.assign(new Error("Breaker is open"), { code: "EOPENBREAKER" })
|
|
92
|
+
this.#state = "half-open"
|
|
93
|
+
}
|
|
94
|
+
try {
|
|
95
|
+
const result = await fn(arg)
|
|
96
|
+
if (this.#state === "half-open") this.#reset("closed")
|
|
97
|
+
else this.#record(true, now)
|
|
98
|
+
return result
|
|
99
|
+
} catch (error) {
|
|
100
|
+
if (!`${error?.response?.status}`.startsWith("4")) this.#onError(now)
|
|
101
|
+
throw error
|
|
102
|
+
}
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
#onError(now) {
|
|
106
|
+
if (this.#state === "half-open") {
|
|
107
|
+
this.#reset("open", now)
|
|
108
|
+
return
|
|
109
|
+
}
|
|
110
|
+
this.#record(false, now)
|
|
111
|
+
if (this.#shouldOpen()) this.#reset("open", now)
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
#record(success, now) {
|
|
115
|
+
const { rollingCountTimeout } = cds.env.agents.circuitBreaker
|
|
116
|
+
const cutoff = now - rollingCountTimeout
|
|
117
|
+
while (this.#window.length && this.#window[0].t < cutoff) this.#window.shift()
|
|
118
|
+
this.#window.push({ success, t: now })
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
#shouldOpen() {
|
|
122
|
+
const { volumeThreshold, errorThresholdPercentage } = cds.env.agents.circuitBreaker
|
|
123
|
+
if (this.#window.length < volumeThreshold) return false
|
|
124
|
+
const failures = this.#window.filter((o) => !o.success).length
|
|
125
|
+
return (failures / this.#window.length) * 100 >= errorThresholdPercentage
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
#reset(state, openedAt = 0) {
|
|
129
|
+
this.#state = state
|
|
130
|
+
this.#openedAt = openedAt
|
|
131
|
+
this.#window = []
|
|
132
|
+
}
|
|
133
|
+
}
|
package/lib/utils/utils.js
CHANGED
|
@@ -10,6 +10,23 @@ function resolveI18n(value, locale) {
|
|
|
10
10
|
return value
|
|
11
11
|
}
|
|
12
12
|
|
|
13
|
+
export function getAgentLogger(srv) {
|
|
14
|
+
const slug = slugified(srv.name)
|
|
15
|
+
const agentSlug = slug.endsWith("-agent") ? slug : `${slug}-agent`
|
|
16
|
+
return cds.log("agents:" + srv.name, { label: agentSlug })
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* Mirrors CAP's internal slug rules used for service path generation.
|
|
21
|
+
*/
|
|
22
|
+
export const slugified = (name) =>
|
|
23
|
+
/[^.]+$/
|
|
24
|
+
.exec(name)[0]
|
|
25
|
+
.replace(/Service$/, "")
|
|
26
|
+
.replace(/_/g, "-")
|
|
27
|
+
.replace(/([a-z0-9])([A-Z])/g, (_m, c, C) => c + "-" + C)
|
|
28
|
+
.toLowerCase()
|
|
29
|
+
|
|
13
30
|
export function getDescription(obj, locale) {
|
|
14
31
|
locale = locale || cds.context?.locale || "en"
|
|
15
32
|
|