@cap-js/agents 0.9.1 → 0.9.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/README.md +2 -1
  2. package/cds-plugin.js +2 -2
  3. package/lib/agents/markdown/backends/outputs-backend.js +16 -12
  4. package/lib/agents/markdown/backends/readonly-backend.js +48 -0
  5. package/lib/agents/markdown/backends/uploads-backend.js +30 -13
  6. package/lib/agents/markdown/deep-agent.js +15 -26
  7. package/lib/agents/middleware/content-filter.js +3 -2
  8. package/lib/agents/middleware/index.js +11 -0
  9. package/lib/agents/middleware/remote-mcp.js +62 -0
  10. package/lib/agents/middleware/tool-wrap.js +52 -0
  11. package/lib/compile.js +6 -2
  12. package/lib/eval/Judge.js +239 -0
  13. package/lib/eval/eval-describe.js +69 -0
  14. package/lib/eval/eval-run.js +147 -0
  15. package/lib/eval/index.js +6 -0
  16. package/lib/eval/metrics.js +53 -0
  17. package/lib/eval/span-collector.js +50 -0
  18. package/lib/index.js +4 -3
  19. package/lib/models/aicore.js +23 -6
  20. package/lib/preview/chat.html +394 -69
  21. package/lib/protocol/agent-card.js +6 -2
  22. package/lib/protocol/persistence/checkpoint-saver.js +4 -1
  23. package/lib/protocol/persistence/cleanup.js +84 -0
  24. package/lib/sidecar.js +1 -1
  25. package/lib/telemetry/chat-tracing.js +61 -8
  26. package/lib/telemetry/mlflow/credentials.js +79 -0
  27. package/lib/telemetry/mlflow/evaluation.js +38 -0
  28. package/lib/telemetry/mlflow/exporter/DatabricksExporter.js +75 -0
  29. package/lib/telemetry/mlflow/exporter/MlflowExporter.js +115 -0
  30. package/lib/telemetry/mlflow/exporter/index.js +23 -0
  31. package/lib/telemetry/mlflow/index.js +16 -0
  32. package/lib/telemetry/mlflow/prompts.js +123 -0
  33. package/lib/telemetry/mlflow/tracing.js +264 -0
  34. package/lib/telemetry/tool-tracing.js +27 -54
  35. package/lib/telemetry/tracing.js +3 -1
  36. package/lib/utils/markdown.js +1 -11
  37. package/lib/utils/resilience.js +133 -0
  38. package/lib/utils/utils.js +17 -0
  39. package/package.json +24 -15
  40. package/{index.cds → srv/entities.cds} +11 -0
  41. package/srv/handlers/chat.js +278 -0
  42. package/srv/handlers/graph-executor.js +48 -37
  43. package/srv/handlers/index.js +8 -0
  44. package/srv/handlers/mcp-tools.js +11 -47
  45. package/srv/handlers/sub-agent-tools.js +12 -3
  46. package/srv/handlers/tools.js +9 -5
  47. package/index.js +0 -0
  48. package/lib/telemetry/mlflow.js +0 -290
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@cap-js/agents",
3
- "version": "0.9.1",
3
+ "version": "0.9.3",
4
4
  "description": "CDS plugin for building agents",
5
5
  "author": "SAP SE (https://www.sap.com)",
6
6
  "license": "Apache-2.0",
@@ -11,16 +11,14 @@
11
11
  },
12
12
  "type": "module",
13
13
  "exports": {
14
- ".": "./index.js",
15
14
  "./package.json": "./package.json",
16
15
  "./cds-plugin": "./cds-plugin.js",
17
- "./index.cds": "./index.cds",
16
+ "./eval": "./lib/eval/index.js",
17
+ "./lib/index.cjs": "./lib/index.cjs",
18
18
  "./lib/models/*": "./lib/models/*.js",
19
- "./srv/langgraph-executor-srv": "./srv/langgraph-executor-srv.js",
20
- "./srv/*": "./srv/*",
21
- "./lib/*": "./lib/*"
19
+ "./srv/entities.cds": "./srv/entities.cds",
20
+ "./srv/push-notification-srv.js": "./srv/push-notification-srv.js"
22
21
  },
23
- "main": "index.js",
24
22
  "scripts": {
25
23
  "lint": "npx eslint .",
26
24
  "test": "CDS_ENV=test npx vitest run --config vitest.config.js",
@@ -30,15 +28,15 @@
30
28
  "watch:hybrid": "DEBUG=agents cds bind --exec -- cds w tests/projects/bookshop",
31
29
  "watch:with-claude": "cds w tests/projects/bookshop --profile with-claude",
32
30
  "watch:deep-agent": "cds w tests/projects/deep-agent --profile hybrid",
31
+ "docs:audit": "node .scripts/generate-audit-docs.js",
32
+ "docs:audit:check": "node .scripts/generate-audit-docs.js --check",
33
33
  "prettier": "npx -y prettier@3 --write .",
34
34
  "prettier:check": "npx -y prettier@3 --check ."
35
35
  },
36
36
  "files": [
37
- "index.js",
38
- "index.cds",
37
+ "cds-plugin.js",
39
38
  "lib",
40
39
  "srv",
41
- "cds-plugin.js",
42
40
  "_i18n"
43
41
  ],
44
42
  "dependencies": {
@@ -52,7 +50,6 @@
52
50
  "@sap-ai-sdk/langchain": "^2.13.0",
53
51
  "@sap-ai-sdk/orchestration": "^2.11.0",
54
52
  "@sap-cloud-sdk/connectivity": "^4.7.0",
55
- "@sap-cloud-sdk/resilience": "^4.7.0",
56
53
  "deepagents": "^1.10.0",
57
54
  "langchain": "^1.5.0",
58
55
  "marked": "^18",
@@ -63,12 +60,15 @@
63
60
  "@cap-js/sqlite": "^3",
64
61
  "@sap/cds-mtxs": "^4",
65
62
  "@toon-format/toon": ">=2.3",
66
- "deepagents": "^1.10.2"
63
+ "acorn": "^8.18.0",
64
+ "deepagents": "^1.10.2",
65
+ "openevals": "^0.2.0"
67
66
  },
68
67
  "peerDependencies": {
69
68
  "@cap-js/audit-logging": ">=1",
70
69
  "@cap-js/telemetry": ">=1",
71
- "@sap/cds": ">=9"
70
+ "@sap/cds": ">=9",
71
+ "openevals": ">=0.2"
72
72
  },
73
73
  "peerDependenciesMeta": {
74
74
  "@cap-js/telemetry": {
@@ -76,6 +76,9 @@
76
76
  },
77
77
  "@cap-js/audit-logging": {
78
78
  "optional": true
79
+ },
80
+ "openevals": {
81
+ "optional": true
79
82
  }
80
83
  },
81
84
  "engines": {
@@ -87,6 +90,7 @@
87
90
  "trace_langchain": true,
88
91
  "mlflow": false,
89
92
  "connect": "auto",
93
+ "retention": "30d",
90
94
  "[development]": {
91
95
  "pushNotifications": {
92
96
  "allowedDomains": [
@@ -112,11 +116,16 @@
112
116
  },
113
117
  "persistAllCheckpointWrites": false,
114
118
  "activeUsersInterval": "24h",
119
+ "circuitBreaker": {
120
+ "errorThresholdPercentage": 50,
121
+ "volumeThreshold": 10,
122
+ "resetTimeout": 30000,
123
+ "rollingCountTimeout": 10000
124
+ },
115
125
  "params": {
116
126
  "max_tokens": 4096,
117
127
  "temperature": 0
118
128
  },
119
- "contentFilter": true,
120
129
  "fileIO": {
121
130
  "enabled": false,
122
131
  "maxInputFileSizeBytes": 2097152,
@@ -156,7 +165,7 @@
156
165
  },
157
166
  "requires": {
158
167
  "agents": {
159
- "model": "@cap-js/agents"
168
+ "model": "@cap-js/agents/srv/entities.cds"
160
169
  },
161
170
  "agent-push-notifications": {
162
171
  "impl": "@cap-js/agents/srv/push-notification-srv.js"
@@ -50,12 +50,23 @@ entity Tasks : managed {
50
50
  * Files written by agent via /outputs/ path for this task.
51
51
  */
52
52
  outputFiles : Composition of many Attachments;
53
+
54
+ /** LangGraph checkpoints created by this task. Cascade-deleted. */
55
+ checkpoints : Composition of many Checkpoints
56
+ on checkpoints.task_id = taskId;
57
+
58
+ /** LangGraph checkpoint writes tied to this task. Cascade-deleted. */
59
+ checkpointWrites : Composition of many CheckpointWrites
60
+ on checkpointWrites.task_id = taskId;
53
61
  }
54
62
 
55
63
  entity Checkpoints : managed {
56
64
  key thread_id : String;
57
65
  key checkpoint_ns : String default '';
58
66
  key checkpoint_id : String;
67
+ task_id : String;
68
+ task : Association to one Tasks
69
+ on task.taskId = task_id;
59
70
  parent_checkpoint_id : String;
60
71
  parent : Association to one Checkpoints
61
72
  on parent.checkpoint_id = parent_checkpoint_id;
@@ -0,0 +1,278 @@
1
+ import cds from "@sap/cds"
2
+ import { startCollection } from "../../lib/eval/span-collector.js"
3
+ import { metricsFromSpans } from "../../lib/eval/metrics.js"
4
+ import { getActiveRunState, logMlflowMetricsForResult } from "../../lib/eval/eval-run.js"
5
+
6
+ export const COLLECT_RESULT = Symbol.for("@cap-js/agents:chat:collect-result")
7
+
8
+ class NoopEventBus {
9
+ constructor() {
10
+ this.events = []
11
+ this[COLLECT_RESULT] = true
12
+ this._graphResult = null
13
+ this._done = false
14
+ this.finished$ = new Promise((resolve, reject) => {
15
+ this._resolve = resolve
16
+ this._reject = reject
17
+ })
18
+ }
19
+
20
+ publish(event) {
21
+ this.events.push(event)
22
+ }
23
+
24
+ finished() {
25
+ this._done = true
26
+ this._resolve?.()
27
+ }
28
+
29
+ error(err) {
30
+ this._reject?.(err)
31
+ }
32
+
33
+ getFinalText() {
34
+ for (let i = this.events.length - 1; i >= 0; i--) {
35
+ const e = this.events[i]
36
+ if (e.kind === "artifact-update" && e.lastChunk === true) {
37
+ const text = e.artifact?.parts
38
+ ?.filter((p) => p.kind === "text")
39
+ .map((p) => p.text)
40
+ .join("")
41
+ if (text) return text
42
+ }
43
+ }
44
+ for (let i = this.events.length - 1; i >= 0; i--) {
45
+ const e = this.events[i]
46
+ if (e.kind === "status-update" && e.status?.message?.parts) {
47
+ const text = e.status.message.parts
48
+ .filter((p) => p.kind === "text")
49
+ .map((p) => p.text)
50
+ .join("")
51
+ if (text) return text
52
+ }
53
+ }
54
+ return ""
55
+ }
56
+
57
+ getStatus() {
58
+ const e = this.events.at(-1)
59
+ if (e && e.kind === "status-update" && e.status?.state) {
60
+ const state = e.status.state
61
+ const msg = e.status?.message?.parts?.find((p) => p.kind === "text")?.text ?? ""
62
+ return { status: state, description: msg }
63
+ }
64
+ return { status: "completed", description: "" }
65
+ }
66
+ }
67
+
68
+ // Derive toolCalls from graph messages by pairing AIMessage.tool_calls with ToolMessage results.
69
+ function toolCallsFromMessages(messages) {
70
+ if (!Array.isArray(messages) || messages.length === 0) return []
71
+
72
+ const resultById = new Map()
73
+ for (const msg of messages) {
74
+ if (msg.tool_call_id && msg.type === "tool") {
75
+ const content = typeof msg.content === "string" ? msg.content : JSON.stringify(msg.content)
76
+ resultById.set(msg.tool_call_id, content)
77
+ }
78
+ }
79
+
80
+ const entries = []
81
+ for (const msg of messages) {
82
+ if (!Array.isArray(msg.tool_calls) || msg.tool_calls.length === 0) continue
83
+ for (const tc of msg.tool_calls) {
84
+ const raw = resultById.get(tc.id)
85
+ let toolResult
86
+ try {
87
+ toolResult = raw !== undefined ? JSON.parse(raw) : undefined
88
+ } catch {
89
+ toolResult = raw
90
+ }
91
+ const entry = {
92
+ tool: tc.name,
93
+ args: tc.args ?? {},
94
+ outcome: "success",
95
+ ...(toolResult !== undefined && { result: toolResult }),
96
+ }
97
+ attachCqn(entry)
98
+ entries.push(entry)
99
+ }
100
+ }
101
+ return entries
102
+ }
103
+
104
+ /**
105
+ * If entry.args.cql attaches cqn as hidden property.
106
+ * No-op when cql is absent or faulty.
107
+ */
108
+ function attachCqn(entry) {
109
+ const cql = entry.args?.cql
110
+ if (typeof cql !== "string") return
111
+ try {
112
+ const cqn = cds.parse.cql(cql)
113
+ Object.defineProperty(entry, "cqn", {
114
+ value: cqn,
115
+ enumerable: false,
116
+ writable: false,
117
+ configurable: true,
118
+ })
119
+ } catch {
120
+ /* unparseable CQL — skip */
121
+ }
122
+ }
123
+
124
+ function buildRequestContext(query, opts = {}) {
125
+ const taskId = opts.taskId || cds.utils.uuid()
126
+ const contextId = opts.contextId || cds.utils.uuid()
127
+ const parts =
128
+ opts.parts ??
129
+ (typeof query === "string"
130
+ ? [{ kind: "text", text: query }]
131
+ : (query?.parts ?? [{ kind: "text", text: String(query) }]))
132
+ return {
133
+ taskId,
134
+ contextId,
135
+ task: opts.task ?? null,
136
+ userMessage: {
137
+ kind: "message",
138
+ messageId: cds.utils.uuid(),
139
+ role: "user",
140
+ taskId,
141
+ contextId,
142
+ parts,
143
+ },
144
+ }
145
+ }
146
+
147
+ /** Install srv.chat(query, previous?) on an @agent service for eval tests. */
148
+ export function registerChat(srv) {
149
+ srv.chat = async function chat(query, previous) {
150
+ // Resolve the small eval helper API:
151
+ // chat(query)
152
+ // chat(query, prevResult) — extract contextId + taskId for HITL resume
153
+ // chat(query, { _details: true }) — internal/test escape hatch outside test profile
154
+ let opts = {}
155
+ if (typeof previous === "string") {
156
+ throw new TypeError(
157
+ "agent.chat: second argument must be a previous chat result object or options object",
158
+ )
159
+ } else if (previous && typeof previous === "object") {
160
+ if ("text" in previous || "contextId" in previous) {
161
+ // prior chat() result — extract conversation ids and HITL state
162
+ opts = {
163
+ contextId: previous.contextId,
164
+ // mark as resume when prior result was input-required
165
+ ...(previous.status === "input-required" && {
166
+ taskId: previous.taskId,
167
+ task: { id: previous.taskId, status: { state: "input-required" } },
168
+ }),
169
+ }
170
+ } else {
171
+ opts = { _details: previous._details === true }
172
+ }
173
+ }
174
+
175
+ const includeDetails = shouldIncludeChatDetails(opts)
176
+ const runState = includeDetails ? getActiveRunState() : null
177
+
178
+ // Open span collection session before execution so the processor
179
+ // captures all spans from this trace.
180
+ const collection = includeDetails ? await startCollection() : null
181
+
182
+ const { LangGraphExecutor } = await import("../langgraph-executor-srv.js")
183
+ const executor = LangGraphExecutor.for(srv)
184
+ const requestContext = buildRequestContext(query, opts)
185
+ const eventBus = new NoopEventBus()
186
+ const mlflowRunId = runState?.mlflowRunId
187
+ let traceId
188
+
189
+ const runInContext = async () => {
190
+ // Link the trace to the MLflow eval run — graph-executor reads this to set mlflow.sourceRun.
191
+ if (mlflowRunId && cds.context) cds.context["_mlflow.evalRunId"] = mlflowRunId
192
+
193
+ let execError = null
194
+ const execPromise = executor.execute(requestContext, eventBus).catch((err) => {
195
+ execError = err
196
+ if (!eventBus._done) eventBus.finished()
197
+ })
198
+ await Promise.race([eventBus.finished$, execPromise])
199
+ if (!eventBus._done) eventBus.finished()
200
+ if (execError) throw execError
201
+
202
+ if (includeDetails) {
203
+ const rootSpan = cds.context?.["_mlflow.rootSpan"]
204
+ if (rootSpan) traceId = rootSpan.spanContext?.()?.traceId
205
+ }
206
+ }
207
+
208
+ if (cds.context) {
209
+ await runInContext()
210
+ } else {
211
+ await cds._with(
212
+ new cds.EventContext({ tenant: "t0", user: new cds.User.Privileged() }),
213
+ runInContext,
214
+ )
215
+ }
216
+
217
+ const { status, description } = eventBus.getStatus()
218
+ if (status === "failed")
219
+ throw new Error(`agent.chat: task failed — ${description || "no message"}`)
220
+
221
+ const text = eventBus.getFinalText()
222
+ const result = {
223
+ text,
224
+ contextId: requestContext.contextId,
225
+ taskId: requestContext.taskId,
226
+ status,
227
+ }
228
+
229
+ if (includeDetails) {
230
+ const allMessages = eventBus._graphResult?.messages ?? []
231
+
232
+ // Find last HumanMessage whose text matches current query; return from there.
233
+ const queryText = String(query)
234
+ let turnStart = 0
235
+ for (let i = allMessages.length - 1; i >= 0; i--) {
236
+ const m = allMessages[i]
237
+ if (m.getType?.() === "human" || m._getType?.() === "human" || m.type === "human") {
238
+ const content = typeof m.content === "string" ? m.content : (m.content?.[0]?.text ?? "")
239
+ if (content.startsWith(queryText) || queryText.startsWith(content.trim())) {
240
+ turnStart = i
241
+ break
242
+ }
243
+ }
244
+ }
245
+ const messages = allMessages.slice(turnStart)
246
+ const toolCalls = toolCallsFromMessages(messages)
247
+
248
+ const allSpans = collection.collect()
249
+ const spans = traceId
250
+ ? allSpans.filter((s) => s.spanContext?.()?.traceId === traceId)
251
+ : allSpans
252
+
253
+ result.query = String(query)
254
+ result.traceId = traceId
255
+ result.toolCalls = toolCalls
256
+ result.messages = messages.map((m) => {
257
+ // Eval prompts expect text content, not content-block objects.
258
+ if (m.type === "ai") {
259
+ if (Array.isArray(m.content) && m.content[0]?.text) {
260
+ m.content = m.content[0].text
261
+ }
262
+ }
263
+ return m
264
+ })
265
+ result.spans = spans
266
+ result.metrics = metricsFromSpans(spans)
267
+ if (runState) result._evalState = runState
268
+ // Post metrics to MLflow ootb — fire-and-forget.
269
+ logMlflowMetricsForResult(result, runState).catch(() => {})
270
+ }
271
+
272
+ return result
273
+ }
274
+ }
275
+
276
+ export function shouldIncludeChatDetails(opts = {}) {
277
+ return cds.env.profiles?.includes("test") || opts._details === true
278
+ }
@@ -2,10 +2,13 @@ import cds from "@sap/cds"
2
2
  import { short, audit, ms4 } from "../../lib/utils/utils.js"
3
3
  import { partsToText, buildChatMessages, firstDataPart } from "../../lib/utils/message-handling.js"
4
4
  import * as metrics from "../../lib/telemetry/metrics.js"
5
- import { mlflowAttrs, mlflowTraceAttrs, setSpanAttrs } from "../../lib/telemetry/mlflow.js"
5
+ import { mlflowAttrs, mlflowTraceAttrs, setSpanAttrs } from "../../lib/telemetry/mlflow/index.js"
6
6
  import { CdsFileStore } from "../../lib/protocol/persistence/file-store.js"
7
7
  import { formatFileSize, sanitizeFilename } from "./tools.js"
8
8
  import { convertUsageData } from "../../lib/telemetry/chat-tracing.js"
9
+ import { triggerCleanup } from "../../lib/protocol/persistence/cleanup.js"
10
+ import { COLLECT_RESULT } from "./chat.js"
11
+ import { linkTraceToPrompt } from "../../lib/telemetry/mlflow/tracing.js"
9
12
 
10
13
  const LOG = cds.log("agents")
11
14
 
@@ -333,16 +336,14 @@ class GraphExecutor {
333
336
 
334
337
  let tokenCount = 0
335
338
  let finalState = null
336
- // Track the current turn (langchain message id) and whether it has emitted a
337
- // tool call. Anthropic-style turns can stream a text preamble BEFORE their
338
- // tool_use block ("Let me first look up …"); we can't tell in advance that
339
- // such a turn is planning rather than the final answer, so we stream those
340
- // tokens optimistically. Once we see tool_call_chunks for the same turn, we
341
- // know retrospectively that the preamble was planning — emit an authoritative
342
- // event-level replace with empty text to wipe the leaked preamble, then skip
343
- // all further text from this turn.
339
+ // Track the current turn (langchain message id). Each new turn opens a fresh
340
+ // bubble on the client via `append:false`; subsequent tokens of the same turn
341
+ // are `append:true` (accumulate). Anthropic-style turns can stream a text
342
+ // preamble BEFORE their tool_use block ("Let me first look up …") — we let
343
+ // that reasoning text stream normally; the client is responsible for the
344
+ // final visual (collapse to the last turn's bubble at task completion).
344
345
  let currentMsgId = null
345
- let turnHasToolCall = false
346
+ let thinkingCount = 0
346
347
 
347
348
  try {
348
349
  if (typeof graph.stream !== "function" || cds.env.agents?.streaming === false) {
@@ -378,30 +379,11 @@ class GraphExecutor {
378
379
 
379
380
  if (msgChunk.id && msgChunk.id !== currentMsgId) {
380
381
  currentMsgId = msgChunk.id
381
- turnHasToolCall = false
382
382
  tokenCount = 0
383
383
  }
384
384
 
385
- // Retroactively invalidate a leaked planning preamble. In a ReAct loop
386
- // the model can emit "Let me look this up …" before its tool_use block
387
- if (msgChunk.tool_call_chunks?.length && !turnHasToolCall) {
388
- turnHasToolCall = true
389
- if (tokenCount > 0) {
390
- eventBus.publish({
391
- kind: "artifact-update",
392
- taskId,
393
- contextId,
394
- append: false,
395
- lastChunk: false,
396
- artifact: {
397
- artifactId: "response",
398
- parts: [{ kind: "text", text: "" }],
399
- },
400
- })
401
- tokenCount = 0
402
- }
403
- }
404
- if (turnHasToolCall) continue
385
+ const lastChunk =
386
+ !!msgChunk.additional_kwargs?.intermediate_results?.llm?.choices[0].finish_reason
405
387
 
406
388
  const text = messageText(msgChunk?.content)
407
389
  if (!text) continue
@@ -414,12 +396,13 @@ class GraphExecutor {
414
396
  taskId,
415
397
  contextId,
416
398
  append: tokenCount > 0,
417
- lastChunk: false,
399
+ lastChunk: lastChunk,
418
400
  artifact: {
419
- artifactId: "response",
401
+ artifactId: `thinking-${thinkingCount}`,
420
402
  parts: [{ kind: "text", text }],
421
403
  },
422
404
  })
405
+ if (lastChunk) thinkingCount++
423
406
  tokenCount++
424
407
  } else if (mode === "updates") {
425
408
  // The updates stream yields per-node deltas — { <node>: { messages: [oneNewMessage] } },
@@ -589,6 +572,10 @@ class GraphExecutor {
589
572
  audit("AgentTaskStarted", {
590
573
  data: { taskId, contextId, service: serviceName, userMessage: requestContext.userMessage },
591
574
  })
575
+ // Lazy scheduling task deletion
576
+ cds.spawn({}, async () => {
577
+ await triggerCleanup(serviceName)
578
+ })
592
579
  }
593
580
 
594
581
  // ── File I/O: persist incoming FileParts to cap.agent.Tasks.inputFiles ──
@@ -694,6 +681,12 @@ class GraphExecutor {
694
681
  }),
695
682
  )
696
683
  setSpanAttrs(rootSpan, mlflowTraceAttrs())
684
+ // Required for MLFLow run linking
685
+ const evalRunId = cds.context?.["_mlflow.evalRunId"]
686
+ if (evalRunId) {
687
+ rootSpan.setAttribute("mlflow.sourceRun", evalRunId)
688
+ wfSpan.setAttribute("mlflow.sourceRun", evalRunId)
689
+ }
697
690
  }
698
691
 
699
692
  let usageData
@@ -748,7 +741,6 @@ class GraphExecutor {
748
741
  contextId,
749
742
  service: serviceName,
750
743
  decision,
751
- userMessage: requestContext.userMessage,
752
744
  },
753
745
  })
754
746
  // On edit, stash a diff note in state; the injector middleware prepends it next turn.
@@ -818,7 +810,7 @@ class GraphExecutor {
818
810
  contextId,
819
811
  service: serviceName,
820
812
  description,
821
- userMessage: requestContext.userMessage,
813
+ interruptData,
822
814
  },
823
815
  })
824
816
 
@@ -873,8 +865,7 @@ class GraphExecutor {
873
865
  service: serviceName,
874
866
  duration,
875
867
  tokenUsage: usageData,
876
- toolCalls: totalToolCalls(result.messages),
877
- output: output?.slice(0, 2000),
868
+ toolCalls: toolCallsShortened(result.messages),
878
869
  task: requestContext.task,
879
870
  },
880
871
  })
@@ -1073,6 +1064,12 @@ class GraphExecutor {
1073
1064
  })
1074
1065
  }
1075
1066
 
1067
+ // Programmatic .chat() path: stash graph result on eventBus so chat.js
1068
+ // can read messages without an extra checkpoint roundtrip.
1069
+ if (eventBus[COLLECT_RESULT]) {
1070
+ eventBus._graphResult = { messages: result.messages || [] }
1071
+ }
1072
+
1076
1073
  eventBus.publish({
1077
1074
  kind: "status-update",
1078
1075
  taskId,
@@ -1240,6 +1237,10 @@ class GraphExecutor {
1240
1237
  final: true,
1241
1238
  })
1242
1239
  } finally {
1240
+ // setSpanAttrs must happen at the end for the linking as else prompt might not yet have been created
1241
+ const rootSpan = cds.context["_mlflow.rootSpan"]
1242
+ setSpanAttrs(rootSpan, linkTraceToPrompt())
1243
+
1243
1244
  this._abortControllers.delete(taskId)
1244
1245
  metrics.concurrentExecutions.add(-1, mAttrs)
1245
1246
 
@@ -1367,3 +1368,13 @@ function totalToolCalls(messages) {
1367
1368
  return acc
1368
1369
  }, 0)
1369
1370
  }
1371
+
1372
+ /**
1373
+ * @param {[import('@langchain/core/messages').Message]} messages
1374
+ */
1375
+ function toolCallsShortened(messages) {
1376
+ return messages.reduce((acc, val) => {
1377
+ if (val.type === "tool") acc.push(val.name)
1378
+ return acc
1379
+ }, [])
1380
+ }
@@ -3,6 +3,8 @@ import { generateTools, createReadFileTool } from "./tools.js"
3
3
  import { buildSystemPrompt } from "./system-prompt.js"
4
4
  import buildMiddleware from "../../lib/agents/middleware/index.js"
5
5
  import { partsToText } from "../../lib/utils/message-handling.js"
6
+ import { cleanupExpiredTasks } from "../../lib/protocol/persistence/cleanup.js"
7
+ import { registerChat } from "./chat.js"
6
8
 
7
9
  const LOG = cds.log("agents")
8
10
 
@@ -170,4 +172,10 @@ export default function registerDefaultAgentHandlers(srv) {
170
172
  },
171
173
  })
172
174
  })
175
+
176
+ srv.on("cleanupTasks", async () => {
177
+ await cleanupExpiredTasks(srv.name)
178
+ })
179
+
180
+ registerChat(srv)
173
181
  }