@cap-js/agents 0.9.2 → 0.9.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/README.md +2 -0
  2. package/cds-plugin.js +2 -2
  3. package/lib/agents/middleware/content-filter.js +3 -2
  4. package/lib/agents/middleware/hitl-decision-note-injector.js +18 -0
  5. package/lib/agents/middleware/hitl.js +2 -2
  6. package/lib/agents/middleware/index.js +11 -0
  7. package/lib/agents/middleware/quota-enforcer.js +9 -0
  8. package/lib/agents/middleware/remote-mcp.js +62 -0
  9. package/lib/agents/middleware/tool-wrap.js +52 -0
  10. package/lib/compile.js +6 -2
  11. package/lib/eval/Judge.js +239 -0
  12. package/lib/eval/eval-describe.js +69 -0
  13. package/lib/eval/eval-run.js +147 -0
  14. package/lib/eval/index.js +6 -0
  15. package/lib/eval/metrics.js +53 -0
  16. package/lib/eval/span-collector.js +50 -0
  17. package/lib/index.js +2 -1
  18. package/lib/models/aicore.js +23 -6
  19. package/lib/models/anthropic.js +27 -12
  20. package/lib/preview/chat.html +373 -66
  21. package/lib/protocol/agent-card.js +6 -2
  22. package/lib/sidecar.js +1 -1
  23. package/lib/telemetry/chat-tracing.js +39 -8
  24. package/lib/telemetry/mlflow/credentials.js +79 -0
  25. package/lib/telemetry/mlflow/evaluation.js +38 -0
  26. package/lib/telemetry/mlflow/exporter/DatabricksExporter.js +75 -0
  27. package/lib/telemetry/mlflow/exporter/MlflowExporter.js +115 -0
  28. package/lib/telemetry/mlflow/exporter/index.js +23 -0
  29. package/lib/telemetry/mlflow/index.js +16 -0
  30. package/lib/telemetry/mlflow/prompts.js +123 -0
  31. package/lib/telemetry/mlflow/tracing.js +264 -0
  32. package/lib/telemetry/tool-tracing.js +27 -54
  33. package/lib/telemetry/tracing.js +3 -1
  34. package/lib/utils/markdown.js +1 -11
  35. package/lib/utils/message-handling.js +14 -0
  36. package/lib/utils/resilience.js +133 -0
  37. package/lib/utils/utils.js +17 -0
  38. package/package.json +26 -14
  39. package/srv/handlers/chat.js +278 -0
  40. package/srv/handlers/graph-executor/hitl.js +263 -0
  41. package/srv/handlers/graph-executor.js +100 -286
  42. package/srv/handlers/index.js +5 -1
  43. package/srv/handlers/mcp-tools.js +7 -45
  44. package/srv/handlers/sub-agent-tools.js +30 -12
  45. package/srv/handlers/tools.js +39 -5
  46. package/index.js +0 -0
  47. package/lib/agents/middleware/hitl-edit-note-injector.js +0 -18
  48. package/lib/telemetry/mlflow.js +0 -290
  49. /package/{index.cds → srv/entities.cds} +0 -0
@@ -0,0 +1,147 @@
1
+ /* global beforeAll, afterEach, afterAll */
2
+ import cds from "@sap/cds"
3
+ import {
4
+ postMlflowAssessment,
5
+ createEvalRun,
6
+ closeEvalRun,
7
+ logMlflowMetrics,
8
+ } from "../telemetry/mlflow/evaluation.js"
9
+ import { flushMlflowTraces } from "../telemetry/mlflow/tracing.js"
10
+
11
+ export function getActiveRunState() {
12
+ return cds._activeEvalRun ?? null
13
+ }
14
+
15
+ export function evalRun(opts = {}) {
16
+ if (typeof beforeAll !== "function" || typeof afterAll !== "function") return
17
+ if (!cds.env.agents?.mlflow) return
18
+
19
+ let state = null
20
+
21
+ function _makeState(runId, mlflowRunId) {
22
+ return {
23
+ runId,
24
+ mlflowRunId,
25
+ validationsByTask: new Map(),
26
+ }
27
+ }
28
+
29
+ beforeAll(async () => {
30
+ const runId = cds.utils.uuid()
31
+ const mlflowRunId = await createEvalRun(opts).catch(() => null)
32
+ state = _makeState(runId, mlflowRunId)
33
+ cds._activeEvalRun = state
34
+ })
35
+
36
+ if (typeof afterEach === "function") {
37
+ afterEach(async () => {
38
+ if (state) await _flushValidations(state)
39
+ })
40
+ }
41
+
42
+ afterAll(async () => {
43
+ if (state) await _flushValidations(state)
44
+ await flushMlflowTraces()
45
+ await closeEvalRun(state?.mlflowRunId).catch(() => {})
46
+ if (cds._activeEvalRun === state) cds._activeEvalRun = null
47
+ state = null
48
+ })
49
+ }
50
+
51
+ async function _flushValidations(state) {
52
+ if (!state?.validationsByTask.size) return
53
+
54
+ const tasks = []
55
+ for (const [, entry] of state.validationsByTask) {
56
+ const { passes, traceId } = entry
57
+ if (!passes.length) continue
58
+
59
+ const success_rate = passes.every(Boolean) ? 1 : 0
60
+ const output_correctness = passes.filter(Boolean).length / passes.length
61
+ const codeOpts = { sourceType: "CODE" }
62
+
63
+ if (state.mlflowRunId) {
64
+ tasks.push(
65
+ logMlflowMetrics(state.mlflowRunId, { success_rate, output_correctness }).catch(() => {}),
66
+ )
67
+ }
68
+
69
+ if (traceId) {
70
+ tasks.push(
71
+ flushMlflowTraces().then(() =>
72
+ Promise.all([
73
+ postMlflowAssessment(traceId, success_rate, "", "success_rate", null, codeOpts).catch(
74
+ () => {},
75
+ ),
76
+ postMlflowAssessment(
77
+ traceId,
78
+ output_correctness,
79
+ "",
80
+ "output_correctness",
81
+ null,
82
+ codeOpts,
83
+ ).catch(() => {}),
84
+ ]),
85
+ ),
86
+ )
87
+ }
88
+ }
89
+
90
+ await Promise.all(tasks)
91
+ state.validationsByTask.clear()
92
+ }
93
+
94
+ export function recordEvaluation(result, assessment = {}) {
95
+ if (!cds.env.agents?.mlflow) return
96
+ const { pass, score, comment, ...config } = assessment
97
+ // Conversation level evaluations shall not be included in the per task roll-up
98
+ if (pass !== undefined && !config.conversationLevel) _addValidation(result, pass)
99
+ if (score === undefined) return
100
+ return _postAssessmentScore(result, score, comment, config)
101
+ }
102
+
103
+ function _addValidation(result, pass) {
104
+ const state = result?._evalState ?? cds._activeEvalRun
105
+ if (!state || !result?.taskId) return
106
+ const key = result.taskId
107
+ if (!state.validationsByTask.has(key)) {
108
+ state.validationsByTask.set(key, { passes: [], traceId: result.traceId })
109
+ }
110
+ state.validationsByTask.get(key).passes.push(pass)
111
+ }
112
+
113
+ async function _postAssessmentScore(result, score, comment, config) {
114
+ const traceId = config.traceId ?? result?.traceId
115
+ if (!traceId) return
116
+
117
+ const opts = { sourceType: config.sourceType }
118
+ await flushMlflowTraces()
119
+
120
+ // conversationLevel: session assessment — post with session metadata only
121
+ if (config.conversationLevel) {
122
+ await postMlflowAssessment(
123
+ traceId,
124
+ score,
125
+ comment ?? "",
126
+ config.assessmentName,
127
+ config.model ?? null,
128
+ { ...opts, metadata: { "mlflow.trace.session": config.sessionId ?? "" } },
129
+ )
130
+ } else {
131
+ // Single-turn: post to this trace only, no session metadata
132
+ await postMlflowAssessment(
133
+ traceId,
134
+ score,
135
+ comment ?? "",
136
+ config.assessmentName,
137
+ config.model ?? null,
138
+ opts,
139
+ )
140
+ }
141
+ }
142
+
143
+ export async function logMlflowMetricsForResult(result, state = null) {
144
+ state = state ?? cds._activeEvalRun
145
+ if (!state?.mlflowRunId) return
146
+ await logMlflowMetrics(state.mlflowRunId, result.metrics).catch(() => {})
147
+ }
@@ -0,0 +1,6 @@
1
+ import { installEvalDescribe } from "./eval-describe.js"
2
+
3
+ installEvalDescribe()
4
+
5
+ export { Judge, matchToolCall } from "./Judge.js"
6
+ export { evalRun } from "./eval-run.js"
@@ -0,0 +1,53 @@
1
+ function _hrtimeToMs(hr) {
2
+ return hr[0] * 1000 + hr[1] / 1e6
3
+ }
4
+
5
+ /**
6
+ * @param {object[]} spans Finished OTel spans from span-collector.js
7
+ * @returns {{ input_tokens, output_tokens, total_tokens, tool_call_count, latency_ms, cost_usd }}
8
+ */
9
+ export function metricsFromSpans(spans) {
10
+ let input_tokens = 0
11
+ let output_tokens = 0
12
+ let tool_call_count = 0
13
+ let cost_usd = 0
14
+ let latency_ms = null
15
+
16
+ for (const span of spans) {
17
+ const attrs = span.attributes ?? {}
18
+ const op = attrs["gen_ai.operation.name"]
19
+
20
+ if (op === "chat") {
21
+ input_tokens += Number(attrs["gen_ai.usage.input_tokens"] ?? 0)
22
+ output_tokens += Number(attrs["gen_ai.usage.output_tokens"] ?? 0)
23
+
24
+ // mlflow.llm.cost is set by apps themselves
25
+ const rawCost = attrs["mlflow.llm.cost"]
26
+ if (rawCost) {
27
+ try {
28
+ const c = typeof rawCost === "string" ? JSON.parse(rawCost) : rawCost
29
+ cost_usd += c.total_cost ?? 0
30
+ } catch {
31
+ /* malformed — skip */
32
+ }
33
+ }
34
+ }
35
+
36
+ if (op === "execute_tool") {
37
+ tool_call_count++
38
+ }
39
+
40
+ if (op === "invoke_agent" && span.endTime && span.startTime) {
41
+ latency_ms = _hrtimeToMs(span.endTime) - _hrtimeToMs(span.startTime)
42
+ }
43
+ }
44
+
45
+ return {
46
+ input_tokens,
47
+ output_tokens,
48
+ total_tokens: input_tokens + output_tokens,
49
+ tool_call_count,
50
+ latency_ms,
51
+ cost_usd: cost_usd > 0 ? cost_usd : null,
52
+ }
53
+ }
@@ -0,0 +1,50 @@
1
+ const _sessions = new Map()
2
+ let _registered = false
3
+ let _sessionCounter = 0
4
+
5
+ const _processor = {
6
+ onStart() {},
7
+ onEnd(span) {
8
+ if (_sessions.size === 0) return
9
+ for (const session of _sessions.values()) {
10
+ session.spans.push(span)
11
+ }
12
+ },
13
+ async forceFlush() {},
14
+ async shutdown() {},
15
+ }
16
+
17
+ async function _ensureRegistered() {
18
+ if (_registered) return
19
+ _registered = true
20
+ try {
21
+ const { trace } = await import("@opentelemetry/api")
22
+ const provider = trace.getTracerProvider()
23
+ const delegate = provider.getDelegate?.() || provider
24
+ if (delegate.constructor?.name === "NoopTracerProvider") return
25
+ if (typeof delegate.addSpanProcessor === "function") {
26
+ delegate.addSpanProcessor(_processor)
27
+ } else if (Array.isArray(delegate._activeSpanProcessor?._spanProcessors)) {
28
+ delegate._activeSpanProcessor._spanProcessors.push(_processor)
29
+ }
30
+ } catch {
31
+ /* OTel not present */
32
+ }
33
+ }
34
+
35
+ function _openSession() {
36
+ const id = ++_sessionCounter
37
+ const session = { spans: [] }
38
+ _sessions.set(id, session)
39
+ return {
40
+ collect() {
41
+ _sessions.delete(id)
42
+ return session.spans
43
+ },
44
+ }
45
+ }
46
+
47
+ export async function startCollection() {
48
+ await _ensureRegistered()
49
+ return _openSession()
50
+ }
package/lib/index.js CHANGED
@@ -161,7 +161,8 @@ export default function A2AProtocolAdapter(srv, options = {}) {
161
161
  }
162
162
 
163
163
  router.get("/.well-known/agent-card.json", (req, res) => {
164
- const url = proxyUrl || `${req.protocol}://${req.get("host")}${req.baseUrl}`
164
+ const proto = req.headers["x-forwarded-proto"]?.split(",")[0].trim() || req.protocol
165
+ const url = proxyUrl || `${proto}://${req.get("host")}${req.baseUrl}`
165
166
  // Regenerate agent card when feature toggles are active (annotations may differ)
166
167
  let card
167
168
  if (cds.context?.features && Object.keys(cds.context.features).length > 0) {
@@ -1,11 +1,13 @@
1
1
  import cds from "@sap/cds"
2
2
  import { OrchestrationClient } from "@sap-ai-sdk/langchain"
3
- import { circuitBreaker, timeout } from "@sap-cloud-sdk/resilience"
3
+ import { circuitBreaker, timeout } from "../utils/resilience.js"
4
4
 
5
5
  import { SystemMessage, ToolMessage, HumanMessage, AIMessage } from "@langchain/core/messages"
6
6
  import { ms4 } from "../utils/utils.js"
7
+ import { syncSystemPrompt } from "../telemetry/mlflow/prompts.js"
7
8
 
8
9
  const LOG = cds.log("agents")
10
+ const DEFAULT_STREAM_DELIMITERS = [".", "!", "?"]
9
11
 
10
12
  class _InstrumentedOrchestrationClient extends OrchestrationClient {
11
13
  constructor(name, options) {
@@ -18,6 +20,7 @@ class _InstrumentedOrchestrationClient extends OrchestrationClient {
18
20
  // and extra ReAct iterations — see PR #188 review).
19
21
  const params =
20
22
  options.params || (deepAgent ? { max_tokens: 4096, temperature: 0 } : cds.env.agents?.params)
23
+ const auditParams = params && typeof params === "object" ? { ...params } : params
21
24
 
22
25
  LOG.debug("Initializing LLM", { model, deepAgent: !!deepAgent })
23
26
 
@@ -42,15 +45,15 @@ class _InstrumentedOrchestrationClient extends OrchestrationClient {
42
45
  {
43
46
  promptTemplating: { model: { name: model, params } },
44
47
  ...(filtering && { filtering }),
48
+ },
49
+ {
45
50
  // `streaming` controls the SDK's auto-stream-and-concatenate in _generate()
46
51
  // (i.e. direct model.invoke() calls). It does NOT gate LangGraph token
47
52
  // streaming — that is driven by graph.stream(streamMode:["messages"]) via a
48
53
  // streaming callback handler + our overridden _streamResponseChunks, and
49
54
  // works for every agent regardless of this flag (deep or managed alike).
50
55
  // Default on; `streaming: false` opts out.
51
- ...(streaming !== false ? { streaming: true } : {}),
52
- },
53
- {
56
+ streaming: streaming !== false,
54
57
  onFailedAttempt: (err) => {
55
58
  // Abort retries when circuit breaker is open (otherwise pRetry delays ~30-60s)
56
59
  if (err.code === "EOPENBREAKER" || err.message === "Breaker is open") {
@@ -62,7 +65,7 @@ class _InstrumentedOrchestrationClient extends OrchestrationClient {
62
65
  destination,
63
66
  )
64
67
  this.name = name
65
- this.options = { ...options, params, contentFilter, flatten }
68
+ this.options = { ...options, params: auditParams, contentFilter, flatten }
66
69
  }
67
70
 
68
71
  async _generate(messages, opts, runManager) {
@@ -88,7 +91,7 @@ class _InstrumentedOrchestrationClient extends OrchestrationClient {
88
91
  opts = _withMiddleware(this, opts)
89
92
  const prepared = _prepareMessages(messages, { flatten, model, opts })
90
93
  const inputMessages = prepared.inputMessages
91
- opts = prepared.opts
94
+ opts = withDefaultStreamDelimiters(prepared.opts)
92
95
 
93
96
  let turnHasToolCall = false
94
97
  for await (const chunk of super._streamResponseChunks(inputMessages, opts, runManager)) {
@@ -144,9 +147,23 @@ function _prepareMessages(messages, { flatten, model, opts }) {
144
147
  opts = { ...opts, tools }
145
148
  }
146
149
  }
150
+ syncSystemPrompt(inputMessages)
147
151
  return { inputMessages, opts }
148
152
  }
149
153
 
154
+ export function withDefaultStreamDelimiters(opts = {}) {
155
+ return {
156
+ ...opts,
157
+ streamOptions: {
158
+ ...opts.streamOptions,
159
+ global: {
160
+ ...opts.streamOptions?.global,
161
+ delimiters: opts.streamOptions?.global?.delimiters ?? DEFAULT_STREAM_DELIMITERS,
162
+ },
163
+ },
164
+ }
165
+ }
166
+
150
167
  // Claude currently only supports caching of type ephemeral. TTL can differ between 5min or 1h but
151
168
  // we use the 5min default
152
169
  const CACHE_CONTROL_EPHEMERAL = { type: "ephemeral" }
@@ -5,6 +5,7 @@ import os from 'node:os'
5
5
  import cds from '@sap/cds'
6
6
 
7
7
  const HOME = os.homedir() || process.env.HOME || process.env.USERPROFILE
8
+ const local = file => file.replace(HOME,'~')
8
9
  const LOG = cds.log('agents')
9
10
 
10
11
  /**
@@ -15,27 +16,36 @@ const LOG = cds.log('agents')
15
16
  */
16
17
  export default class ChatAnthropicService extends ChatAnthropic {
17
18
  constructor (name, options) {
18
- // REVISIT: may be better handled via options.credentials?
19
- let config = { ...options, ...fromEnv() }
19
+ let config = { ...options, ...options?.credentials, ...fromEnv() }
20
20
  if (!config.anthropicApiUrl) config = {
21
21
  ...fromClaude() || fromOpencode(),
22
22
  ...config
23
23
  }
24
- LOG.debug (`Using effective config for ChatAnthropic:`, config)
24
+ if (LOG._debug) {
25
+ const { kind, model, anthropicApiUrl, apiKey } = config
26
+ LOG.info (`Using effective config:`, {
27
+ kind,
28
+ model,
29
+ credentials: {
30
+ anthropicApiUrl,
31
+ apiKey: apiKey ? '***' : undefined
32
+ }
33
+ })
34
+ }
25
35
  super (config)
26
36
  this.name = name
27
37
  this.options = config
28
38
  }
29
39
  }
30
40
 
31
- function fromEnv (env = process.env) {
41
+ function fromEnv (env = process.env, silent) {
32
42
  let any, config = {}
33
43
  if ((any = env.ANTHROPIC_BASE_URL)) config.anthropicApiUrl = any
34
44
  if ((any = env.ANTHROPIC_AUTH_TOKEN)) config.apiKey = any
35
45
  if ((any = env.ANTHROPIC_API_KEY)) config.apiKey = any
36
46
  if ((any = env.ANTHROPIC_MODEL)) config.model = any
37
47
  if (!Object.keys(config).length) return null
38
- LOG.debug (`Loaded Anthropic settings from env:`, config)
48
+ if (!silent) LOG.debug (`Loaded Anthropic settings from env:`, config)
39
49
  return config
40
50
  }
41
51
 
@@ -43,12 +53,17 @@ function fromClaude() {
43
53
  if ('cached' in fromClaude) return fromClaude.cached
44
54
  const settings_json = path.join (HOME,'.claude/settings.json')
45
55
  try {
46
- let conf = JSON.parse (fs.readFileSync (settings_json,'utf8'))
56
+ let settings = JSON.parse (fs.readFileSync (settings_json,'utf8'))
47
57
  // https://www.schemastore.org/claude-code-settings.json
48
- fromClaude.cached = conf = fromEnv (conf?.env)
49
- LOG.debug(`Loaded Claude settings from`, settings_json, ':', conf)
58
+
59
+ let conf = fromClaude.cached = fromEnv (settings?.env, 'silent')
60
+ if (!conf.model && settings.env) {
61
+ let family = (settings.model||'sonnet').toUpperCase()
62
+ conf.model = settings.env[`ANTHROPIC_DEFAULT_${family}_MODEL`] || settings?.model
63
+ }
64
+ LOG.debug(`Loaded Claude settings from`, local(settings_json), ':', conf)
50
65
  } catch {
51
- LOG.debug(`Failed loading Claude settings from`, settings_json)
66
+ LOG.debug(`Failed loading Claude settings from`, local(settings_json))
52
67
  fromClaude.cached = null
53
68
  }
54
69
  return fromClaude.cached
@@ -59,7 +74,7 @@ function fromOpencode() {
59
74
  const opencode_json = path.join (HOME,'.config/opencode/opencode.json')
60
75
  try {
61
76
  let conf = JSON.parse (fs.readFileSync (opencode_json,'utf8'))
62
- LOG.debug(`Loaded OpenCode settings from`, opencode_json)
77
+ LOG.debug(`Loaded OpenCode settings from`, local(opencode_json))
63
78
  // https://opencode.ai/config.json
64
79
  let o = conf?.provider?.anthropic?.options
65
80
  if (!o) return fromOpencode.cached = null
@@ -68,9 +83,9 @@ function fromOpencode() {
68
83
  if ((any = o.anthropicApiKey ?? o.apiKey)) config.apiKey = any
69
84
  if ((any = conf?.model)) config.model = any.replace('anthropic/','')
70
85
  fromOpencode.cached = Object.keys(config).length ? config : null
71
- LOG.debug(`Loaded OpenCode settings from`, opencode_json, ':', conf)
86
+ LOG.debug(`Loaded OpenCode settings from`, local(opencode_json), ':', conf)
72
87
  } catch {
73
- LOG.debug(`Failed loading OpenCode settings from`, opencode_json)
88
+ LOG.debug(`Failed loading OpenCode settings from`, local(opencode_json))
74
89
  fromOpencode.cached = null
75
90
  }
76
91
  return fromOpencode.cached