@cap-js/agents 0.9.1 → 0.9.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -1
- package/cds-plugin.js +2 -2
- package/lib/agents/markdown/backends/outputs-backend.js +16 -12
- package/lib/agents/markdown/backends/readonly-backend.js +48 -0
- package/lib/agents/markdown/backends/uploads-backend.js +30 -13
- package/lib/agents/markdown/deep-agent.js +15 -26
- package/lib/agents/middleware/content-filter.js +3 -2
- package/lib/agents/middleware/index.js +11 -0
- package/lib/agents/middleware/remote-mcp.js +62 -0
- package/lib/agents/middleware/tool-wrap.js +52 -0
- package/lib/compile.js +6 -2
- package/lib/eval/Judge.js +239 -0
- package/lib/eval/eval-describe.js +69 -0
- package/lib/eval/eval-run.js +147 -0
- package/lib/eval/index.js +6 -0
- package/lib/eval/metrics.js +53 -0
- package/lib/eval/span-collector.js +50 -0
- package/lib/index.js +4 -3
- package/lib/models/aicore.js +23 -6
- package/lib/preview/chat.html +394 -69
- package/lib/protocol/agent-card.js +6 -2
- package/lib/protocol/persistence/checkpoint-saver.js +4 -1
- package/lib/protocol/persistence/cleanup.js +84 -0
- package/lib/sidecar.js +1 -1
- package/lib/telemetry/chat-tracing.js +61 -8
- package/lib/telemetry/mlflow/credentials.js +79 -0
- package/lib/telemetry/mlflow/evaluation.js +38 -0
- package/lib/telemetry/mlflow/exporter/DatabricksExporter.js +75 -0
- package/lib/telemetry/mlflow/exporter/MlflowExporter.js +115 -0
- package/lib/telemetry/mlflow/exporter/index.js +23 -0
- package/lib/telemetry/mlflow/index.js +16 -0
- package/lib/telemetry/mlflow/prompts.js +123 -0
- package/lib/telemetry/mlflow/tracing.js +264 -0
- package/lib/telemetry/tool-tracing.js +27 -54
- package/lib/telemetry/tracing.js +3 -1
- package/lib/utils/markdown.js +1 -11
- package/lib/utils/resilience.js +133 -0
- package/lib/utils/utils.js +17 -0
- package/package.json +24 -15
- package/{index.cds → srv/entities.cds} +11 -0
- package/srv/handlers/chat.js +278 -0
- package/srv/handlers/graph-executor.js +48 -37
- package/srv/handlers/index.js +8 -0
- package/srv/handlers/mcp-tools.js +11 -47
- package/srv/handlers/sub-agent-tools.js +12 -3
- package/srv/handlers/tools.js +9 -5
- package/index.js +0 -0
- package/lib/telemetry/mlflow.js +0 -290
|
@@ -0,0 +1,239 @@
|
|
|
1
|
+
import cds from "@sap/cds"
|
|
2
|
+
import { recordEvaluation } from "./eval-run.js"
|
|
3
|
+
|
|
4
|
+
const LOG = cds.log("agents-judge")
|
|
5
|
+
const CRITERIA_SEPARATOR = "\n\n"
|
|
6
|
+
|
|
7
|
+
// ─── Base Judge ───────────────────────────────────────────────────────────────
|
|
8
|
+
|
|
9
|
+
export class Judge {
|
|
10
|
+
constructor(opts) {
|
|
11
|
+
_assertSingleConstructorArg(arguments)
|
|
12
|
+
const { criteria, assessmentName, continuous, invertedScala, type } = _judgeOptions(
|
|
13
|
+
opts,
|
|
14
|
+
"ANSWER_RELEVANCE_PROMPT",
|
|
15
|
+
arguments.length,
|
|
16
|
+
)
|
|
17
|
+
if (!criteria || typeof criteria !== "string") {
|
|
18
|
+
throw new TypeError("Judge: 'criteria' is required and must be a string")
|
|
19
|
+
}
|
|
20
|
+
if (type !== undefined && type !== "trajectory") {
|
|
21
|
+
throw new TypeError("Judge: 'type' must be 'trajectory' when provided")
|
|
22
|
+
}
|
|
23
|
+
this._criteria = criteria
|
|
24
|
+
this._type = type
|
|
25
|
+
const isInvertedScala = (criteria) => {
|
|
26
|
+
if (_promptKeyFromCriteria(criteria) === "TOXICITY_PROMPT") {
|
|
27
|
+
return true
|
|
28
|
+
}
|
|
29
|
+
return false
|
|
30
|
+
}
|
|
31
|
+
this._invertedScala = invertedScala ?? isInvertedScala(criteria)
|
|
32
|
+
|
|
33
|
+
this._assessmentName =
|
|
34
|
+
assessmentName ?? _assessmentNameFromCriteria(this._criteria, this._defaultAssessmentName())
|
|
35
|
+
this._continuous = continuous ?? true
|
|
36
|
+
this._judgeImpl = null
|
|
37
|
+
this._sessionJudgeImpl = null
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
_defaultAssessmentName() {
|
|
41
|
+
return this._type === "trajectory" ? "trajectory" : "relevance"
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
async _ensureJudge(type) {
|
|
45
|
+
const judgeType = arguments.length === 0 ? this._type : type
|
|
46
|
+
if (judgeType === this._type && this._judgeImpl) return this._judgeImpl
|
|
47
|
+
if (judgeType !== this._type && this._sessionJudgeImpl) return this._sessionJudgeImpl
|
|
48
|
+
const judgeImpl = await _loadJudgeImpl(
|
|
49
|
+
this._criteria,
|
|
50
|
+
this._assessmentName,
|
|
51
|
+
this._continuous,
|
|
52
|
+
judgeType,
|
|
53
|
+
)
|
|
54
|
+
if (judgeType === this._type) this._judgeImpl = judgeImpl
|
|
55
|
+
else this._sessionJudgeImpl = judgeImpl
|
|
56
|
+
return judgeImpl
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/** Sibling with appended prompt/criteria. */
|
|
60
|
+
criteria(criteria) {
|
|
61
|
+
if (!criteria || typeof criteria !== "string") {
|
|
62
|
+
throw new TypeError("Judge.criteria: argument 'criteria' is required and must be a string")
|
|
63
|
+
}
|
|
64
|
+
const sibling = new this.constructor({
|
|
65
|
+
criteria: `${this._criteria}${CRITERIA_SEPARATOR}${criteria}`,
|
|
66
|
+
assessmentName: this._assessmentName,
|
|
67
|
+
continuous: this._continuous,
|
|
68
|
+
type: this._type,
|
|
69
|
+
})
|
|
70
|
+
return sibling
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/** @returns {Promise<{score, comment, pass}>} */
|
|
74
|
+
async evaluate(result) {
|
|
75
|
+
if (!result) throw new Error("evaluate: result is required")
|
|
76
|
+
if (Array.isArray(result)) return this._evaluateSession(result)
|
|
77
|
+
const judgeImpl = await this._ensureJudge()
|
|
78
|
+
const judgement = await judgeImpl(this._buildInput(result))
|
|
79
|
+
const { score, pass, comment } = this._judgementResult(judgement)
|
|
80
|
+
LOG.debug(`[${this._assessmentName}] score=${score} pass=${pass} — ${comment}`)
|
|
81
|
+
await recordEvaluation(result, {
|
|
82
|
+
pass,
|
|
83
|
+
score,
|
|
84
|
+
comment,
|
|
85
|
+
assessmentName: this._assessmentName,
|
|
86
|
+
})
|
|
87
|
+
return { score, comment, pass }
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
async _evaluateSession(results) {
|
|
91
|
+
if (results.length === 0) {
|
|
92
|
+
throw new Error(
|
|
93
|
+
"Judge.evaluate: session assessment requires a non-empty array of chat() results",
|
|
94
|
+
)
|
|
95
|
+
}
|
|
96
|
+
const judgeImpl = await this._ensureJudge(undefined)
|
|
97
|
+
const judgement = await judgeImpl({ outputs: results.flatMap((r) => r.messages ?? []) })
|
|
98
|
+
const { score, pass, comment } = this._judgementResult(judgement)
|
|
99
|
+
LOG.info(`[${this._assessmentName}] pass=${pass} score=${score} — ${comment}`)
|
|
100
|
+
const first = results[0]
|
|
101
|
+
await recordEvaluation(first, {
|
|
102
|
+
score: pass,
|
|
103
|
+
comment,
|
|
104
|
+
sessionId: first.contextId,
|
|
105
|
+
assessmentName: this._assessmentName,
|
|
106
|
+
conversationLevel: true,
|
|
107
|
+
})
|
|
108
|
+
return { score, comment, pass }
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
_judgementResult(judgement) {
|
|
112
|
+
const raw = judgement?.score
|
|
113
|
+
const score = typeof raw === "boolean" ? raw : (raw ?? 0)
|
|
114
|
+
const pass = this._invertedScala
|
|
115
|
+
? typeof score === "boolean"
|
|
116
|
+
? !score
|
|
117
|
+
: score <= 0.5
|
|
118
|
+
: typeof score === "boolean"
|
|
119
|
+
? score
|
|
120
|
+
: score >= 0.5
|
|
121
|
+
const comment = judgement?.comment ?? ""
|
|
122
|
+
return { score, comment, pass }
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
_buildInput(result) {
|
|
126
|
+
if (this._type === "trajectory") {
|
|
127
|
+
return {
|
|
128
|
+
inputs: result.query ?? "",
|
|
129
|
+
outputs: result.messages ?? [],
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
return {
|
|
133
|
+
inputs: result.query ?? "",
|
|
134
|
+
outputs: result.text,
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
// ─── matchToolCall ───────────────────────────────────────────────────────────
|
|
140
|
+
|
|
141
|
+
/** Deterministic tool call assertion. Contributes to success_rate rollup. */
|
|
142
|
+
export function matchToolCall(result, toolName, matcher) {
|
|
143
|
+
const match = (result?.toolCalls ?? []).find((c) => {
|
|
144
|
+
if (c.tool !== toolName) return false
|
|
145
|
+
if (matcher === undefined) return true
|
|
146
|
+
if (typeof matcher === "function") return !!matcher(c.args)
|
|
147
|
+
return _partialMatch(c.args, matcher)
|
|
148
|
+
})
|
|
149
|
+
const pass = !!match
|
|
150
|
+
recordEvaluation(result, { pass })
|
|
151
|
+
return pass
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
function _partialMatch(actual, expected) {
|
|
155
|
+
if (!expected || typeof expected !== "object") return actual === expected
|
|
156
|
+
for (const [k, v] of Object.entries(expected)) {
|
|
157
|
+
if (actual?.[k] !== v) return false
|
|
158
|
+
}
|
|
159
|
+
return true
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
function _judgeOptions(opts, defaultCriteria = "ANSWER_RELEVANCE_PROMPT", argCount = 1) {
|
|
163
|
+
if (argCount === 0 || opts === undefined) return { criteria: defaultCriteria }
|
|
164
|
+
if (typeof opts === "string") return { criteria: opts }
|
|
165
|
+
if (opts && typeof opts === "object")
|
|
166
|
+
return { ...opts, criteria: opts.criteria ?? defaultCriteria }
|
|
167
|
+
throw new TypeError("Judge: constructor argument must be a string or an object")
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
function _assertSingleConstructorArg(args) {
|
|
171
|
+
if (args.length > 1) throw new TypeError("Judge: constructor accepts a single argument")
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
function _assessmentNameFromCriteria(criteria, fallback) {
|
|
175
|
+
const key = _promptKeyFromCriteria(criteria)
|
|
176
|
+
return key ? key.replace(/_PROMPT$/, "").toLowerCase() : fallback
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
function _promptKeyFromCriteria(criteria) {
|
|
180
|
+
if (!criteria || typeof criteria !== "string") return null
|
|
181
|
+
const key = criteria.split(CRITERIA_SEPARATOR, 1)[0].trim()
|
|
182
|
+
return key.endsWith("_PROMPT") ? key : null
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
const INPUT_OUTPUTS_PLACEHOLDER = `<input>
|
|
186
|
+
{inputs}
|
|
187
|
+
</input>
|
|
188
|
+
|
|
189
|
+
<output>
|
|
190
|
+
{outputs}
|
|
191
|
+
</output>`
|
|
192
|
+
|
|
193
|
+
function _resolvePrompt(openevals, criteria) {
|
|
194
|
+
if (Object.prototype.hasOwnProperty.call(openevals, criteria)) return openevals[criteria]
|
|
195
|
+
const key = _promptKeyFromCriteria(criteria)
|
|
196
|
+
if (!key || !Object.prototype.hasOwnProperty.call(openevals, key))
|
|
197
|
+
return `${criteria}${CRITERIA_SEPARATOR}${INPUT_OUTPUTS_PLACEHOLDER}`
|
|
198
|
+
const rest = criteria.slice(criteria.indexOf(key) + key.length).trimStart()
|
|
199
|
+
return rest ? `${openevals[key]}${CRITERIA_SEPARATOR}${rest}` : openevals[key]
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
// ─── Helpers ──────────────────────────────────────────────────────────────────
|
|
203
|
+
|
|
204
|
+
async function _loadOpenevals() {
|
|
205
|
+
const savedExpect = globalThis.expect
|
|
206
|
+
let openevals
|
|
207
|
+
try {
|
|
208
|
+
openevals = await import("openevals")
|
|
209
|
+
} catch (err) {
|
|
210
|
+
throw new Error(
|
|
211
|
+
"openevals is required for Judge.evaluate(). Install it as a devDependency:\n npm install --save-dev openevals\n" +
|
|
212
|
+
`Original error: ${err.message}`,
|
|
213
|
+
{ cause: err },
|
|
214
|
+
)
|
|
215
|
+
}
|
|
216
|
+
if (savedExpect && globalThis.expect !== savedExpect) globalThis.expect = savedExpect
|
|
217
|
+
return openevals
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
async function _buildLlm() {
|
|
221
|
+
const name = "llm"
|
|
222
|
+
const { kind, impl, ...options } = cds.env.requires[name] ?? {}
|
|
223
|
+
const providerImpl = impl ?? cds.env.requires.kinds?.[kind]?.impl
|
|
224
|
+
if (!providerImpl) throw new Error("No service implementation found for " + name)
|
|
225
|
+
const { default: LLMProvider } = await import(providerImpl)
|
|
226
|
+
const llm = new LLMProvider(name, options)
|
|
227
|
+
llm[Symbol.for("@cap-js/agents:instrumented")] = true
|
|
228
|
+
return llm
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
async function _loadJudgeImpl(criteria, assessmentName, continuous, type) {
|
|
232
|
+
const openevals = await _loadOpenevals()
|
|
233
|
+
const llm = await _buildLlm()
|
|
234
|
+
const prompt = _resolvePrompt(openevals, criteria)
|
|
235
|
+
if (type === "trajectory") {
|
|
236
|
+
return openevals.createTrajectoryLLMAsJudge({ judge: llm, prompt, assessmentName })
|
|
237
|
+
}
|
|
238
|
+
return openevals.createLLMAsJudge({ judge: llm, prompt, continuous, assessmentName })
|
|
239
|
+
}
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
import { evalRun } from "../eval/eval-run.js"
|
|
2
|
+
|
|
3
|
+
const PATCHED = Symbol.for("@cap-js/agents:eval-describe-patched")
|
|
4
|
+
|
|
5
|
+
export function installEvalDescribe({
|
|
6
|
+
target = globalThis,
|
|
7
|
+
evalRun: registerEvalRun = evalRun,
|
|
8
|
+
} = {}) {
|
|
9
|
+
const original = target.describe
|
|
10
|
+
if (typeof original !== "function") return false
|
|
11
|
+
if (original[PATCHED]) return false
|
|
12
|
+
|
|
13
|
+
let depth = 0
|
|
14
|
+
|
|
15
|
+
const wrapSuite = (suite) => {
|
|
16
|
+
if (typeof suite !== "function") return suite
|
|
17
|
+
|
|
18
|
+
const wrapped = function evalDescribe(name, factory, ...args) {
|
|
19
|
+
if (typeof factory !== "function") return suite.call(this, name, factory, ...args)
|
|
20
|
+
|
|
21
|
+
return suite.call(
|
|
22
|
+
this,
|
|
23
|
+
name,
|
|
24
|
+
function evalSuite(...suiteArgs) {
|
|
25
|
+
depth += 1
|
|
26
|
+
const topLevel = depth === 1
|
|
27
|
+
try {
|
|
28
|
+
if (topLevel && typeof name === "string") registerEvalRun({ name })
|
|
29
|
+
return factory.apply(this, suiteArgs)
|
|
30
|
+
} finally {
|
|
31
|
+
depth -= 1
|
|
32
|
+
}
|
|
33
|
+
},
|
|
34
|
+
...args,
|
|
35
|
+
)
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
copySuiteProperties(suite, wrapped, wrapSuite)
|
|
39
|
+
return wrapped
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
target.describe = wrapSuite(original)
|
|
43
|
+
target.describe[PATCHED] = true
|
|
44
|
+
target.describe._original = original
|
|
45
|
+
return true
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
function copySuiteProperties(source, target, wrapSuite) {
|
|
49
|
+
for (const key of Reflect.ownKeys(source)) {
|
|
50
|
+
if (["length", "name", "prototype"].includes(key)) continue
|
|
51
|
+
const descriptor = Object.getOwnPropertyDescriptor(source, key)
|
|
52
|
+
if (!descriptor) continue
|
|
53
|
+
if (typeof descriptor.value === "function") {
|
|
54
|
+
if (["each", "skipIf", "runIf"].includes(key)) {
|
|
55
|
+
descriptor.value = function describeFactory(...args) {
|
|
56
|
+
return wrapSuite(source[key].apply(this, args))
|
|
57
|
+
}
|
|
58
|
+
} else {
|
|
59
|
+
descriptor.value = wrapSuite(descriptor.value)
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
try {
|
|
63
|
+
Object.defineProperty(target, key, descriptor)
|
|
64
|
+
} catch {
|
|
65
|
+
// Vitest may expose non-configurable helper properties. The base describe
|
|
66
|
+
// still works, so ignore properties that cannot be mirrored.
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
}
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
/* global beforeAll, afterEach, afterAll */
|
|
2
|
+
import cds from "@sap/cds"
|
|
3
|
+
import {
|
|
4
|
+
postMlflowAssessment,
|
|
5
|
+
createEvalRun,
|
|
6
|
+
closeEvalRun,
|
|
7
|
+
logMlflowMetrics,
|
|
8
|
+
} from "../telemetry/mlflow/evaluation.js"
|
|
9
|
+
import { flushMlflowTraces } from "../telemetry/mlflow/tracing.js"
|
|
10
|
+
|
|
11
|
+
export function getActiveRunState() {
|
|
12
|
+
return cds._activeEvalRun ?? null
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
export function evalRun(opts = {}) {
|
|
16
|
+
if (typeof beforeAll !== "function" || typeof afterAll !== "function") return
|
|
17
|
+
if (!cds.env.agents?.mlflow) return
|
|
18
|
+
|
|
19
|
+
let state = null
|
|
20
|
+
|
|
21
|
+
function _makeState(runId, mlflowRunId) {
|
|
22
|
+
return {
|
|
23
|
+
runId,
|
|
24
|
+
mlflowRunId,
|
|
25
|
+
validationsByTask: new Map(),
|
|
26
|
+
}
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
beforeAll(async () => {
|
|
30
|
+
const runId = cds.utils.uuid()
|
|
31
|
+
const mlflowRunId = await createEvalRun(opts).catch(() => null)
|
|
32
|
+
state = _makeState(runId, mlflowRunId)
|
|
33
|
+
cds._activeEvalRun = state
|
|
34
|
+
})
|
|
35
|
+
|
|
36
|
+
if (typeof afterEach === "function") {
|
|
37
|
+
afterEach(async () => {
|
|
38
|
+
if (state) await _flushValidations(state)
|
|
39
|
+
})
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
afterAll(async () => {
|
|
43
|
+
if (state) await _flushValidations(state)
|
|
44
|
+
await flushMlflowTraces()
|
|
45
|
+
await closeEvalRun(state?.mlflowRunId).catch(() => {})
|
|
46
|
+
if (cds._activeEvalRun === state) cds._activeEvalRun = null
|
|
47
|
+
state = null
|
|
48
|
+
})
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
async function _flushValidations(state) {
|
|
52
|
+
if (!state?.validationsByTask.size) return
|
|
53
|
+
|
|
54
|
+
const tasks = []
|
|
55
|
+
for (const [, entry] of state.validationsByTask) {
|
|
56
|
+
const { passes, traceId } = entry
|
|
57
|
+
if (!passes.length) continue
|
|
58
|
+
|
|
59
|
+
const success_rate = passes.every(Boolean) ? 1 : 0
|
|
60
|
+
const output_correctness = passes.filter(Boolean).length / passes.length
|
|
61
|
+
const codeOpts = { sourceType: "CODE" }
|
|
62
|
+
|
|
63
|
+
if (state.mlflowRunId) {
|
|
64
|
+
tasks.push(
|
|
65
|
+
logMlflowMetrics(state.mlflowRunId, { success_rate, output_correctness }).catch(() => {}),
|
|
66
|
+
)
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
if (traceId) {
|
|
70
|
+
tasks.push(
|
|
71
|
+
flushMlflowTraces().then(() =>
|
|
72
|
+
Promise.all([
|
|
73
|
+
postMlflowAssessment(traceId, success_rate, "", "success_rate", null, codeOpts).catch(
|
|
74
|
+
() => {},
|
|
75
|
+
),
|
|
76
|
+
postMlflowAssessment(
|
|
77
|
+
traceId,
|
|
78
|
+
output_correctness,
|
|
79
|
+
"",
|
|
80
|
+
"output_correctness",
|
|
81
|
+
null,
|
|
82
|
+
codeOpts,
|
|
83
|
+
).catch(() => {}),
|
|
84
|
+
]),
|
|
85
|
+
),
|
|
86
|
+
)
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
await Promise.all(tasks)
|
|
91
|
+
state.validationsByTask.clear()
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
export function recordEvaluation(result, assessment = {}) {
|
|
95
|
+
if (!cds.env.agents?.mlflow) return
|
|
96
|
+
const { pass, score, comment, ...config } = assessment
|
|
97
|
+
// Conversation level evaluations shall not be included in the per task roll-up
|
|
98
|
+
if (pass !== undefined && !config.conversationLevel) _addValidation(result, pass)
|
|
99
|
+
if (score === undefined) return
|
|
100
|
+
return _postAssessmentScore(result, score, comment, config)
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
function _addValidation(result, pass) {
|
|
104
|
+
const state = result?._evalState ?? cds._activeEvalRun
|
|
105
|
+
if (!state || !result?.taskId) return
|
|
106
|
+
const key = result.taskId
|
|
107
|
+
if (!state.validationsByTask.has(key)) {
|
|
108
|
+
state.validationsByTask.set(key, { passes: [], traceId: result.traceId })
|
|
109
|
+
}
|
|
110
|
+
state.validationsByTask.get(key).passes.push(pass)
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
async function _postAssessmentScore(result, score, comment, config) {
|
|
114
|
+
const traceId = config.traceId ?? result?.traceId
|
|
115
|
+
if (!traceId) return
|
|
116
|
+
|
|
117
|
+
const opts = { sourceType: config.sourceType }
|
|
118
|
+
await flushMlflowTraces()
|
|
119
|
+
|
|
120
|
+
// conversationLevel: session assessment — post with session metadata only
|
|
121
|
+
if (config.conversationLevel) {
|
|
122
|
+
await postMlflowAssessment(
|
|
123
|
+
traceId,
|
|
124
|
+
score,
|
|
125
|
+
comment ?? "",
|
|
126
|
+
config.assessmentName,
|
|
127
|
+
config.model ?? null,
|
|
128
|
+
{ ...opts, metadata: { "mlflow.trace.session": config.sessionId ?? "" } },
|
|
129
|
+
)
|
|
130
|
+
} else {
|
|
131
|
+
// Single-turn: post to this trace only, no session metadata
|
|
132
|
+
await postMlflowAssessment(
|
|
133
|
+
traceId,
|
|
134
|
+
score,
|
|
135
|
+
comment ?? "",
|
|
136
|
+
config.assessmentName,
|
|
137
|
+
config.model ?? null,
|
|
138
|
+
opts,
|
|
139
|
+
)
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
export async function logMlflowMetricsForResult(result, state = null) {
|
|
144
|
+
state = state ?? cds._activeEvalRun
|
|
145
|
+
if (!state?.mlflowRunId) return
|
|
146
|
+
await logMlflowMetrics(state.mlflowRunId, result.metrics).catch(() => {})
|
|
147
|
+
}
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
function _hrtimeToMs(hr) {
|
|
2
|
+
return hr[0] * 1000 + hr[1] / 1e6
|
|
3
|
+
}
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* @param {object[]} spans Finished OTel spans from span-collector.js
|
|
7
|
+
* @returns {{ input_tokens, output_tokens, total_tokens, tool_call_count, latency_ms, cost_usd }}
|
|
8
|
+
*/
|
|
9
|
+
export function metricsFromSpans(spans) {
|
|
10
|
+
let input_tokens = 0
|
|
11
|
+
let output_tokens = 0
|
|
12
|
+
let tool_call_count = 0
|
|
13
|
+
let cost_usd = 0
|
|
14
|
+
let latency_ms = null
|
|
15
|
+
|
|
16
|
+
for (const span of spans) {
|
|
17
|
+
const attrs = span.attributes ?? {}
|
|
18
|
+
const op = attrs["gen_ai.operation.name"]
|
|
19
|
+
|
|
20
|
+
if (op === "chat") {
|
|
21
|
+
input_tokens += Number(attrs["gen_ai.usage.input_tokens"] ?? 0)
|
|
22
|
+
output_tokens += Number(attrs["gen_ai.usage.output_tokens"] ?? 0)
|
|
23
|
+
|
|
24
|
+
// mlflow.llm.cost is set by apps themselves
|
|
25
|
+
const rawCost = attrs["mlflow.llm.cost"]
|
|
26
|
+
if (rawCost) {
|
|
27
|
+
try {
|
|
28
|
+
const c = typeof rawCost === "string" ? JSON.parse(rawCost) : rawCost
|
|
29
|
+
cost_usd += c.total_cost ?? 0
|
|
30
|
+
} catch {
|
|
31
|
+
/* malformed — skip */
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
if (op === "execute_tool") {
|
|
37
|
+
tool_call_count++
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
if (op === "invoke_agent" && span.endTime && span.startTime) {
|
|
41
|
+
latency_ms = _hrtimeToMs(span.endTime) - _hrtimeToMs(span.startTime)
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
return {
|
|
46
|
+
input_tokens,
|
|
47
|
+
output_tokens,
|
|
48
|
+
total_tokens: input_tokens + output_tokens,
|
|
49
|
+
tool_call_count,
|
|
50
|
+
latency_ms,
|
|
51
|
+
cost_usd: cost_usd > 0 ? cost_usd : null,
|
|
52
|
+
}
|
|
53
|
+
}
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
const _sessions = new Map()
|
|
2
|
+
let _registered = false
|
|
3
|
+
let _sessionCounter = 0
|
|
4
|
+
|
|
5
|
+
const _processor = {
|
|
6
|
+
onStart() {},
|
|
7
|
+
onEnd(span) {
|
|
8
|
+
if (_sessions.size === 0) return
|
|
9
|
+
for (const session of _sessions.values()) {
|
|
10
|
+
session.spans.push(span)
|
|
11
|
+
}
|
|
12
|
+
},
|
|
13
|
+
async forceFlush() {},
|
|
14
|
+
async shutdown() {},
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
async function _ensureRegistered() {
|
|
18
|
+
if (_registered) return
|
|
19
|
+
_registered = true
|
|
20
|
+
try {
|
|
21
|
+
const { trace } = await import("@opentelemetry/api")
|
|
22
|
+
const provider = trace.getTracerProvider()
|
|
23
|
+
const delegate = provider.getDelegate?.() || provider
|
|
24
|
+
if (delegate.constructor?.name === "NoopTracerProvider") return
|
|
25
|
+
if (typeof delegate.addSpanProcessor === "function") {
|
|
26
|
+
delegate.addSpanProcessor(_processor)
|
|
27
|
+
} else if (Array.isArray(delegate._activeSpanProcessor?._spanProcessors)) {
|
|
28
|
+
delegate._activeSpanProcessor._spanProcessors.push(_processor)
|
|
29
|
+
}
|
|
30
|
+
} catch {
|
|
31
|
+
/* OTel not present */
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
function _openSession() {
|
|
36
|
+
const id = ++_sessionCounter
|
|
37
|
+
const session = { spans: [] }
|
|
38
|
+
_sessions.set(id, session)
|
|
39
|
+
return {
|
|
40
|
+
collect() {
|
|
41
|
+
_sessions.delete(id)
|
|
42
|
+
return session.spans
|
|
43
|
+
},
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
export async function startCollection() {
|
|
48
|
+
await _ensureRegistered()
|
|
49
|
+
return _openSession()
|
|
50
|
+
}
|
package/lib/index.js
CHANGED
|
@@ -122,7 +122,7 @@ export default function A2AProtocolAdapter(srv, options = {}) {
|
|
|
122
122
|
}
|
|
123
123
|
})
|
|
124
124
|
|
|
125
|
-
if (
|
|
125
|
+
if (cds.env?.server?.index) {
|
|
126
126
|
linkProviders.push((entity, endpoint) => {
|
|
127
127
|
if (entity || endpoint?.kind !== "agent") return undefined
|
|
128
128
|
return {
|
|
@@ -161,7 +161,8 @@ export default function A2AProtocolAdapter(srv, options = {}) {
|
|
|
161
161
|
}
|
|
162
162
|
|
|
163
163
|
router.get("/.well-known/agent-card.json", (req, res) => {
|
|
164
|
-
const
|
|
164
|
+
const proto = req.headers["x-forwarded-proto"]?.split(",")[0].trim() || req.protocol
|
|
165
|
+
const url = proxyUrl || `${proto}://${req.get("host")}${req.baseUrl}`
|
|
165
166
|
// Regenerate agent card when feature toggles are active (annotations may differ)
|
|
166
167
|
let card
|
|
167
168
|
if (cds.context?.features && Object.keys(cds.context.features).length > 0) {
|
|
@@ -182,7 +183,7 @@ export default function A2AProtocolAdapter(srv, options = {}) {
|
|
|
182
183
|
res.json(card)
|
|
183
184
|
})
|
|
184
185
|
|
|
185
|
-
if (
|
|
186
|
+
if (cds.env?.server?.index) {
|
|
186
187
|
router.use("/preview", previewAuthChallenge(srv), preview(agentCard.name || srv.name))
|
|
187
188
|
}
|
|
188
189
|
|
package/lib/models/aicore.js
CHANGED
|
@@ -1,11 +1,13 @@
|
|
|
1
1
|
import cds from "@sap/cds"
|
|
2
2
|
import { OrchestrationClient } from "@sap-ai-sdk/langchain"
|
|
3
|
-
import { circuitBreaker, timeout } from "
|
|
3
|
+
import { circuitBreaker, timeout } from "../utils/resilience.js"
|
|
4
4
|
|
|
5
5
|
import { SystemMessage, ToolMessage, HumanMessage, AIMessage } from "@langchain/core/messages"
|
|
6
6
|
import { ms4 } from "../utils/utils.js"
|
|
7
|
+
import { syncSystemPrompt } from "../telemetry/mlflow/prompts.js"
|
|
7
8
|
|
|
8
9
|
const LOG = cds.log("agents")
|
|
10
|
+
const DEFAULT_STREAM_DELIMITERS = [".", "!", "?"]
|
|
9
11
|
|
|
10
12
|
class _InstrumentedOrchestrationClient extends OrchestrationClient {
|
|
11
13
|
constructor(name, options) {
|
|
@@ -18,6 +20,7 @@ class _InstrumentedOrchestrationClient extends OrchestrationClient {
|
|
|
18
20
|
// and extra ReAct iterations — see PR #188 review).
|
|
19
21
|
const params =
|
|
20
22
|
options.params || (deepAgent ? { max_tokens: 4096, temperature: 0 } : cds.env.agents?.params)
|
|
23
|
+
const auditParams = params && typeof params === "object" ? { ...params } : params
|
|
21
24
|
|
|
22
25
|
LOG.debug("Initializing LLM", { model, deepAgent: !!deepAgent })
|
|
23
26
|
|
|
@@ -42,15 +45,15 @@ class _InstrumentedOrchestrationClient extends OrchestrationClient {
|
|
|
42
45
|
{
|
|
43
46
|
promptTemplating: { model: { name: model, params } },
|
|
44
47
|
...(filtering && { filtering }),
|
|
48
|
+
},
|
|
49
|
+
{
|
|
45
50
|
// `streaming` controls the SDK's auto-stream-and-concatenate in _generate()
|
|
46
51
|
// (i.e. direct model.invoke() calls). It does NOT gate LangGraph token
|
|
47
52
|
// streaming — that is driven by graph.stream(streamMode:["messages"]) via a
|
|
48
53
|
// streaming callback handler + our overridden _streamResponseChunks, and
|
|
49
54
|
// works for every agent regardless of this flag (deep or managed alike).
|
|
50
55
|
// Default on; `streaming: false` opts out.
|
|
51
|
-
|
|
52
|
-
},
|
|
53
|
-
{
|
|
56
|
+
streaming: streaming !== false,
|
|
54
57
|
onFailedAttempt: (err) => {
|
|
55
58
|
// Abort retries when circuit breaker is open (otherwise pRetry delays ~30-60s)
|
|
56
59
|
if (err.code === "EOPENBREAKER" || err.message === "Breaker is open") {
|
|
@@ -62,7 +65,7 @@ class _InstrumentedOrchestrationClient extends OrchestrationClient {
|
|
|
62
65
|
destination,
|
|
63
66
|
)
|
|
64
67
|
this.name = name
|
|
65
|
-
this.options = { ...options, params, contentFilter, flatten }
|
|
68
|
+
this.options = { ...options, params: auditParams, contentFilter, flatten }
|
|
66
69
|
}
|
|
67
70
|
|
|
68
71
|
async _generate(messages, opts, runManager) {
|
|
@@ -88,7 +91,7 @@ class _InstrumentedOrchestrationClient extends OrchestrationClient {
|
|
|
88
91
|
opts = _withMiddleware(this, opts)
|
|
89
92
|
const prepared = _prepareMessages(messages, { flatten, model, opts })
|
|
90
93
|
const inputMessages = prepared.inputMessages
|
|
91
|
-
opts = prepared.opts
|
|
94
|
+
opts = withDefaultStreamDelimiters(prepared.opts)
|
|
92
95
|
|
|
93
96
|
let turnHasToolCall = false
|
|
94
97
|
for await (const chunk of super._streamResponseChunks(inputMessages, opts, runManager)) {
|
|
@@ -144,9 +147,23 @@ function _prepareMessages(messages, { flatten, model, opts }) {
|
|
|
144
147
|
opts = { ...opts, tools }
|
|
145
148
|
}
|
|
146
149
|
}
|
|
150
|
+
syncSystemPrompt(inputMessages)
|
|
147
151
|
return { inputMessages, opts }
|
|
148
152
|
}
|
|
149
153
|
|
|
154
|
+
export function withDefaultStreamDelimiters(opts = {}) {
|
|
155
|
+
return {
|
|
156
|
+
...opts,
|
|
157
|
+
streamOptions: {
|
|
158
|
+
...opts.streamOptions,
|
|
159
|
+
global: {
|
|
160
|
+
...opts.streamOptions?.global,
|
|
161
|
+
delimiters: opts.streamOptions?.global?.delimiters ?? DEFAULT_STREAM_DELIMITERS,
|
|
162
|
+
},
|
|
163
|
+
},
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
|
|
150
167
|
// Claude currently only supports caching of type ephemeral. TTL can differ between 5min or 1h but
|
|
151
168
|
// we use the 5min default
|
|
152
169
|
const CACHE_CONTROL_EPHEMERAL = { type: "ephemeral" }
|