@cap-js/agents 0.9.6 → 0.9.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. package/README.md +2 -131
  2. package/_i18n/messages.properties +2 -0
  3. package/_i18n/messages_ar.properties +48 -0
  4. package/_i18n/messages_bg.properties +48 -0
  5. package/_i18n/messages_cs.properties +48 -0
  6. package/_i18n/messages_da.properties +48 -0
  7. package/_i18n/messages_de.properties +48 -0
  8. package/_i18n/messages_el.properties +48 -0
  9. package/_i18n/messages_en.properties +48 -0
  10. package/_i18n/messages_es.properties +48 -0
  11. package/_i18n/messages_es_MX.properties +48 -0
  12. package/_i18n/messages_fi.properties +48 -0
  13. package/_i18n/messages_fr.properties +48 -0
  14. package/_i18n/messages_he.properties +48 -0
  15. package/_i18n/messages_hr.properties +48 -0
  16. package/_i18n/messages_hu.properties +48 -0
  17. package/_i18n/messages_it.properties +48 -0
  18. package/_i18n/messages_ja.properties +48 -0
  19. package/_i18n/messages_kk.properties +48 -0
  20. package/_i18n/messages_ko.properties +48 -0
  21. package/_i18n/messages_ms.properties +48 -0
  22. package/_i18n/messages_nl.properties +48 -0
  23. package/_i18n/messages_no.properties +48 -0
  24. package/_i18n/messages_pl.properties +48 -0
  25. package/_i18n/messages_pt.properties +48 -0
  26. package/_i18n/messages_ro.properties +48 -0
  27. package/_i18n/messages_ru.properties +48 -0
  28. package/_i18n/messages_sh.properties +48 -0
  29. package/_i18n/messages_sk.properties +48 -0
  30. package/_i18n/messages_sl.properties +48 -0
  31. package/_i18n/messages_sv.properties +48 -0
  32. package/_i18n/messages_th.properties +48 -0
  33. package/_i18n/messages_tr.properties +48 -0
  34. package/_i18n/messages_uk.properties +48 -0
  35. package/_i18n/messages_vi.properties +48 -0
  36. package/_i18n/messages_zh_CN.properties +48 -0
  37. package/_i18n/messages_zh_TW.properties +48 -0
  38. package/cds-plugin.js +15 -10
  39. package/lib/agents/middleware/content-filter.js +4 -0
  40. package/lib/agents/middleware/hitl-decision-note-injector.js +18 -6
  41. package/lib/agents/middleware/hitl.js +17 -6
  42. package/lib/agents/middleware/index.js +3 -1
  43. package/lib/agents/middleware/masking.js +127 -0
  44. package/lib/agents/middleware/remote-mcp.js +10 -5
  45. package/lib/agents/middleware/status-update.js +209 -76
  46. package/lib/agents/middleware/tool-wrap.js +1 -2
  47. package/lib/agents/summarize-on-timeout.js +5 -2
  48. package/lib/config/local.js +121 -8
  49. package/lib/eval/eval-run.js +50 -4
  50. package/lib/masking/index.js +90 -0
  51. package/lib/masking/store.js +80 -0
  52. package/lib/masking/structured/findElements.js +459 -0
  53. package/lib/masking/structured/index.js +170 -0
  54. package/lib/masking/unstructured/dpi.js +84 -0
  55. package/lib/masking/unstructured/hana.js +72 -0
  56. package/lib/masking/unstructured/index.js +52 -0
  57. package/lib/models/aicore.js +16 -3
  58. package/lib/models/openai.js +9 -0
  59. package/lib/preview/chat.html +623 -119
  60. package/lib/protocol/persistence/cleanup.js +13 -4
  61. package/lib/telemetry/chat-tracing.js +9 -2
  62. package/lib/telemetry/mlflow/evaluation.js +83 -4
  63. package/lib/telemetry/mlflow/exporter/DatabricksExporter.js +39 -5
  64. package/lib/telemetry/mlflow/exporter/MlflowExporter.js +43 -4
  65. package/lib/telemetry/mlflow/index.js +1 -7
  66. package/lib/telemetry/mlflow/prompts.js +15 -11
  67. package/lib/telemetry/mlflow/tracing.js +3 -3
  68. package/lib/telemetry/span-masking.js +165 -0
  69. package/lib/telemetry/tool-tracing.js +4 -4
  70. package/lib/utils/markdown.js +3 -10
  71. package/lib/utils/resilience.js +2 -1
  72. package/lib/utils/toml.js +67 -0
  73. package/lib/utils/usage.js +34 -0
  74. package/lib/utils/utils.js +43 -1
  75. package/package.json +33 -26
  76. package/srv/handlers/graph-executor/crash-handler.js +47 -0
  77. package/srv/handlers/graph-executor/hitl.js +27 -0
  78. package/srv/handlers/graph-executor.js +69 -20
  79. package/srv/handlers/index.js +6 -1
  80. package/srv/handlers/mcp-tools.js +11 -2
  81. package/srv/handlers/subagent-tools.js +13 -4
  82. package/srv/handlers/system-prompt.js +30 -20
  83. package/srv/handlers/tools.js +14 -13
@@ -1,63 +1,103 @@
1
1
  import cds from "@sap/cds"
2
2
  import { createMiddleware } from "langchain"
3
- import { ToolMessage } from "@langchain/core/messages"
3
+ import { AIMessage, ToolMessage } from "@langchain/core/messages"
4
+ import { resolvePseudonyms } from "../../masking/index.js"
5
+ import { serviceLabel, declaredServiceLabel, mcpToolKind } from "../../utils/utils.js"
4
6
 
5
7
  /**
6
- * Resolves a human-readable label for a tool call.
7
- *
8
- * - query tool → resolves entity label from CDS model: @Common.Label, @title, or i18n
9
- * - actions → resolves label from action definition
10
- * - fallback → tool name as-is
8
+ * Resolve the metadata for a tool call, or undefined when the tool is unknown.
9
+ * Static tools come from the precomputed `toolMeta` map; remote MCP tools are
10
+ * resolved per-request from cds.context.__mcpDynamicTools.
11
11
  */
12
- function resolveToolLabel(tc) {
13
- const serviceName = cds.context?.["agent.service"]
14
- const srv = serviceName && cds.services?.[serviceName]
15
- const model = cds.context?.model ?? cds.model
12
+ function resolveMeta(tc, toolMeta) {
13
+ const meta = toolMeta.get(tc.name)
14
+ if (meta) return meta
16
15
 
17
- // Query tool: extract entity target and resolve its label
18
- if (tc.name === "query" || tc.name?.endsWith("_query")) {
19
- let entityName = tc.args?.entity ? `${serviceName}.${tc.args.entity}` : undefined
20
-
21
- // SQL format: no entity arg — parse SQL to extract FROM target
22
- if (!entityName && tc.args?.cql) {
23
- try {
24
- const cqn = cds.parse.cql(tc.args.cql)
25
- const ref0 = cqn.SELECT?.from?.ref?.[0]
26
- const targetName = ref0?.id ?? ref0
27
- if (targetName && serviceName && !targetName.startsWith(serviceName + ".")) {
28
- entityName = `${serviceName}.${targetName}`
29
- } else {
30
- entityName = targetName
31
- }
32
- } catch {
33
- /* fallback to tc.name below */
34
- }
35
- }
36
- if (entityName) {
37
- const entityDef = model.definitions?.[entityName]
38
- if (entityDef) {
39
- const label = cds.i18n?.labels?.at(entityDef)
40
- if (label) return label
41
- }
42
- }
43
- // Fallback: strip service prefix if present
44
- if (entityName && srv?.name && entityName.startsWith(srv.name + ".")) {
45
- return entityName.slice(srv.name.length + 1)
16
+ const cache = cds.context?.__mcpDynamicTools
17
+ if (cache) {
18
+ for (const entry of Object.values(cache)) {
19
+ const t = entry.tools?.find((t) => t.name === tc.name)
20
+ if (t) return t.metadata
46
21
  }
47
- return entityName || tc.name
48
22
  }
23
+ return undefined
24
+ }
25
+
26
+ /**
27
+ * Resolve the kind ("query" | "describe" | "action" | "agent") for a tool call.
28
+ * Prefer the build-time `meta.kind` (classified from the unprefixed name, so it
29
+ * survives prefixing/truncation); else sniff `tc.name` for own-service tools.
30
+ */
31
+ function resolveToolKind(tc, meta) {
32
+ if (meta?.kind) return meta.kind
33
+ return mcpToolKind(tc.name)
34
+ }
35
+
36
+ /** Resolve an entity's display label (declared i18n label, else its simple name). */
37
+ function entityLabel(name, serviceName, model) {
38
+ const fq = serviceName && !name.startsWith(serviceName + ".") ? `${serviceName}.${name}` : name
39
+ const def = model?.definitions?.[fq]
40
+ return (
41
+ (def && cds.i18n?.labels?.at(def)) ||
42
+ (fq.includes(".") ? fq.slice(fq.lastIndexOf(".") + 1) : fq)
43
+ )
44
+ }
49
45
 
50
- // Action/function tools: resolve from service definition
51
- if (srv) {
52
- // Try as action on service
53
- const actionDef = model.definitions[`${srv.name}.${tc.name}`]
54
- if (actionDef) {
55
- const label = cds.i18n?.labels?.at(actionDef)
56
- if (label) return label
46
+ /** Resolve the query target: the `entity` arg, or the FROM target parsed from `cql`. */
47
+ function queryTarget(tc) {
48
+ if (tc.args?.entity) return tc.args.entity
49
+ if (tc.args?.cql) {
50
+ try {
51
+ const ref0 = cds.parse.cql(tc.args.cql).SELECT?.from?.ref?.[0]
52
+ return ref0?.id ?? ref0 ?? undefined
53
+ } catch {
54
+ /* no target derivable */
57
55
  }
58
56
  }
57
+ return undefined
58
+ }
59
+
60
+ /**
61
+ * Resolve a human-readable label for a tool call:
62
+ * - agent → declared service label, else the subagent's agent-card name
63
+ * - describe → service label (declared, else the FQ service name)
64
+ * - query → entity label, prefixed with the service only when it has a declared label
65
+ * - action → action label, prefixed with the service only when it has a declared label
66
+ */
67
+ function resolveToolLabel(tc, meta, kind = resolveToolKind(tc, meta)) {
68
+ const model = cds.context?.model ?? cds.model
69
+ const serviceName = meta?.serviceName ?? cds.context?.["agent.service"] // REVISIT: why are we using cds.context here?
70
+
71
+ if (kind === "agent") {
72
+ return declaredServiceLabel(meta?.serviceName) || meta?.agentName || tc.name
73
+ }
74
+
75
+ if (kind === "describe") {
76
+ let parts = []
77
+ if (tc.args?.entities) parts.push(...tc.args.entities.map(e => entityLabel(e, serviceName, model)))
78
+ if (tc.args?.actions) parts.push(...tc.args.actions.map(a => entityLabel(a, serviceName, model)))
79
+ if (!parts.length) parts.push(serviceLabel(serviceName) || tc.name)
80
+ return parts.join(", ")
81
+ }
82
+
83
+ const svcLabel = declaredServiceLabel(serviceName)
84
+ const withService = (detail) => (svcLabel ? `${svcLabel} · ${detail}` : detail)
85
+
86
+ if (kind === "query") {
87
+ const target = queryTarget(tc)
88
+ if (target) return withService(entityLabel(target, serviceName, model))
89
+ return svcLabel || tc.name
90
+ }
59
91
 
60
- return tc.name
92
+ // meta.actionName is the un-prefixed name for MCP action tools (tc.name is
93
+ // "{service}_{action}"); own-service tools already carry the bare name.
94
+ const actionName = meta?.actionName ?? tc.args?.action ?? tc.name
95
+ if (serviceName) {
96
+ const actionDef = model?.definitions?.[`${serviceName}.${actionName}`]
97
+ const declared = actionDef && cds.i18n?.labels?.at(actionDef) || actionName
98
+ if (declared) return withService(declared)
99
+ }
100
+ return withService(actionName)
61
101
  }
62
102
 
63
103
  /**
@@ -86,29 +126,106 @@ export function publishStatus(text) {
86
126
  }
87
127
 
88
128
  /**
89
- * beforeModel hook: emit "Processing tool response" when tools just finished.
129
+ * Reads tool-call visibility config from the current request's metadata.
130
+ * Controlled entirely by the client: presence of userMessage.metadata["tool-status-update"] enables it.
131
+ */
132
+ function toolCallConfig() {
133
+ const requestMeta = cds.context?.["agent.request.metadata"]?.["tool-status-update"]
134
+ return {
135
+ enabled: requestMeta !== undefined,
136
+ args: requestMeta?.args ?? true,
137
+ result: requestMeta?.result ?? true,
138
+ }
139
+ }
140
+
141
+ function serialize(v) {
142
+ return resolvePseudonyms(typeof v === "string" ? v : JSON.stringify(v))
143
+ }
144
+
145
+ /**
146
+ * Publishes an artifact-update for a tool call (start or end).
147
+ */
148
+ function publishToolCallArtifact(tc, { meta, status, lastChunk, result }) {
149
+ const eventBus = cds.context?.["agent.eventBus"]
150
+ const config = toolCallConfig()
151
+ if (!eventBus || !config.enabled) return
152
+
153
+ let kind = resolveToolKind(tc, meta)
154
+ let label = resolveToolLabel(tc, meta)
155
+ eventBus.publish({
156
+ kind: "artifact-update",
157
+ taskId: cds.context["agent.task.id"],
158
+ contextId: cds.context["agent.context.id"],
159
+ append: false,
160
+ lastChunk,
161
+ artifact: {
162
+ artifactId: `tool-call-${tc.id}`,
163
+ parts: [
164
+ {
165
+ kind: "data",
166
+ data: {
167
+ type: "tool-call",
168
+ name: tc.name,
169
+ kind,
170
+ label,
171
+ status,
172
+ ...(config.args && { args: serialize(tc.args) }),
173
+ ...(result !== undefined && config.result && { result: serialize(result) }),
174
+ },
175
+ },
176
+ ],
177
+ },
178
+ })
179
+ }
180
+
181
+ /**
182
+ * beforeModel hook: emit "Processing tool response" + tool-call completion events.
90
183
  */
91
- export function beforeModelHook(state) {
184
+ export function beforeModelHook(state, toolMeta) {
92
185
  if (!cds.context?.["agent.eventBus"]) return {}
93
186
 
94
187
  const msgs = state.messages
95
188
  if (!msgs?.length) return {}
96
- const lastMsg = msgs[msgs.length - 1]
97
- if (!ToolMessage.isInstance(lastMsg)) return {}
189
+ if (!ToolMessage.isInstance(msgs[msgs.length - 1])) return {}
98
190
 
99
- // Plural if second-to-last is also a ToolMessage
100
- const plural = msgs.length >= 2 && ToolMessage.isInstance(msgs[msgs.length - 2])
191
+ const firstToolIdx = msgs.findLastIndex((m) => !ToolMessage.isInstance(m)) + 1
192
+ const toolMsgs = msgs.slice(firstToolIdx)
193
+
194
+ const tcMap = {}
195
+ const msg = msgs.findLast((m) => AIMessage.isInstance(m) && m.tool_calls?.length)
196
+ if (msg) for (const tc of msg.tool_calls) tcMap[tc.id] = tc
197
+
198
+ for (const tm of toolMsgs) {
199
+ const tc = tcMap[tm.tool_call_id]
200
+ if (!tc) continue
201
+ const meta = resolveMeta(tc, toolMeta)
202
+ if (meta) {
203
+ publishToolCallArtifact(tc, {
204
+ meta,
205
+ status: tm.status === "error" ? "error" : "done",
206
+ lastChunk: true,
207
+ result: tm.content,
208
+ })
209
+ }
210
+ }
211
+
212
+ const plural = toolMsgs.length >= 2
101
213
  const key = plural ? "agent_status_processing_responses" : "agent_status_processing_response"
102
- const text = cds.i18n.messages.at(key)
103
- publishStatus(text)
214
+ publishStatus(cds.i18n.messages.at(key))
104
215
 
105
216
  return {}
106
217
  }
107
218
 
219
+ /** De-duplicate labels while preserving order. */
220
+ function uniqueLabels(calls) {
221
+ return [...new Set(calls.map(({ label }) => label))].join(", ")
222
+ }
223
+
108
224
  /**
109
- * afterModel hook: emit tool-call status updates (querying/calling).
225
+ * afterModel hook: emit tool-call status updates (querying/inspecting/calling)
226
+ * + tool-call start events.
110
227
  */
111
- export function afterModelHook(state) {
228
+ export function afterModelHook(state, toolMeta) {
112
229
  if (!cds.context?.["agent.eventBus"]) return {}
113
230
 
114
231
  const msgs = state.messages
@@ -118,36 +235,52 @@ export function afterModelHook(state) {
118
235
  const toolCalls = lastAI?.tool_calls
119
236
  if (!toolCalls?.length) return {}
120
237
 
121
- // Separate query calls from action calls
122
- const queryCalls = toolCalls.filter((tc) => tc.name === "query" || tc.name?.endsWith("_query"))
123
- const otherCalls = toolCalls.filter((tc) => tc.name !== "query" && !tc.name?.endsWith("_query"))
238
+ const resolved = toolCalls
239
+ .map((tc) => ({ tc, meta: resolveMeta(tc, toolMeta) }))
240
+ .filter(({ meta }) => meta)
241
+ .map(({ tc, meta }) => {
242
+ let kind = resolveToolKind(tc, meta)
243
+ let label = resolveToolLabel(tc, meta, kind)
244
+ return { tc, meta, kind, label }
245
+ })
246
+
247
+ const queryCalls = resolved.filter(({ kind }) => kind === "query")
248
+ const describeCalls = resolved.filter(({ kind }) => kind === "describe")
249
+ const otherCalls = resolved.filter(({ kind }) => kind !== "query" && kind !== "describe")
124
250
 
125
- // Emit "Querying {entities}" for query tools
126
251
  if (queryCalls.length) {
127
- const labels = queryCalls.map((tc) => resolveToolLabel(tc))
128
- const text = cds.i18n.messages.at("agent_status_querying", [labels.join(", ")])
129
- publishStatus(text)
252
+ publishStatus(cds.i18n.messages.at("agent_status_querying", [uniqueLabels(queryCalls)]))
253
+ }
254
+ if (describeCalls.length) {
255
+ publishStatus(cds.i18n.messages.at("agent_status_describing", [uniqueLabels(describeCalls)]))
130
256
  }
131
-
132
- // Emit "Calling {action labels}" for other tools
133
257
  if (otherCalls.length) {
134
- const labels = otherCalls.map((tc) => resolveToolLabel(tc))
135
- const text = cds.i18n.messages.at("agent_status_calling_tools", [labels.join(", ")])
136
- publishStatus(text)
258
+ publishStatus(cds.i18n.messages.at("agent_status_calling_tools", [uniqueLabels(otherCalls)]))
259
+ }
260
+
261
+ for (const { tc, meta } of resolved) {
262
+ publishToolCallArtifact(tc, { meta, status: "running", lastChunk: false })
137
263
  }
138
264
 
139
265
  return {}
140
266
  }
141
267
 
142
268
  /**
143
- * Middleware factory that emits non-final status-update events during agent execution:
144
- * - beforeModel: "Processing tool response" after tools finish
145
- * - afterModel: "Querying <entity>" / "Calling <action>" before tools are invoked
269
+ * Middleware emitting non-final status-update events during agent execution:
270
+ * - beforeModel: "Processing tool response" + tool-call completion events
271
+ * - afterModel: "Querying"/"Inspecting"/"Calling" + tool-call start events
272
+ *
273
+ * Static tools are captured here; remote MCP tools are resolved per-request from
274
+ * cds.context.__mcpDynamicTools by resolveMeta().
146
275
  */
147
- export async function statusUpdateMiddleware() {
276
+ export async function statusUpdateMiddleware(tools = []) {
277
+ // Presence in the map marks a tool as "known"; own-service tools carry no
278
+ // metadata (→ {}), so resolveToolKind sniffs their kind from the name.
279
+ const toolMeta = new Map(tools.map((t) => [t.name, t.metadata ?? {}]))
280
+
148
281
  return createMiddleware({
149
282
  name: "statusUpdateMiddleware",
150
- beforeModel: { hook: beforeModelHook },
151
- afterModel: { hook: afterModelHook },
283
+ beforeModel: { hook: (state) => beforeModelHook(state, toolMeta) },
284
+ afterModel: { hook: (state) => afterModelHook(state, toolMeta) },
152
285
  })
153
286
  }
@@ -14,9 +14,8 @@ export function toolWrapMiddleware(srv) {
14
14
  return createMiddleware({
15
15
  name: "ToolWrapMiddleware",
16
16
  wrapToolCall: async function (request, handler) {
17
- const { name, id, args } = request.toolCall
17
+ const { name, id } = request.toolCall
18
18
  try {
19
- LOG.debug(srv.name, "calling tool", name, args)
20
19
  const result = await handler(request)
21
20
  if (ToolMessage.isInstance(result) && result.artifact?.isError === true) {
22
21
  result.status = "error"
@@ -60,14 +60,17 @@ export async function summarizePartialWork({
60
60
  reason === "timeOut"
61
61
  ? `Agent task was not completed within its time limit. Write short, precise progress summary so user can decide whether to continue. State completed work and immediate next work. Do not claim work not shown. Start with: Agent did not finish within time! End with: Continue running or stop? No other questions to the user allowed!`
62
62
  : `Agent task was interrupted. Reason: ${reason}. Based on conversation history, provide brief summary of completed work and remaining work. Be concise.`
63
- summaryPrompt += `\n\n Conversation Snippet: \n\n ${conversationSnippet.trim()}`
63
+ summaryPrompt += `\n\n Conversation Snippet: \n\n`
64
64
 
65
65
  let summaryTimer
66
66
  try {
67
67
  const response = await Promise.race([
68
68
  (async () => {
69
+ const instruction = new HumanMessage(summaryPrompt)
70
+ instruction.name = "cap-js-agents-summarize"
69
71
  const model = await getModel()
70
- return model.invoke([new HumanMessage(summaryPrompt)])
72
+ // Two messages, so the first one can be logged as the prompt for summarization in MLflow
73
+ return model.invoke([instruction, new HumanMessage(conversationSnippet.trim())])
71
74
  })(),
72
75
  new Promise((_, reject) => {
73
76
  summaryTimer = setTimeout(() => reject(new Error("Summary LLM call timed out")), timeout)
@@ -2,26 +2,35 @@ import path from "node:path"
2
2
  import fs from "node:fs"
3
3
  import os from "node:os"
4
4
  import cds from "@sap/cds"
5
+ import { toml } from "../utils/toml.js"
5
6
 
6
7
  const HOME = os.homedir() || process.env.HOME || process.env.USERPROFILE
7
8
  const local = (file) => file.replace(HOME, "~")
8
9
  const LOG = cds.log("agents")
9
10
 
10
11
  /**
11
- * `cds.connect.to` compliant langchain model
12
- * for connecting to an Anthropic compatible API,
13
- * with autoconfiguration based on env, options,
14
- * ~/.claude/settings.json and ~/.config/opencode/opencode.json
12
+ * Autoconfiguration for anthropic -> openai -> mock
15
13
  */
16
14
  export function resolve_config(options) {
17
- let config = fromEnv()
15
+ let config = resolve_anthropic_config(options)
16
+ if (config?.credentials?.anthropicApiUrl || config?.credentials?.apiKey) return config
17
+ config = resolve_openai_config(options)
18
+ if (config?.credentials?.baseURL || config?.credentials?.apiKey) return config
19
+ return { kind: "mock" }
20
+ }
21
+
22
+ /**
23
+ * Anthropic autoconfiguration based on env, options,
24
+ * ~/.claude/settings.json and ~/.config/opencode/opencode.json
25
+ */
26
+ export function resolve_anthropic_config(options) {
27
+ let config = fromAnthropicEnv()
18
28
  if (!config?.anthropicApiUrl)
19
29
  config = {
20
30
  ...config,
21
31
  ...(fromClaude() || fromOpencode()),
22
32
  }
23
33
  let { model, ...credentials } = config
24
- if (!model && !options?.model) return { kind: "mock" }
25
34
  return {
26
35
  kind: "anthropic",
27
36
  model: options?.model || model,
@@ -32,7 +41,7 @@ export function resolve_config(options) {
32
41
  }
33
42
  }
34
43
 
35
- function fromEnv(env = process.env, silent) {
44
+ function fromAnthropicEnv(env = process.env, silent) {
36
45
  let any,
37
46
  config = {}
38
47
  if ((any = env.ANTHROPIC_BASE_URL)) config.anthropicApiUrl = any
@@ -51,7 +60,7 @@ function fromClaude() {
51
60
  let settings = JSON.parse(fs.readFileSync(settings_json, "utf8"))
52
61
  // https://www.schemastore.org/claude-code-settings.json
53
62
 
54
- let conf = (fromClaude.cached = fromEnv(settings?.env, "silent"))
63
+ let conf = (fromClaude.cached = fromAnthropicEnv(settings?.env, "silent"))
55
64
  if (!conf.model && settings.env) {
56
65
  let family = (settings.model || "sonnet").toUpperCase()
57
66
  conf.model = settings.env[`ANTHROPIC_DEFAULT_${family}_MODEL`] || settings?.model
@@ -88,6 +97,110 @@ function fromOpencode() {
88
97
  return fromOpencode.cached
89
98
  }
90
99
 
100
+ /**
101
+ * OpenAI autoconfiguration based on env, options,
102
+ * ~/.codex/config.toml and ~/.config/opencode/opencode.json
103
+ */
104
+ export function resolve_openai_config(options = {}) {
105
+ let config = fromOpenAIEnv()
106
+ if (!config?.baseURL)
107
+ config = {
108
+ ...config,
109
+ ...fromCodex(),
110
+ ...fromOpencodeOpenAI(),
111
+ }
112
+ let { model, ...credentials } = config
113
+ return {
114
+ kind: "openai",
115
+ model: options?.model || model,
116
+ credentials: {
117
+ ...credentials,
118
+ ...options?.credentials,
119
+ },
120
+ }
121
+ }
122
+
123
+ function fromOpenAIEnv(env = process.env) {
124
+ let any,
125
+ config = {}
126
+ if ((any = env.OPENAI_BASE_URL)) config.baseURL = any
127
+ if ((any = env.OPENAI_API_KEY)) config.apiKey = any
128
+ if ((any = env.OPENAI_MODEL)) config.model = any
129
+ if (!Object.keys(config).length) return null
130
+ LOG.debug(`Loaded OpenAI settings from env:`, config)
131
+ return config
132
+ }
133
+
134
+ function fromCodex() {
135
+ if ("cached" in fromCodex) return fromCodex.cached
136
+ const codex_toml = path.join(HOME, ".codex/config.toml")
137
+ try {
138
+ let conf = toml.parse(fs.readFileSync(codex_toml, "utf8"))
139
+ LOG.debug(`Loaded Codex settings from`, local(codex_toml))
140
+ const provider = conf.model_provider ?? "openai"
141
+ const p = conf.model_providers?.[provider] ?? {}
142
+ let apiKey = (p.env_key && process.env[p.env_key]) || p.experimental_bearer_token
143
+ if (!apiKey && p.requires_openai_auth) apiKey = fromCodexAuth()
144
+ let any,
145
+ config = {}
146
+ if ((any = p.base_url)) config.baseURL = any
147
+ if (
148
+ (any =
149
+ (p.env_key && process.env[p.env_key]) ??
150
+ p.experimental_bearer_token ??
151
+ (p.requires_openai_auth && fromCodexAuth()))
152
+ )
153
+ config.apiKey = any
154
+ if ((any = conf.model)) config.model = any
155
+ fromCodex.cached = Object.keys(config).length ? config : null
156
+ LOG.debug(`Loaded config from`, local(codex_toml), ":", sanitized(config))
157
+ } catch {
158
+ LOG.debug(`Failed loading Codex settings from`, local(codex_toml))
159
+ fromCodex.cached = null
160
+ }
161
+ return fromCodex.cached
162
+ }
163
+
164
+ function fromCodexAuth() {
165
+ const auth_json = path.join(HOME, ".codex/auth.json")
166
+ try {
167
+ let auth = JSON.parse(fs.readFileSync(auth_json, "utf8"))
168
+ // auth.json stores the key under OPENAI_API_KEY (or a custom env var name)
169
+ return auth?.OPENAI_API_KEY ?? null
170
+ } catch {
171
+ LOG.debug(`Failed loading Codex auth from`, local(auth_json))
172
+ return null
173
+ }
174
+ }
175
+
176
+ function fromOpencodeOpenAI() {
177
+ if ("cached" in fromOpencodeOpenAI) return fromOpencodeOpenAI.cached
178
+ const opencode_json = path.join(HOME, ".config/opencode/opencode.json")
179
+ try {
180
+ let conf = JSON.parse(fs.readFileSync(opencode_json, "utf8"))
181
+ LOG.debug(`Loaded OpenCode settings from`, local(opencode_json))
182
+ // https://opencode.ai/config.json
183
+ let o = conf?.provider?.openai?.options
184
+ if (!o) return (fromOpencodeOpenAI.cached = null)
185
+ let any,
186
+ config = {}
187
+ if ((any = o.baseURL ?? o.baseURL)) config.baseURL = any
188
+ if ((any = o.apiKey)) config.apiKey = any
189
+ if ((any = conf?.model)) config.model = any.replace("openai/", "")
190
+ fromOpencodeOpenAI.cached = Object.keys(config).length ? config : null
191
+ LOG.debug(
192
+ `Loaded config from`,
193
+ local(opencode_json),
194
+ ":",
195
+ sanitized(fromOpencodeOpenAI.cached || {}),
196
+ )
197
+ } catch {
198
+ LOG.debug(`Failed loading OpenCode settings from`, local(opencode_json))
199
+ fromOpencodeOpenAI.cached = null
200
+ }
201
+ return fromOpencodeOpenAI.cached
202
+ }
203
+
91
204
  const sanitized = ({ apiKey, ...rest }) => ({
92
205
  ...rest,
93
206
  apiKey: apiKey ? "***" : undefined,
@@ -5,6 +5,7 @@ import {
5
5
  createEvalRun,
6
6
  closeEvalRun,
7
7
  logMlflowMetrics,
8
+ logMlflowRunMetadata,
8
9
  } from "../telemetry/mlflow/evaluation.js"
9
10
  import { flushMlflowTraces } from "../telemetry/mlflow/tracing.js"
10
11
 
@@ -23,6 +24,9 @@ export function evalRun(opts = {}) {
23
24
  runId,
24
25
  mlflowRunId,
25
26
  validationsByTask: new Map(),
27
+ metricKeys: new Set(),
28
+ mlflowMetadataLogged: false,
29
+ prompts: [],
26
30
  }
27
31
  }
28
32
 
@@ -34,15 +38,22 @@ export function evalRun(opts = {}) {
34
38
  })
35
39
 
36
40
  if (typeof afterEach === "function") {
37
- afterEach(async () => {
38
- if (state) await _flushValidations(state)
41
+ afterEach(async (testState) => {
42
+ if (state) {
43
+ await _flushValidations(state)
44
+ // Report test failure/success, so aggregated output_correctness respects static asserts
45
+ _addValidation(
46
+ { _evalState: state, taskId: "code_asserts" },
47
+ testState.task.result.state === "pass",
48
+ )
49
+ }
39
50
  })
40
51
  }
41
52
 
42
53
  afterAll(async () => {
43
54
  if (state) await _flushValidations(state)
44
55
  await flushMlflowTraces()
45
- await closeEvalRun(state?.mlflowRunId).catch(() => {})
56
+ await closeEvalRun(state).catch(() => {})
46
57
  if (cds._activeEvalRun === state) cds._activeEvalRun = null
47
58
  state = null
48
59
  })
@@ -148,5 +159,40 @@ async function _postAssessmentScore(result, score, comment, config) {
148
159
  export async function logMlflowMetricsForResult(result, state = null) {
149
160
  state = state ?? cds._activeEvalRun
150
161
  if (!state?.mlflowRunId) return
151
- await logMlflowMetrics(state.mlflowRunId, result.metrics).catch(() => {})
162
+ state.prompts = state.prompts.concat(_extractPrompts(result?.spans))
163
+ await _logMlflowRunMetadataOnce(result, state)
164
+ await logMlflowMetrics(state, result.metrics).catch(() => {})
165
+ }
166
+
167
+ // REVISIT: Properly log models as Registered Models in an eval Run to cover the change that an eval run contains multiple models
168
+ async function _logMlflowRunMetadataOnce(result, state) {
169
+ if (state.mlflowMetadataLogged) return
170
+ const metadata = _extractMlflowRunMetadata(result?.spans)
171
+ if (!metadata) return
172
+ state.mlflowMetadataLogged = true
173
+ await logMlflowRunMetadata(state.mlflowRunId, metadata).catch(() => {})
174
+ }
175
+
176
+ function _extractMlflowRunMetadata(spans) {
177
+ const attrs = spans?.find(
178
+ (span) => span.attributes?.["gen_ai.operation.name"] === "chat",
179
+ )?.attributes
180
+ if (!attrs) return null
181
+
182
+ const model = attrs["gen_ai.response.model"]
183
+ const provider = attrs["gen_ai.provider.name"]
184
+ const params =
185
+ attrs["gen_ai.request.model_params"] && JSON.parse(attrs["gen_ai.request.model_params"])
186
+
187
+ if (!model && !provider && !Object.keys(params).length) return null
188
+ return { model, provider, params }
189
+ }
190
+
191
+ function _extractPrompts(spans) {
192
+ const attrs = spans?.find(
193
+ (span) => span.attributes?.["gen_ai.operation.name"] === "invoke_agent",
194
+ )?.attributes
195
+ if (!attrs) return []
196
+ const prompts = JSON.parse(attrs["mlflow.traceTag.mlflow.linkedPrompts"]) ?? []
197
+ return prompts
152
198
  }