@viveka10/cap-ai-token-optimizer 0.1.1 → 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.cdsrc.js +27 -0
- package/CHANGELOG.md +32 -1
- package/README.md +16 -7
- package/cds-plugin.js +5 -19
- package/docs/existing-cap-app-on-btp.md +4 -3
- package/docs/integration-guide.md +2 -1
- package/lib/metrics/classifier.js +13 -1
- package/lib/models.js +24 -0
- package/lib/optimizer.js +50 -2
- package/package.json +2 -1
- package/srv/admin-service.js +22 -3
package/.cdsrc.js
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
'use strict'
|
|
2
|
+
/**
|
|
3
|
+
* Plugin configuration defaults, read by cds.env. cds.env is also rebuilt from these files
|
|
4
|
+
* after plugins are loaded (`cds build`, `cds deploy`), so anything cds-plugin.js wrote into
|
|
5
|
+
* cds.env would be lost there.
|
|
6
|
+
*
|
|
7
|
+
* `model` is therefore a getter: CAP's config merge keeps getters, and CAP reads `model`
|
|
8
|
+
* when it resolves the app's models, after the app's own configuration and profiles have been
|
|
9
|
+
* merged. Models the app sets explicitly are kept.
|
|
10
|
+
*/
|
|
11
|
+
const { modelsFor } = require('./lib/models')
|
|
12
|
+
|
|
13
|
+
module.exports = {
|
|
14
|
+
requires: {
|
|
15
|
+
tokenOptimizer: {
|
|
16
|
+
kind: 'token-optimizer',
|
|
17
|
+
get model () {
|
|
18
|
+
const own = [].concat(this._model ?? [])
|
|
19
|
+
const models = [...new Set([...own, ...modelsFor(this)])]
|
|
20
|
+
return models.length ? models : undefined
|
|
21
|
+
},
|
|
22
|
+
set model (value) {
|
|
23
|
+
Object.defineProperty(this, '_model', { value, writable: true, configurable: true, enumerable: false })
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
}
|
|
27
|
+
}
|
package/CHANGELOG.md
CHANGED
|
@@ -6,6 +6,36 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.1.2] - 2026-10-11
|
|
10
|
+
|
|
11
|
+
### Fixed
|
|
12
|
+
|
|
13
|
+
- **Token usage of streaming calls.** `stream()` calls (including LangChain models with
|
|
14
|
+
streaming, as used by `@cap-js/agents`) were recorded when the stream opened, without token
|
|
15
|
+
usage. They are now recorded when the stream ends, with AI Core's usage, finish reason, request
|
|
16
|
+
id and cost.
|
|
17
|
+
- **Errors inside a stream** were not recorded, and the call counted as a success. They are now
|
|
18
|
+
recorded as errors and classified from the error event AI Core sends (for example `SERVER`,
|
|
19
|
+
`RATE_LIMIT`, `CONTENT_FILTER`).
|
|
20
|
+
- **`cds build` and `cds deploy` left out the plugin's models.** The `TokenUsageLog` entity and
|
|
21
|
+
the admin service were added to `cds.env` when the plugin loaded, but `cds build` rebuilds
|
|
22
|
+
`cds.env` afterwards, so the MTA had no usage-log table and no admin service. The models now
|
|
23
|
+
come from the plugin's `.cdsrc.js` and are computed from the app's final configuration
|
|
24
|
+
(profiles and environment variables included). Apps that worked around this by listing the
|
|
25
|
+
models in `cds.requires.tokenOptimizer.model` keep working and can remove the list.
|
|
26
|
+
- **Admin service for Fiori elements:** the in-memory entities (`UseCases`, `Models`,
|
|
27
|
+
`ErrorCategories`, `StepSavings`, `PromptPrefixChanges`, `Exporters`) answered `$count=true`
|
|
28
|
+
with 0, so list reports showed empty tables, and `UseCases` and `Models` returned `tokens` as a
|
|
29
|
+
nested object instead of the `tokens_*` columns of their OData metadata.
|
|
30
|
+
|
|
31
|
+
### Changed
|
|
32
|
+
|
|
33
|
+
- `latencyMs` of streaming calls now covers the whole answer instead of the time until the
|
|
34
|
+
stream opened. A stream closed early by the caller, or not read within 10 minutes, is recorded
|
|
35
|
+
at that point without usage and with a note in `savings.notes`.
|
|
36
|
+
- The mock AI Core server answers streaming requests as server-sent events and has a
|
|
37
|
+
`stream_error` scenario.
|
|
38
|
+
|
|
9
39
|
## [0.1.1] - 2026-10-10
|
|
10
40
|
|
|
11
41
|
### Added
|
|
@@ -78,6 +108,7 @@ First public release.
|
|
|
78
108
|
- The `http` exporter supports Internet destinations only.
|
|
79
109
|
- Not yet verified against SAP Cloud ALM; live AI Core tests are opt-in.
|
|
80
110
|
|
|
81
|
-
[Unreleased]: https://github.com/viveka10/npm_token_optimizer/compare/v0.1.
|
|
111
|
+
[Unreleased]: https://github.com/viveka10/npm_token_optimizer/compare/v0.1.2...HEAD
|
|
112
|
+
[0.1.2]: https://github.com/viveka10/npm_token_optimizer/compare/v0.1.1...v0.1.2
|
|
82
113
|
[0.1.1]: https://github.com/viveka10/npm_token_optimizer/compare/v0.1.0...v0.1.1
|
|
83
114
|
[0.1.0]: https://github.com/viveka10/npm_token_optimizer/releases/tag/v0.1.0
|
package/README.md
CHANGED
|
@@ -104,7 +104,10 @@ Wrap clients once (for example in your service's `init()`) and reuse them.
|
|
|
104
104
|
|
|
105
105
|
- **Supported clients:** `OrchestrationClient` (`chatCompletion`, `stream`) and
|
|
106
106
|
`AzureOpenAiChatClient` (`run`, `stream`). Other methods are forwarded untouched.
|
|
107
|
-
- **Streaming:** `stream()` calls are passed through unoptimized
|
|
107
|
+
- **Streaming:** `stream()` calls are passed through unoptimized. They are recorded when the
|
|
108
|
+
stream has been read to the end, with AI Core's token usage, the finish reason and the latency
|
|
109
|
+
of the whole answer. A failure inside the stream is recorded as an error. A stream closed early,
|
|
110
|
+
or not read within 10 minutes, is recorded at that point without usage and with a note.
|
|
108
111
|
- **Errors:** SDK errors are rethrown unchanged. Use `classifyError(err)` to get the failure category.
|
|
109
112
|
|
|
110
113
|
```js
|
|
@@ -141,7 +144,7 @@ This hooks the `@sap-ai-sdk` client classes `OrchestrationClient` (`chatCompleti
|
|
|
141
144
|
In short:
|
|
142
145
|
- only these two classes from SDK 2.x, and only the SDK copy resolved from the app root;
|
|
143
146
|
- process-wide, including libraries;
|
|
144
|
-
- streaming is measured but not optimized.
|
|
147
|
+
- streaming is measured (with token usage) but not optimized.
|
|
145
148
|
|
|
146
149
|
### withUseCase
|
|
147
150
|
|
|
@@ -433,9 +436,10 @@ optimizer.configure({
|
|
|
433
436
|
})
|
|
434
437
|
```
|
|
435
438
|
|
|
436
|
-
The `db` exporter and the admin service also need their CDS models
|
|
437
|
-
|
|
438
|
-
|
|
439
|
+
The `db` exporter and the admin service also need their CDS models. The plugin adds them to
|
|
440
|
+
the app's model (also in `cds build` and `cds deploy`) when they are enabled in the
|
|
441
|
+
configuration, including profiles and environment variables. Enable them in the
|
|
442
|
+
configuration, not only in code.
|
|
439
443
|
|
|
440
444
|
`createOptimizer(config)` creates a separate instance with its own stats and exporters.
|
|
441
445
|
|
|
@@ -462,7 +466,8 @@ One event per call:
|
|
|
462
466
|
|
|
463
467
|
- Prompt and completion text are never recorded by default; only a SHA-256 hash and the length.
|
|
464
468
|
`user` is a salted hash of the user id.
|
|
465
|
-
- `tokens.prompt`/`completion`/`total`/`cached` come from the AI Core response
|
|
469
|
+
- `tokens.prompt`/`completion`/`total`/`cached` come from the AI Core response, for streams
|
|
470
|
+
from the last chunk. Orchestration streams don't report cached tokens (`cached` is 0).
|
|
466
471
|
`promptEstimatedBefore`/`After` are estimates before and after optimization (js-tiktoken for
|
|
467
472
|
OpenAI model families, characters ÷ 4 for other models).
|
|
468
473
|
- `savings.estimatedTokens` is estimated prompt tokens saved plus, for cache hits, the cached
|
|
@@ -554,7 +559,8 @@ Writes one row per call to the entity `cap.ai.token_optimizer.TokenUsageLog`
|
|
|
554
559
|
(table `CAP_AI_TOKEN_OPTIMIZER_TOKENUSAGELOG`) through CAP's database service.
|
|
555
560
|
|
|
556
561
|
- **Databases:** SAP HANA, SQLite or PostgreSQL. `hana` is an alias for `db`.
|
|
557
|
-
- **Model:** the plugin adds the entity to your model when the exporter is configured
|
|
562
|
+
- **Model:** the plugin adds the entity to your model when the exporter is configured, also in
|
|
563
|
+
`cds build`, so it is part of the MTA's db deployer.
|
|
558
564
|
Deploy as usual (`cds deploy`, or the MTA db deployer), and run your MTX upgrade for multitenant apps.
|
|
559
565
|
- **Multitenancy:** rows are inserted in batches on the flush interval with a privileged
|
|
560
566
|
user. Each tenant's rows go to its own container.
|
|
@@ -594,6 +600,9 @@ for users with the role `admin.role` (default `TokenOptimizerAdmin`):
|
|
|
594
600
|
| `UsageLog` | the `TokenUsageLog` rows, if the `db` exporter is enabled (supports `$filter`, `$orderby`, `$top`, …) |
|
|
595
601
|
|
|
596
602
|
All entries except `UsageLog` are in-memory stats of the app instance that answers, since it started.
|
|
603
|
+
They support key lookups and `$count`, which is enough for a Fiori elements list report, but not
|
|
604
|
+
`$filter`, `$orderby`, `$search` or paging. With flat OData (the default), `tokens` is exposed as
|
|
605
|
+
`tokens_prompt`, `tokens_completion`, `tokens_total` and `tokens_cached`.
|
|
597
606
|
|
|
598
607
|
To grant access, add the role to `xs-security.json` and a role collection:
|
|
599
608
|
|
package/cds-plugin.js
CHANGED
|
@@ -6,35 +6,21 @@
|
|
|
6
6
|
const cds = globalThis.cds ?? require('@sap/cds')
|
|
7
7
|
const { optimizer, countText } = require('./index')
|
|
8
8
|
const log = require('./lib/util/log')
|
|
9
|
-
const path = require('node:path')
|
|
10
9
|
|
|
11
10
|
try {
|
|
12
11
|
optimizer.configure(cds.env.requires?.tokenOptimizer)
|
|
13
12
|
log.debug('configured, mode =', optimizer.config.mode, 'exporters =', optimizer.config.metrics.exporters)
|
|
14
|
-
|
|
13
|
+
setupAdminRole(optimizer.config)
|
|
15
14
|
} catch (e) {
|
|
16
15
|
log.error('could not configure token optimizer:', e?.message)
|
|
17
16
|
}
|
|
18
17
|
|
|
19
18
|
/**
|
|
20
|
-
*
|
|
21
|
-
*
|
|
22
|
-
* UsageLog projection when both are on. Plugins load before the model, so adding a `model`
|
|
23
|
-
* to our cds.requires entry is enough.
|
|
19
|
+
* The CDS models (usage log entity, admin service) are added through `model` in .cdsrc.js, so
|
|
20
|
+
* they are also part of `cds build` and `cds deploy`. Here we only set the admin service's role.
|
|
24
21
|
*/
|
|
25
|
-
function
|
|
26
|
-
|
|
27
|
-
const dbLog = exporters.includes('db') || exporters.includes('hana')
|
|
28
|
-
const admin = cfg.admin?.enabled
|
|
29
|
-
const models = []
|
|
30
|
-
if (dbLog) models.push('db/token-usage-log')
|
|
31
|
-
if (admin) models.push('srv/admin-service')
|
|
32
|
-
if (admin && dbLog) models.push('srv/admin-usage-log')
|
|
33
|
-
if (!models.length) return
|
|
34
|
-
|
|
35
|
-
const req = cds.env.requires.tokenOptimizer ??= {}
|
|
36
|
-
req.model = [...new Set([].concat(req.model ?? [], models.map(m => path.join(__dirname, m))))]
|
|
37
|
-
if (!admin) return
|
|
22
|
+
function setupAdminRole (cfg) {
|
|
23
|
+
if (!cfg.admin?.enabled) return
|
|
38
24
|
const role = cfg.admin.role || 'TokenOptimizerAdmin'
|
|
39
25
|
cds.on('loaded', (csn) => {
|
|
40
26
|
const srv = csn?.definitions?.TokenOptimizerService
|
|
@@ -179,9 +179,10 @@ the `db` exporter:
|
|
|
179
179
|
"[production]": { "metrics": { "exporters": ["console", "otel", "db"] }, "admin": { "enabled": true } }
|
|
180
180
|
```
|
|
181
181
|
|
|
182
|
-
- The plugin adds the entity `cap.ai.token_optimizer.TokenUsageLog` to your model
|
|
183
|
-
deployed with your normal HANA deployment (the `db-deployer`
|
|
184
|
-
no extra step.
|
|
182
|
+
- The plugin adds the entity `cap.ai.token_optimizer.TokenUsageLog` to your model, also in
|
|
183
|
+
`cds build --production`. It is deployed with your normal HANA deployment (the `db-deployer`
|
|
184
|
+
module in the MTA), so there's no extra step. With version 0.1.1 the entity was missing from
|
|
185
|
+
`cds build`; there, list the models in `cds.requires.tokenOptimizer.model` as a workaround.
|
|
185
186
|
- **Multitenant apps (MTX):** each tenant's rows go to that tenant's container. Run your
|
|
186
187
|
usual tenant upgrade after deploying, so existing tenants get the new table.
|
|
187
188
|
- Rows older than 30 days are deleted (`metrics.db.retentionDays`). No prompt or answer text
|
|
@@ -130,7 +130,8 @@ for the same requests.
|
|
|
130
130
|
- **Scope:** the hooks apply to every caller in the process, including libraries. If you want
|
|
131
131
|
to choose exactly which calls are optimized, leave `autoInstrument` off and use `wrap()`,
|
|
132
132
|
`withUseCase()` and `wrapLangChain()` instead.
|
|
133
|
-
- **Streaming:** `stream()` calls are measured
|
|
133
|
+
- **Streaming:** `stream()` calls are measured, with token usage once the stream has been read
|
|
134
|
+
to the end, but not optimized.
|
|
134
135
|
- **Use cases:** without `withUseCase()`, all hooked calls share the use case set in
|
|
135
136
|
`autoInstrument.useCase` (default `default`), so per-use-case settings apply to all of them.
|
|
136
137
|
- **Start-up:** the hooks are installed when the CAP plugin loads. Outside CAP, call
|
|
@@ -7,6 +7,10 @@
|
|
|
7
7
|
* (orchestration: `{ error: { code, message, location } }`, Azure OpenAI: `{ error: { code, message, innererror } }`).
|
|
8
8
|
* Deployment-resolution and OAuth failures wrap further levels of `cause`. We therefore walk the
|
|
9
9
|
* whole cause chain (and AggregateError.errors) and collect every signal before deciding.
|
|
10
|
+
*
|
|
11
|
+
* Errors inside a stream are thrown by the SDK's SSE parser as
|
|
12
|
+
* `Error("Error received from the server.\n<JSON error>")`, wrapped in
|
|
13
|
+
* `ErrorWithCause("Error while iterating over SSE stream.")`; the JSON is read like a response body.
|
|
10
14
|
*/
|
|
11
15
|
|
|
12
16
|
const CATEGORIES = Object.freeze([
|
|
@@ -24,6 +28,7 @@ const RE_TIMEOUT = /timeout of \d+ms exceeded|timed? ?out|deadline exceeded/i
|
|
|
24
28
|
const RE_DEPLOYMENT = /no deployment matched|deployment not found|DeploymentNotFound|has no deployment url/i
|
|
25
29
|
const RE_AUTH = /unauthori[sz]ed|forbidden|invalid[_ ]client|could not fetch client credentials token/i
|
|
26
30
|
const RE_NETWORK = /socket hang up|network error|fetch failed|getaddrinfo|connect ECONN/i
|
|
31
|
+
const RE_SSE_ERROR = /^Error received from the server\.\n(\{[\s\S]*\})$/
|
|
27
32
|
|
|
28
33
|
const TIMEOUT_CODES = new Set(['ECONNABORTED', 'ETIMEDOUT', 'ESOCKETTIMEDOUT', 'ERR_TIMEOUT', 'UND_ERR_CONNECT_TIMEOUT', 'UND_ERR_HEADERS_TIMEOUT', 'UND_ERR_BODY_TIMEOUT'])
|
|
29
34
|
const NETWORK_CODES = new Set(['ECONNREFUSED', 'ECONNRESET', 'ENOTFOUND', 'EAI_AGAIN', 'EPIPE', 'EHOSTUNREACH', 'ENETUNREACH', 'ENETDOWN', 'ERR_NETWORK', 'UND_ERR_SOCKET'])
|
|
@@ -97,7 +102,7 @@ function collectSignals (root) {
|
|
|
97
102
|
if (typeof e.message === 'string' && e.message) s.texts.push(e.message)
|
|
98
103
|
s.retryAfter ??= header(e.response?.headers, 'retry-after') ?? header(e.headers, 'retry-after')
|
|
99
104
|
|
|
100
|
-
const body = parseBody(e.response?.data ?? e.body)
|
|
105
|
+
const body = parseBody(e.response?.data ?? e.body ?? sseErrorBody(e.message))
|
|
101
106
|
const errors = providerErrors(body)
|
|
102
107
|
// a non-JSON body (e.g. a gateway's "Payload Too Large") is the most useful message we have
|
|
103
108
|
if (!errors.length && typeof body?.message === 'string' && body.message) {
|
|
@@ -132,6 +137,13 @@ function providerErrors (body) {
|
|
|
132
137
|
return []
|
|
133
138
|
}
|
|
134
139
|
|
|
140
|
+
/** The `{ error }` body of an error event in a stream, from the SDK's error message. */
|
|
141
|
+
function sseErrorBody (message) {
|
|
142
|
+
const json = typeof message === 'string' ? RE_SSE_ERROR.exec(message)?.[1] : null
|
|
143
|
+
if (!json) return undefined
|
|
144
|
+
try { return { error: JSON.parse(json) } } catch { return undefined }
|
|
145
|
+
}
|
|
146
|
+
|
|
135
147
|
function parseBody (data) {
|
|
136
148
|
if (typeof data === 'string') {
|
|
137
149
|
try { return JSON.parse(data) } catch { return { message: data } }
|
package/lib/models.js
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
'use strict'
|
|
2
|
+
/**
|
|
3
|
+
* The CDS models this plugin adds to the app, depending on its configuration: the
|
|
4
|
+
* TokenUsageLog entity for the db exporter, the read-only admin service, and the UsageLog
|
|
5
|
+
* projection when both are on.
|
|
6
|
+
*/
|
|
7
|
+
const { name } = require('../package.json')
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* @param {object} [cfg] cds.requires.tokenOptimizer (raw, as in cds.env)
|
|
11
|
+
* @returns {string[]} model ids, resolved from the app root like `using from '<id>'`
|
|
12
|
+
*/
|
|
13
|
+
function modelsFor (cfg) {
|
|
14
|
+
const exporters = [].concat(cfg?.metrics?.exporters ?? [])
|
|
15
|
+
const dbLog = exporters.includes('db') || exporters.includes('hana')
|
|
16
|
+
const admin = cfg?.admin?.enabled === true
|
|
17
|
+
const models = []
|
|
18
|
+
if (dbLog) models.push(`${name}/db/token-usage-log`)
|
|
19
|
+
if (admin) models.push(`${name}/srv/admin-service`)
|
|
20
|
+
if (admin && dbLog) models.push(`${name}/srv/admin-usage-log`)
|
|
21
|
+
return models
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
module.exports = { modelsFor }
|
package/lib/optimizer.js
CHANGED
|
@@ -10,6 +10,10 @@
|
|
|
10
10
|
*
|
|
11
11
|
* Rule: nothing the optimizer does may break the AI call. If the pipeline fails, the original
|
|
12
12
|
* request is sent. Only errors thrown by the SDK itself reach the caller, unchanged.
|
|
13
|
+
*
|
|
14
|
+
* Streaming calls are recorded when the caller has read the stream to the end, so the event
|
|
15
|
+
* has AI Core's token usage, finish reason and the full latency. A stream that fails, is
|
|
16
|
+
* closed early or is not read within STREAM_RECORD_TIMEOUT_MS is recorded at that point.
|
|
13
17
|
*/
|
|
14
18
|
const { resolveConfig, forUseCase } = require('./config')
|
|
15
19
|
const { MetricsCollector } = require('./metrics/collector')
|
|
@@ -30,6 +34,7 @@ const WRAPPED = Symbol.for('cap-ai-token-optimizer.wrapped')
|
|
|
30
34
|
const LANGCHAIN_WRAPPED = Symbol.for('cap-ai-token-optimizer.langchain')
|
|
31
35
|
const MAX_PREFIXES = 1000
|
|
32
36
|
const ADVISORY_LOG_INTERVAL_MS = 10 * 60 * 1000
|
|
37
|
+
const STREAM_RECORD_TIMEOUT_MS = 10 * 60 * 1000
|
|
33
38
|
|
|
34
39
|
class TokenOptimizer {
|
|
35
40
|
/**
|
|
@@ -40,6 +45,7 @@ class TokenOptimizer {
|
|
|
40
45
|
this.collector = new MetricsCollector()
|
|
41
46
|
this._deps = deps
|
|
42
47
|
this._sleep = deps.sleep ?? ((ms) => new Promise(resolve => setTimeout(resolve, ms)))
|
|
48
|
+
this._streamRecordTimeoutMs = deps.streamRecordTimeoutMs ?? STREAM_RECORD_TIMEOUT_MS
|
|
43
49
|
this._configured = false
|
|
44
50
|
this.cache = null
|
|
45
51
|
// last prompt prefix (system + tools) per use case and model, for the prompt-caching advisory
|
|
@@ -253,10 +259,11 @@ class TokenOptimizer {
|
|
|
253
259
|
// the original method: a hooked one would route the call through an optimizer again
|
|
254
260
|
const response = await instrument.originalOf(plan.client[methodName]).apply(plan.client, callArgs)
|
|
255
261
|
if (ctx.cacheKey) this._cacheSet(ctx, response)
|
|
262
|
+
if (method.operation === 'stream' && safe(() => this._trackStream(ctx, { started, retries, response }), logRecordError)) return response
|
|
256
263
|
safe(() => this._record(ctx, { started, retries, response }), logRecordError)
|
|
257
264
|
return response
|
|
258
265
|
} catch (err) {
|
|
259
|
-
const classified = safe(() => classifyError(err)) ??
|
|
266
|
+
const classified = safe(() => classifyError(err)) ?? unknownError(err)
|
|
260
267
|
const delay = this._retryDelay(cfg, classified, retries)
|
|
261
268
|
if (delay != null) {
|
|
262
269
|
retries++
|
|
@@ -270,6 +277,44 @@ class TokenOptimizer {
|
|
|
270
277
|
}
|
|
271
278
|
}
|
|
272
279
|
|
|
280
|
+
/**
|
|
281
|
+
* Record a streaming call when its stream ends: replaces `response.stream` with a stream of
|
|
282
|
+
* the same class that passes every chunk through unchanged and records the event once, when
|
|
283
|
+
* the stream is read to the end (with AI Core's usage), fails (the error is classified and
|
|
284
|
+
* rethrown), is closed early by the caller, or is not read within the record timeout.
|
|
285
|
+
* @returns {boolean} false if the stream can't be tracked; the caller then records right away
|
|
286
|
+
*/
|
|
287
|
+
_trackStream (ctx, { started, retries, response }) {
|
|
288
|
+
const original = response?.stream
|
|
289
|
+
const Stream = original?.constructor
|
|
290
|
+
if (typeof original?.[Symbol.asyncIterator] !== 'function' || typeof Stream !== 'function' || Stream === Object) return false
|
|
291
|
+
|
|
292
|
+
let recorded = false
|
|
293
|
+
const record = (outcome = {}) => {
|
|
294
|
+
if (recorded) return
|
|
295
|
+
recorded = true
|
|
296
|
+
clearTimeout(timer)
|
|
297
|
+
safe(() => this._record(ctx, { started, retries, response, ...outcome }), logRecordError)
|
|
298
|
+
}
|
|
299
|
+
const timer = setTimeout(() => record({ note: 'stream: not read to the end within the record timeout' }), this._streamRecordTimeoutMs)
|
|
300
|
+
timer.unref?.()
|
|
301
|
+
|
|
302
|
+
async function * tracked () {
|
|
303
|
+
let ended = false
|
|
304
|
+
try {
|
|
305
|
+
for await (const chunk of original) yield chunk
|
|
306
|
+
ended = true
|
|
307
|
+
} catch (err) {
|
|
308
|
+
record({ error: safe(() => classifyError(err)) ?? unknownError(err) })
|
|
309
|
+
throw err
|
|
310
|
+
} finally {
|
|
311
|
+
record(ended ? {} : { note: 'stream: closed by the caller before the end' })
|
|
312
|
+
}
|
|
313
|
+
}
|
|
314
|
+
response.stream = new Stream(() => tracked(), original.controller)
|
|
315
|
+
return true
|
|
316
|
+
}
|
|
317
|
+
|
|
273
318
|
/**
|
|
274
319
|
* Everything known before the call: call info, estimates, pipeline result and cache key.
|
|
275
320
|
* Never throws: a failure leaves the call unoptimized.
|
|
@@ -375,7 +420,7 @@ class TokenOptimizer {
|
|
|
375
420
|
return Math.min(r.maxDelayMs, Math.round(backoff / 2 + Math.random() * backoff / 2))
|
|
376
421
|
}
|
|
377
422
|
|
|
378
|
-
_record (ctx, { started, retries, response, error, cacheEntry }) {
|
|
423
|
+
_record (ctx, { started, retries, response, error, cacheEntry, note }) {
|
|
379
424
|
const latencyMs = Math.round(performance.now() - started)
|
|
380
425
|
const cfg = ctx.cfg
|
|
381
426
|
const info = ctx.info ?? {}
|
|
@@ -398,6 +443,7 @@ class TokenOptimizer {
|
|
|
398
443
|
const model = info.model ?? null
|
|
399
444
|
const notes = [...(pipeline?.notes ?? [])]
|
|
400
445
|
if (ctx.pipelineFailed) notes.push('pipeline failed; original request sent')
|
|
446
|
+
if (note) notes.push(note)
|
|
401
447
|
const advisory = cacheEntry ? null : safe(() => this._checkPrefix(ctx))
|
|
402
448
|
|
|
403
449
|
const event = buildEvent({
|
|
@@ -453,6 +499,8 @@ function prefixOf (messages, tools) {
|
|
|
453
499
|
|
|
454
500
|
const logRecordError = (e) => log.warn('could not record call metrics:', e?.message)
|
|
455
501
|
|
|
502
|
+
const unknownError = (err) => ({ category: 'UNKNOWN', httpStatus: null, code: null, message: String(err?.message ?? err), retryAfterMs: null })
|
|
503
|
+
|
|
456
504
|
/**
|
|
457
505
|
* Run `fn`, returning undefined instead of throwing.
|
|
458
506
|
* @template T
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@viveka10/cap-ai-token-optimizer",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.2",
|
|
4
4
|
"description": "SAP CAP plugin to reduce, measure and report SAP AI Core LLM token usage: prompt optimization, response cache, cost tracking, and metrics for SAP Cloud ALM / OpenTelemetry",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"sap",
|
|
@@ -60,6 +60,7 @@
|
|
|
60
60
|
"index.js",
|
|
61
61
|
"index.d.ts",
|
|
62
62
|
"cds-plugin.js",
|
|
63
|
+
".cdsrc.js",
|
|
63
64
|
"lib/",
|
|
64
65
|
"srv/",
|
|
65
66
|
"db/",
|
package/srv/admin-service.js
CHANGED
|
@@ -13,11 +13,30 @@ const totals = (t) => ({
|
|
|
13
13
|
costEstimate: t.costEstimate
|
|
14
14
|
})
|
|
15
15
|
|
|
16
|
-
/**
|
|
16
|
+
/**
|
|
17
|
+
* OData exposes structured elements (`tokens`) as flat columns (`tokens_prompt`, …) unless the
|
|
18
|
+
* app uses structured OData (`odata.structs`, e.g. flavor x4). Return rows in the same shape.
|
|
19
|
+
*/
|
|
20
|
+
function shape (row) {
|
|
21
|
+
if (cds.env.effective?.odata?.structs || !row.tokens) return row
|
|
22
|
+
const { tokens, ...rest } = row
|
|
23
|
+
for (const [k, v] of Object.entries(tokens)) rest[`tokens_${k}`] = v
|
|
24
|
+
return rest
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* Rows for a READ, honouring a key lookup (e.g. UseCases('ask')). Lists carry `$count`, so
|
|
29
|
+
* `$count=true` (used by Fiori elements to size its tables) reports the number of rows.
|
|
30
|
+
*/
|
|
17
31
|
function rows (req, list, key) {
|
|
18
32
|
const wanted = req.data?.[key]
|
|
19
|
-
if (wanted
|
|
20
|
-
|
|
33
|
+
if (wanted !== undefined) {
|
|
34
|
+
const row = list.find(r => r[key] === wanted)
|
|
35
|
+
return row ? shape(row) : null
|
|
36
|
+
}
|
|
37
|
+
const result = list.map(shape)
|
|
38
|
+
result.$count = result.length
|
|
39
|
+
return result
|
|
21
40
|
}
|
|
22
41
|
|
|
23
42
|
module.exports = class TokenOptimizerService extends cds.ApplicationService {
|