@viveka10/cap-ai-token-optimizer 0.1.0 → 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.cdsrc.js +27 -0
- package/CHANGELOG.md +44 -1
- package/README.md +24 -8
- package/cds-plugin.js +5 -19
- package/docs/existing-cap-app-on-btp.md +426 -0
- package/docs/integration-guide.md +2 -1
- package/lib/metrics/classifier.js +13 -1
- package/lib/models.js +24 -0
- package/lib/optimizer.js +50 -2
- package/package.json +14 -3
- package/srv/admin-service.js +22 -3
package/.cdsrc.js
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
'use strict'
|
|
2
|
+
/**
|
|
3
|
+
* Plugin configuration defaults, read by cds.env. cds.env is also rebuilt from these files
|
|
4
|
+
* after plugins are loaded (`cds build`, `cds deploy`), so anything cds-plugin.js wrote into
|
|
5
|
+
* cds.env would be lost there.
|
|
6
|
+
*
|
|
7
|
+
* `model` is therefore a getter: CAP's config merge keeps getters, and CAP reads `model`
|
|
8
|
+
* when it resolves the app's models, after the app's own configuration and profiles have been
|
|
9
|
+
* merged. Models the app sets explicitly are kept.
|
|
10
|
+
*/
|
|
11
|
+
const { modelsFor } = require('./lib/models')
|
|
12
|
+
|
|
13
|
+
module.exports = {
|
|
14
|
+
requires: {
|
|
15
|
+
tokenOptimizer: {
|
|
16
|
+
kind: 'token-optimizer',
|
|
17
|
+
get model () {
|
|
18
|
+
const own = [].concat(this._model ?? [])
|
|
19
|
+
const models = [...new Set([...own, ...modelsFor(this)])]
|
|
20
|
+
return models.length ? models : undefined
|
|
21
|
+
},
|
|
22
|
+
set model (value) {
|
|
23
|
+
Object.defineProperty(this, '_model', { value, writable: true, configurable: true, enumerable: false })
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
}
|
|
27
|
+
}
|
package/CHANGELOG.md
CHANGED
|
@@ -6,6 +6,47 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.1.2] - 2026-10-11
|
|
10
|
+
|
|
11
|
+
### Fixed
|
|
12
|
+
|
|
13
|
+
- **Token usage of streaming calls.** `stream()` calls (including LangChain models with
|
|
14
|
+
streaming, as used by `@cap-js/agents`) were recorded when the stream opened, without token
|
|
15
|
+
usage. They are now recorded when the stream ends, with AI Core's usage, finish reason, request
|
|
16
|
+
id and cost.
|
|
17
|
+
- **Errors inside a stream** were not recorded, and the call counted as a success. They are now
|
|
18
|
+
recorded as errors and classified from the error event AI Core sends (for example `SERVER`,
|
|
19
|
+
`RATE_LIMIT`, `CONTENT_FILTER`).
|
|
20
|
+
- **`cds build` and `cds deploy` left out the plugin's models.** The `TokenUsageLog` entity and
|
|
21
|
+
the admin service were added to `cds.env` when the plugin loaded, but `cds build` rebuilds
|
|
22
|
+
`cds.env` afterwards, so the MTA had no usage-log table and no admin service. The models now
|
|
23
|
+
come from the plugin's `.cdsrc.js` and are computed from the app's final configuration
|
|
24
|
+
(profiles and environment variables included). Apps that worked around this by listing the
|
|
25
|
+
models in `cds.requires.tokenOptimizer.model` keep working and can remove the list.
|
|
26
|
+
- **Admin service for Fiori elements:** the in-memory entities (`UseCases`, `Models`,
|
|
27
|
+
`ErrorCategories`, `StepSavings`, `PromptPrefixChanges`, `Exporters`) answered `$count=true`
|
|
28
|
+
with 0, so list reports showed empty tables, and `UseCases` and `Models` returned `tokens` as a
|
|
29
|
+
nested object instead of the `tokens_*` columns of their OData metadata.
|
|
30
|
+
|
|
31
|
+
### Changed
|
|
32
|
+
|
|
33
|
+
- `latencyMs` of streaming calls now covers the whole answer instead of the time until the
|
|
34
|
+
stream opened. A stream closed early by the caller, or not read within 10 minutes, is recorded
|
|
35
|
+
at that point without usage and with a note in `savings.notes`.
|
|
36
|
+
- The mock AI Core server answers streaming requests as server-sent events and has a
|
|
37
|
+
`stream_error` scenario.
|
|
38
|
+
|
|
39
|
+
## [0.1.1] - 2026-10-10
|
|
40
|
+
|
|
41
|
+
### Added
|
|
42
|
+
|
|
43
|
+
- Guide "Add to an existing CAP app on SAP BTP" (`docs/existing-cap-app-on-btp.md`): AI Core
|
|
44
|
+
service binding, `mta.yaml`, secrets, HANA usage log, admin role, and SAP Cloud ALM setup.
|
|
45
|
+
|
|
46
|
+
### Changed
|
|
47
|
+
|
|
48
|
+
- Package description, keywords and README introduction, so the package is easier to find.
|
|
49
|
+
|
|
9
50
|
## [0.1.0] - 2026-10-10
|
|
10
51
|
|
|
11
52
|
First public release.
|
|
@@ -67,5 +108,7 @@ First public release.
|
|
|
67
108
|
- The `http` exporter supports Internet destinations only.
|
|
68
109
|
- Not yet verified against SAP Cloud ALM; live AI Core tests are opt-in.
|
|
69
110
|
|
|
70
|
-
[Unreleased]: https://github.com/viveka10/npm_token_optimizer/compare/v0.1.
|
|
111
|
+
[Unreleased]: https://github.com/viveka10/npm_token_optimizer/compare/v0.1.2...HEAD
|
|
112
|
+
[0.1.2]: https://github.com/viveka10/npm_token_optimizer/compare/v0.1.1...v0.1.2
|
|
113
|
+
[0.1.1]: https://github.com/viveka10/npm_token_optimizer/compare/v0.1.0...v0.1.1
|
|
71
114
|
[0.1.0]: https://github.com/viveka10/npm_token_optimizer/releases/tag/v0.1.0
|
package/README.md
CHANGED
|
@@ -5,7 +5,12 @@
|
|
|
5
5
|
[](https://github.com/viveka10/npm_token_optimizer/blob/main/cap-ai-token-optimizer/LICENSE)
|
|
6
6
|

|
|
7
7
|
|
|
8
|
-
|
|
8
|
+
**SAP CAP plugin** that reduces, measures and reports the LLM tokens your SAP CAP Node.js app
|
|
9
|
+
spends on **SAP AI Core**, and with them your generative AI costs.
|
|
10
|
+
|
|
11
|
+
```sh
|
|
12
|
+
npm add @viveka10/cap-ai-token-optimizer
|
|
13
|
+
```
|
|
9
14
|
|
|
10
15
|
- **Optimize** prompts without changing what they mean: whitespace and JSON, duplicate
|
|
11
16
|
instructions, long histories, large tool lists, output limits, and an exact-match response cache.
|
|
@@ -20,6 +25,8 @@ Node.js. Wraps the SAP Cloud SDK for AI (`@sap-ai-sdk/orchestration`, `@sap-ai-s
|
|
|
20
25
|
> **Status: pre-release.** Not affiliated with or endorsed by SAP.
|
|
21
26
|
|
|
22
27
|
**New here?** [Add it to an existing CAP app in 5 minutes](https://github.com/viveka10/npm_token_optimizer/blob/main/cap-ai-token-optimizer/docs/integration-guide.md): no code changes needed.
|
|
28
|
+
Deploying on SAP BTP with an AI Core binding and SAP Cloud ALM? See
|
|
29
|
+
[Existing CAP app on SAP BTP](https://github.com/viveka10/npm_token_optimizer/blob/main/cap-ai-token-optimizer/docs/existing-cap-app-on-btp.md).
|
|
23
30
|
|
|
24
31
|
**Contents:** [Requirements](#requirements) · [Setup](#setup) · [Usage](#usage) ·
|
|
25
32
|
[Existing code and LangChain](#instrumenting-existing-code-and-langchain) · [Compatibility](#compatibility) ·
|
|
@@ -97,7 +104,10 @@ Wrap clients once (for example in your service's `init()`) and reuse them.
|
|
|
97
104
|
|
|
98
105
|
- **Supported clients:** `OrchestrationClient` (`chatCompletion`, `stream`) and
|
|
99
106
|
`AzureOpenAiChatClient` (`run`, `stream`). Other methods are forwarded untouched.
|
|
100
|
-
- **Streaming:** `stream()` calls are passed through unoptimized
|
|
107
|
+
- **Streaming:** `stream()` calls are passed through unoptimized. They are recorded when the
|
|
108
|
+
stream has been read to the end, with AI Core's token usage, the finish reason and the latency
|
|
109
|
+
of the whole answer. A failure inside the stream is recorded as an error. A stream closed early,
|
|
110
|
+
or not read within 10 minutes, is recorded at that point without usage and with a note.
|
|
101
111
|
- **Errors:** SDK errors are rethrown unchanged. Use `classifyError(err)` to get the failure category.
|
|
102
112
|
|
|
103
113
|
```js
|
|
@@ -134,7 +144,7 @@ This hooks the `@sap-ai-sdk` client classes `OrchestrationClient` (`chatCompleti
|
|
|
134
144
|
In short:
|
|
135
145
|
- only these two classes from SDK 2.x, and only the SDK copy resolved from the app root;
|
|
136
146
|
- process-wide, including libraries;
|
|
137
|
-
- streaming is measured but not optimized.
|
|
147
|
+
- streaming is measured (with token usage) but not optimized.
|
|
138
148
|
|
|
139
149
|
### withUseCase
|
|
140
150
|
|
|
@@ -426,9 +436,10 @@ optimizer.configure({
|
|
|
426
436
|
})
|
|
427
437
|
```
|
|
428
438
|
|
|
429
|
-
The `db` exporter and the admin service also need their CDS models
|
|
430
|
-
|
|
431
|
-
|
|
439
|
+
The `db` exporter and the admin service also need their CDS models. The plugin adds them to
|
|
440
|
+
the app's model (also in `cds build` and `cds deploy`) when they are enabled in the
|
|
441
|
+
configuration, including profiles and environment variables. Enable them in the
|
|
442
|
+
configuration, not only in code.
|
|
432
443
|
|
|
433
444
|
`createOptimizer(config)` creates a separate instance with its own stats and exporters.
|
|
434
445
|
|
|
@@ -455,7 +466,8 @@ One event per call:
|
|
|
455
466
|
|
|
456
467
|
- Prompt and completion text are never recorded by default; only a SHA-256 hash and the length.
|
|
457
468
|
`user` is a salted hash of the user id.
|
|
458
|
-
- `tokens.prompt`/`completion`/`total`/`cached` come from the AI Core response
|
|
469
|
+
- `tokens.prompt`/`completion`/`total`/`cached` come from the AI Core response, for streams
|
|
470
|
+
from the last chunk. Orchestration streams don't report cached tokens (`cached` is 0).
|
|
459
471
|
`promptEstimatedBefore`/`After` are estimates before and after optimization (js-tiktoken for
|
|
460
472
|
OpenAI model families, characters ÷ 4 for other models).
|
|
461
473
|
- `savings.estimatedTokens` is estimated prompt tokens saved plus, for cache hits, the cached
|
|
@@ -547,7 +559,8 @@ Writes one row per call to the entity `cap.ai.token_optimizer.TokenUsageLog`
|
|
|
547
559
|
(table `CAP_AI_TOKEN_OPTIMIZER_TOKENUSAGELOG`) through CAP's database service.
|
|
548
560
|
|
|
549
561
|
- **Databases:** SAP HANA, SQLite or PostgreSQL. `hana` is an alias for `db`.
|
|
550
|
-
- **Model:** the plugin adds the entity to your model when the exporter is configured
|
|
562
|
+
- **Model:** the plugin adds the entity to your model when the exporter is configured, also in
|
|
563
|
+
`cds build`, so it is part of the MTA's db deployer.
|
|
551
564
|
Deploy as usual (`cds deploy`, or the MTA db deployer), and run your MTX upgrade for multitenant apps.
|
|
552
565
|
- **Multitenancy:** rows are inserted in batches on the flush interval with a privileged
|
|
553
566
|
user. Each tenant's rows go to its own container.
|
|
@@ -587,6 +600,9 @@ for users with the role `admin.role` (default `TokenOptimizerAdmin`):
|
|
|
587
600
|
| `UsageLog` | the `TokenUsageLog` rows, if the `db` exporter is enabled (supports `$filter`, `$orderby`, `$top`, …) |
|
|
588
601
|
|
|
589
602
|
All entries except `UsageLog` are in-memory stats of the app instance that answers, since it started.
|
|
603
|
+
They support key lookups and `$count`, which is enough for a Fiori elements list report, but not
|
|
604
|
+
`$filter`, `$orderby`, `$search` or paging. With flat OData (the default), `tokens` is exposed as
|
|
605
|
+
`tokens_prompt`, `tokens_completion`, `tokens_total` and `tokens_cached`.
|
|
590
606
|
|
|
591
607
|
To grant access, add the role to `xs-security.json` and a role collection:
|
|
592
608
|
|
package/cds-plugin.js
CHANGED
|
@@ -6,35 +6,21 @@
|
|
|
6
6
|
const cds = globalThis.cds ?? require('@sap/cds')
|
|
7
7
|
const { optimizer, countText } = require('./index')
|
|
8
8
|
const log = require('./lib/util/log')
|
|
9
|
-
const path = require('node:path')
|
|
10
9
|
|
|
11
10
|
try {
|
|
12
11
|
optimizer.configure(cds.env.requires?.tokenOptimizer)
|
|
13
12
|
log.debug('configured, mode =', optimizer.config.mode, 'exporters =', optimizer.config.metrics.exporters)
|
|
14
|
-
|
|
13
|
+
setupAdminRole(optimizer.config)
|
|
15
14
|
} catch (e) {
|
|
16
15
|
log.error('could not configure token optimizer:', e?.message)
|
|
17
16
|
}
|
|
18
17
|
|
|
19
18
|
/**
|
|
20
|
-
*
|
|
21
|
-
*
|
|
22
|
-
* UsageLog projection when both are on. Plugins load before the model, so adding a `model`
|
|
23
|
-
* to our cds.requires entry is enough.
|
|
19
|
+
* The CDS models (usage log entity, admin service) are added through `model` in .cdsrc.js, so
|
|
20
|
+
* they are also part of `cds build` and `cds deploy`. Here we only set the admin service's role.
|
|
24
21
|
*/
|
|
25
|
-
function
|
|
26
|
-
|
|
27
|
-
const dbLog = exporters.includes('db') || exporters.includes('hana')
|
|
28
|
-
const admin = cfg.admin?.enabled
|
|
29
|
-
const models = []
|
|
30
|
-
if (dbLog) models.push('db/token-usage-log')
|
|
31
|
-
if (admin) models.push('srv/admin-service')
|
|
32
|
-
if (admin && dbLog) models.push('srv/admin-usage-log')
|
|
33
|
-
if (!models.length) return
|
|
34
|
-
|
|
35
|
-
const req = cds.env.requires.tokenOptimizer ??= {}
|
|
36
|
-
req.model = [...new Set([].concat(req.model ?? [], models.map(m => path.join(__dirname, m))))]
|
|
37
|
-
if (!admin) return
|
|
22
|
+
function setupAdminRole (cfg) {
|
|
23
|
+
if (!cfg.admin?.enabled) return
|
|
38
24
|
const role = cfg.admin.role || 'TokenOptimizerAdmin'
|
|
39
25
|
cds.on('loaded', (csn) => {
|
|
40
26
|
const srv = csn?.definitions?.TokenOptimizerService
|
|
@@ -0,0 +1,426 @@
|
|
|
1
|
+
# Add @viveka10/cap-ai-token-optimizer to an existing CAP app on SAP BTP
|
|
2
|
+
|
|
3
|
+
This guide is for a CAP Node.js app on SAP BTP Cloud Foundry that already calls SAP AI Core
|
|
4
|
+
through the SAP Cloud SDK for AI (`@sap-ai-sdk/*`) with an **AI Core service binding**. It
|
|
5
|
+
covers the changes to the app, deployment (`mta.yaml`, `xs-security.json`) and SAP Cloud ALM.
|
|
6
|
+
|
|
7
|
+
**What stays the same**
|
|
8
|
+
|
|
9
|
+
- **AI Core binding:** no change. The SDK keeps reading it from `VCAP_SERVICES`.
|
|
10
|
+
- **AI code:** no change needed. Automatic instrumentation measures the existing calls.
|
|
11
|
+
- **Prompts:** you start in `shadow` mode, which sends your original prompts unchanged.
|
|
12
|
+
|
|
13
|
+
**Contents**
|
|
14
|
+
|
|
15
|
+
1. [Check the prerequisites](#1-check-the-prerequisites)
|
|
16
|
+
2. [Install the package](#2-install-the-package)
|
|
17
|
+
3. [Configure it](#3-configure-it)
|
|
18
|
+
4. [Code changes (optional)](#4-code-changes-optional)
|
|
19
|
+
5. [Local testing with the AI Core binding](#5-local-testing-with-the-ai-core-binding)
|
|
20
|
+
6. [Secrets and environment](#6-secrets-and-environment)
|
|
21
|
+
7. [Usage log in SAP HANA (optional)](#7-usage-log-in-sap-hana-optional)
|
|
22
|
+
8. [Admin service and roles (optional)](#8-admin-service-and-roles-optional)
|
|
23
|
+
9. [SAP Cloud ALM](#9-sap-cloud-alm)
|
|
24
|
+
10. [Deploy and verify](#10-deploy-and-verify)
|
|
25
|
+
11. [Switch from shadow to active](#11-switch-from-shadow-to-active)
|
|
26
|
+
12. [Turn it off or roll back](#12-turn-it-off-or-roll-back)
|
|
27
|
+
|
|
28
|
+
---
|
|
29
|
+
|
|
30
|
+
## 1. Check the prerequisites
|
|
31
|
+
|
|
32
|
+
In the app's root folder:
|
|
33
|
+
|
|
34
|
+
```sh
|
|
35
|
+
node --version
|
|
36
|
+
npm ls @sap/cds @sap-ai-sdk/orchestration @sap-ai-sdk/foundation-models @sap-ai-sdk/langchain
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
| Requirement | Why |
|
|
40
|
+
|---|---|
|
|
41
|
+
| Node.js **22.12 or later**, locally and on Cloud Foundry | The SAP Cloud SDK for AI is ESM-only; Node 22.12+ can load it from CommonJS. |
|
|
42
|
+
| `"engines": { "node": ">=22.12" }` in the app's `package.json` | The Cloud Foundry Node.js buildpack picks the Node version from `engines`. |
|
|
43
|
+
| `@sap/cds` 9 or 10 | Tested versions. |
|
|
44
|
+
| `@sap-ai-sdk/*` **2.8 or later**, all packages on the **same** version, each listed **once** | The packages share internal APIs. Automatic instrumentation only hooks the copy resolved from the app root. If `npm ls` shows nested copies, run `npm dedupe` or align the versions. |
|
|
45
|
+
|
|
46
|
+
## 2. Install the package
|
|
47
|
+
|
|
48
|
+
```sh
|
|
49
|
+
npm add @viveka10/cap-ai-token-optimizer
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
It is a CAP plugin, so there's nothing to register: CAP loads it at startup.
|
|
53
|
+
|
|
54
|
+
## 3. Configure it
|
|
55
|
+
|
|
56
|
+
Add the `tokenOptimizer` block to `cds.requires` in the app's `package.json` (or `.cdsrc.json`).
|
|
57
|
+
This starting configuration measures everything without changing any AI call:
|
|
58
|
+
|
|
59
|
+
```json
|
|
60
|
+
{
|
|
61
|
+
"cds": {
|
|
62
|
+
"requires": {
|
|
63
|
+
"tokenOptimizer": {
|
|
64
|
+
"mode": "shadow",
|
|
65
|
+
"autoInstrument": true,
|
|
66
|
+
"pricing": {
|
|
67
|
+
"gpt-4.1": { "inputPer1k": 0.0, "outputPer1k": 0.0 }
|
|
68
|
+
},
|
|
69
|
+
"metrics": { "exporters": ["console"] },
|
|
70
|
+
"[production]": {
|
|
71
|
+
"metrics": { "exporters": ["console", "otel"] },
|
|
72
|
+
"admin": { "enabled": true }
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
| Setting | What it does |
|
|
81
|
+
|---|---|
|
|
82
|
+
| `mode: "shadow"` | Runs the optimizer and records what it *would* save, but sends the original prompts. |
|
|
83
|
+
| `autoInstrument: true` | Hooks the SDK's `OrchestrationClient` and `AzureOpenAiChatClient`, so existing calls are measured without code changes. |
|
|
84
|
+
| `pricing` | Price per 1k tokens per model, from your SAP contract. Leave it out if you don't need cost estimates. |
|
|
85
|
+
| `metrics.exporters` | Where events go. `console` writes one JSON line per AI call to the app log (`cf logs`). `otel` sends OpenTelemetry metrics and spans (see [SAP Cloud ALM](#9-sap-cloud-alm)). `db` writes a usage table (see [step 7](#7-usage-log-in-sap-hana-optional)). |
|
|
86
|
+
| `admin.enabled` | Serves the read-only statistics service at `/odata/v4/token-optimizer` (see [step 8](#8-admin-service-and-roles-optional)). |
|
|
87
|
+
|
|
88
|
+
All options are described in the [configuration reference](https://github.com/viveka10/npm_token_optimizer/blob/main/cap-ai-token-optimizer/README.md#configuration-reference).
|
|
89
|
+
|
|
90
|
+
## 4. Code changes (optional)
|
|
91
|
+
|
|
92
|
+
With `autoInstrument: true`, no code change is needed. All hooked calls are recorded under
|
|
93
|
+
the use case `default`. These optional changes give you better reporting and per-use-case
|
|
94
|
+
settings:
|
|
95
|
+
|
|
96
|
+
**Name the use cases** where your handlers call AI Core:
|
|
97
|
+
|
|
98
|
+
```js
|
|
99
|
+
const { optimizer } = require('@viveka10/cap-ai-token-optimizer')
|
|
100
|
+
|
|
101
|
+
this.on('summarizeTicket', req =>
|
|
102
|
+
optimizer.withUseCase('ticket-summary', () => this.summarize(req.data.ID)))
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
**LangChain models** from `@sap-ai-sdk/langchain`: wrap them once where you create them:
|
|
106
|
+
|
|
107
|
+
```js
|
|
108
|
+
const model = optimizer.wrapLangChain(new OrchestrationClient(config), { useCase: 'support-agent' })
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
**Better error responses** (optional): the SDK's errors are passed through unchanged, and
|
|
112
|
+
`classifyError` tells you what happened:
|
|
113
|
+
|
|
114
|
+
```js
|
|
115
|
+
const { classifyError } = require('@viveka10/cap-ai-token-optimizer')
|
|
116
|
+
try { /* AI call */ } catch (err) {
|
|
117
|
+
const { category } = classifyError(err) // RATE_LIMIT, CONTENT_FILTER, CONTEXT_LENGTH, TIMEOUT, …
|
|
118
|
+
return req.reject(category === 'RATE_LIMIT' ? 429 : 502, `AI service error: ${category}`)
|
|
119
|
+
}
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
Use cases can then get their own settings, for example
|
|
123
|
+
`"useCases": { "ticket-summary": { "steps": { "outputCap": { "default": 300 } } } }`.
|
|
124
|
+
|
|
125
|
+
## 5. Local testing with the AI Core binding
|
|
126
|
+
|
|
127
|
+
Nothing changes in how the app gets its AI Core credentials. The SDK reads the `aicore` binding
|
|
128
|
+
from `VCAP_SERVICES`. If you already test locally with `cds bind`, keep doing that:
|
|
129
|
+
|
|
130
|
+
```sh
|
|
131
|
+
cf login …
|
|
132
|
+
cds bind aicore --to <your-aicore-instance>:<service-key> # once; writes .cdsrc-private.json (keep it out of git)
|
|
133
|
+
cds bind --exec -- cds watch --profile hybrid # VCAP_SERVICES is set for the SDK
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
Call the feature that uses AI Core. The console shows one line per call:
|
|
137
|
+
|
|
138
|
+
```json
|
|
139
|
+
{"type":"ai-core-call","useCase":"default","mode":"shadow","model":"gpt-4.1",
|
|
140
|
+
"tokens":{"promptEstimatedBefore":2015,"promptEstimatedAfter":805,"prompt":2010,"completion":41,…},
|
|
141
|
+
"savings":{"estimatedTokens":1210,"byStep":{"history":774,"dedupe":340,…},…},"status":"success",…}
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
`savings` shows what `active` mode would have saved on that call.
|
|
145
|
+
|
|
146
|
+
## 6. Secrets and environment
|
|
147
|
+
|
|
148
|
+
**User-id hash salt.** User ids are recorded as a hash. Set a random salt per landscape, but
|
|
149
|
+
**not** in `package.json` or `mta.yaml` if they are in git. Use an MTA extension descriptor that
|
|
150
|
+
you keep out of git:
|
|
151
|
+
|
|
152
|
+
```yaml
|
|
153
|
+
# secrets.mtaext (do not commit)
|
|
154
|
+
_schema-version: '3.1'
|
|
155
|
+
ID: my-cap-app.secrets
|
|
156
|
+
extends: my-cap-app
|
|
157
|
+
modules:
|
|
158
|
+
- name: my-cap-app-srv
|
|
159
|
+
properties:
|
|
160
|
+
cds_requires_tokenOptimizer_metrics_userHashSalt: <long random string>
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
```sh
|
|
164
|
+
cf deploy mta_archives/my-cap-app_1.0.0.mtar -e secrets.mtaext
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
**Overriding settings without a rebuild.** Any setting can be overridden with an environment
|
|
168
|
+
variable in the same format, for example `cds_requires_tokenOptimizer_mode=off`, followed by
|
|
169
|
+
`cf restart`.
|
|
170
|
+
|
|
171
|
+
Never set `captureContent: true` in production. It adds prompt and answer text to events.
|
|
172
|
+
|
|
173
|
+
## 7. Usage log in SAP HANA (optional)
|
|
174
|
+
|
|
175
|
+
To keep one row per AI call in the app's database (for reporting or your own analytics), add
|
|
176
|
+
the `db` exporter:
|
|
177
|
+
|
|
178
|
+
```json
|
|
179
|
+
"[production]": { "metrics": { "exporters": ["console", "otel", "db"] }, "admin": { "enabled": true } }
|
|
180
|
+
```
|
|
181
|
+
|
|
182
|
+
- The plugin adds the entity `cap.ai.token_optimizer.TokenUsageLog` to your model, also in
|
|
183
|
+
`cds build --production`. It is deployed with your normal HANA deployment (the `db-deployer`
|
|
184
|
+
module in the MTA), so there's no extra step. With version 0.1.1 the entity was missing from
|
|
185
|
+
`cds build`; there, list the models in `cds.requires.tokenOptimizer.model` as a workaround.
|
|
186
|
+
- **Multitenant apps (MTX):** each tenant's rows go to that tenant's container. Run your
|
|
187
|
+
usual tenant upgrade after deploying, so existing tenants get the new table.
|
|
188
|
+
- Rows older than 30 days are deleted (`metrics.db.retentionDays`). No prompt or answer text
|
|
189
|
+
is stored.
|
|
190
|
+
- With the admin service enabled, the rows are readable at `/odata/v4/token-optimizer/UsageLog`.
|
|
191
|
+
- To expose the table in your own service:
|
|
192
|
+
```cds
|
|
193
|
+
using { cap.ai.token_optimizer.TokenUsageLog } from '@viveka10/cap-ai-token-optimizer/db/token-usage-log';
|
|
194
|
+
service ReportingService { @readonly entity AiUsage as projection on TokenUsageLog; }
|
|
195
|
+
```
|
|
196
|
+
|
|
197
|
+
> The usage log has been tested with SQLite. It hasn't been tested on SAP HANA or with MTX
|
|
198
|
+
> yet, so check it in a development space first.
|
|
199
|
+
|
|
200
|
+
## 8. Admin service and roles (optional)
|
|
201
|
+
|
|
202
|
+
With `admin.enabled`, the app serves `/odata/v4/token-optimizer`:
|
|
203
|
+
|
|
204
|
+
| Path | Content |
|
|
205
|
+
|---|---|
|
|
206
|
+
| `summary()` | totals, mode, latency percentiles |
|
|
207
|
+
| `UseCases`, `Models`, `ErrorCategories`, `StepSavings`, `PromptPrefixChanges`, `Exporters` | statistics of the app instance that answers (in memory, since its start) |
|
|
208
|
+
| `UsageLog` | the usage table, when the `db` exporter is on |
|
|
209
|
+
|
|
210
|
+
It requires the role `TokenOptimizerAdmin` (configurable with `admin.role`). Add it to
|
|
211
|
+
`xs-security.json`:
|
|
212
|
+
|
|
213
|
+
```json
|
|
214
|
+
{
|
|
215
|
+
"scopes": [
|
|
216
|
+
{ "name": "$XSAPPNAME.TokenOptimizerAdmin", "description": "Read AI token usage statistics" }
|
|
217
|
+
],
|
|
218
|
+
"role-templates": [
|
|
219
|
+
{ "name": "TokenOptimizerAdmin", "scope-references": ["$XSAPPNAME.TokenOptimizerAdmin"] }
|
|
220
|
+
],
|
|
221
|
+
"role-collections": [
|
|
222
|
+
{ "name": "AI_Token_Optimizer_Admin", "role-template-references": ["$XSAPPNAME.TokenOptimizerAdmin"] }
|
|
223
|
+
]
|
|
224
|
+
}
|
|
225
|
+
```
|
|
226
|
+
|
|
227
|
+
Merge this into your existing file; don't replace it. After deploying, assign the role
|
|
228
|
+
collection to the right users in the BTP cockpit (Security → Role Collections).
|
|
229
|
+
|
|
230
|
+
## 9. SAP Cloud ALM
|
|
231
|
+
|
|
232
|
+
The optimizer has no Cloud ALM-specific code. Its `otel` exporter writes OpenTelemetry metrics
|
|
233
|
+
and one span per AI call through whatever OpenTelemetry setup the app has. For Cloud ALM, that
|
|
234
|
+
setup is SAP's Cloud ALM agent plus `@cap-js/telemetry`. The agent setup below follows SAP's
|
|
235
|
+
documentation:
|
|
236
|
+
[Data Collection Infrastructure](https://support.sap.com/en/alm/sap-cloud-alm/operations/expert-portal/data-collection-infrastructure.html) and
|
|
237
|
+
[Auto-instrumentation for Node.js](https://support.sap.com/en/alm/sap-cloud-alm/operations/expert-portal/data-collection-infrastructure/autoinstrumentation-nodejs.html).
|
|
238
|
+
Check those pages for changes before you start.
|
|
239
|
+
|
|
240
|
+
### 9.1 Prerequisites
|
|
241
|
+
|
|
242
|
+
- An SAP Cloud ALM tenant, with the **SAP Cloud ALM API** enabled and a service key for it
|
|
243
|
+
([Enabling SAP Cloud ALM API](https://help.sap.com/docs/cloud-alm/setup-administration/enabling-sap-cloud-alm-api)).
|
|
244
|
+
- A technical user for SAP's **Repository Based Shipment Channel** (RBSC), which distributes
|
|
245
|
+
the agent. Request it at https://ui.repositories.cloud.sap/www/webapp/users. The agent isn't
|
|
246
|
+
on the public npm registry.
|
|
247
|
+
|
|
248
|
+
### 9.2 Get access to SAP's package repository
|
|
249
|
+
|
|
250
|
+
Set `SAP_NPM_AUTH` to the "NPM Base64 Credentials" of the technical user (`base64(user:password)`),
|
|
251
|
+
in your shell and in your CI/build pipeline (as a secret), and add to the app's `.npmrc`:
|
|
252
|
+
|
|
253
|
+
```
|
|
254
|
+
//73555000100200018064.npmsrv.cdn.repositories.cloud.sap/:_auth=${SAP_NPM_AUTH}
|
|
255
|
+
```
|
|
256
|
+
|
|
257
|
+
### 9.3 Add the agent and @cap-js/telemetry
|
|
258
|
+
|
|
259
|
+
In the app's `package.json`:
|
|
260
|
+
|
|
261
|
+
```json
|
|
262
|
+
{
|
|
263
|
+
"dependencies": {
|
|
264
|
+
"@sap/xotel-agent-ext-js": "https://73555000100200018064.npmsrv.cdn.repositories.cloud.sap/@sap/xotel-agent-ext-js/-/xotel-agent-ext-js-2.0.4.tgz",
|
|
265
|
+
"@cap-js/telemetry": "^2.1.0"
|
|
266
|
+
},
|
|
267
|
+
"scripts": {
|
|
268
|
+
"start": "node ${NODE_ARGS} ./node_modules/.bin/cds-serve"
|
|
269
|
+
}
|
|
270
|
+
}
|
|
271
|
+
```
|
|
272
|
+
|
|
273
|
+
- **Agent version:** 2.0.4 or later. That is the version from which `@cap-js/telemetry`
|
|
274
|
+
supports it, and it uses OpenTelemetry 2.x.
|
|
275
|
+
- **`@cap-js/telemetry` version:** 2.1.0 or later. It detects the agent and adds its span
|
|
276
|
+
processor, metric reader and log processor to the agent's pipeline instead of starting a
|
|
277
|
+
second OpenTelemetry SDK. Version 2.0.x doesn't have this integration.
|
|
278
|
+
- **Start script:** loads the agent before CAP via `NODE_ARGS` (set in `mta.yaml` below).
|
|
279
|
+
After `cds build`, check that `gen/srv/package.json` contains this start script.
|
|
280
|
+
- **Dependency conflicts:** SAP notes that the agent's OpenTelemetry 2.x libraries can
|
|
281
|
+
conflict with other OpenTelemetry packages in the app. `@viveka10/cap-ai-token-optimizer`
|
|
282
|
+
itself only uses `@opentelemetry/api`.
|
|
283
|
+
|
|
284
|
+
### 9.4 Enable the otel exporter
|
|
285
|
+
|
|
286
|
+
```json
|
|
287
|
+
"tokenOptimizer": { "[production]": { "metrics": { "exporters": ["console", "otel"] } } }
|
|
288
|
+
```
|
|
289
|
+
|
|
290
|
+
### 9.5 mta.yaml
|
|
291
|
+
|
|
292
|
+
```yaml
|
|
293
|
+
modules:
|
|
294
|
+
- name: my-cap-app-srv
|
|
295
|
+
type: nodejs
|
|
296
|
+
path: gen/srv
|
|
297
|
+
properties:
|
|
298
|
+
SAP_CALM_SERVICE_NAME: my-cap-app-srv # name shown in Cloud ALM
|
|
299
|
+
SAP_CALM_SERVICE_TYPE: SAP_CP_CF
|
|
300
|
+
NODE_ARGS: -r @sap/xotel-agent-ext-js
|
|
301
|
+
requires:
|
|
302
|
+
- name: my-cap-app-aicore # existing AI Core binding, unchanged
|
|
303
|
+
- name: my-cap-app-auth # existing XSUAA
|
|
304
|
+
- name: my-cap-app-db # existing, only needed for the db exporter
|
|
305
|
+
- name: my-cap-app-destination # new: used by the Cloud ALM agent
|
|
306
|
+
|
|
307
|
+
resources:
|
|
308
|
+
- name: my-cap-app-destination
|
|
309
|
+
type: org.cloudfoundry.managed-service
|
|
310
|
+
parameters:
|
|
311
|
+
service: destination
|
|
312
|
+
service-plan: lite
|
|
313
|
+
```
|
|
314
|
+
|
|
315
|
+
**Staging needs the agent's repository.** Cloud Foundry installs dependencies during staging,
|
|
316
|
+
so it has to reach SAP's repository too. SAP suggests, if staging fails, shipping
|
|
317
|
+
`node_modules` and `package-lock.json` inside the MTA archive: remove them from the module's
|
|
318
|
+
`build-parameters.ignore` list, and install them into `gen/srv` with a `before-all` custom
|
|
319
|
+
build step (`npm install --production` and `npm install --prefix ./gen/srv`).
|
|
320
|
+
|
|
321
|
+
### 9.6 Destination for Cloud ALM
|
|
322
|
+
|
|
323
|
+
In the **provider subaccount** (BTP cockpit → Connectivity → Destinations), create:
|
|
324
|
+
|
|
325
|
+
| Field | Value |
|
|
326
|
+
|---|---|
|
|
327
|
+
| Name | `CALM_datacollector` (exactly this name) |
|
|
328
|
+
| Type | HTTP |
|
|
329
|
+
| URL | the `Api` URL from the Cloud ALM API service key, e.g. `https://eu10.alm.cloud.sap/api` |
|
|
330
|
+
| Proxy Type | Internet |
|
|
331
|
+
| Authentication | OAuth2ClientCredentials |
|
|
332
|
+
| Client ID / Client Secret | from the Cloud ALM API service key |
|
|
333
|
+
| Token Service URL | the `url` from the service key |
|
|
334
|
+
|
|
335
|
+
SAP notes that "Check Connection" can return 404 even when the destination is correct.
|
|
336
|
+
|
|
337
|
+
### 9.7 Activate monitoring in Cloud ALM
|
|
338
|
+
|
|
339
|
+
In SAP Cloud ALM, activate the monitoring use cases you need for the service (for example
|
|
340
|
+
Health Monitoring and Exception Monitoring). Optional agent settings and how to turn the agent
|
|
341
|
+
off (`SAP_CALM_INSTRUMENTATION_ENABLED=false`) are on SAP's auto-instrumentation page.
|
|
342
|
+
|
|
343
|
+
### 9.8 What you'll see, and what isn't verified
|
|
344
|
+
|
|
345
|
+
- **App log after deploying:** a line from `@cap-js/telemetry` saying it found
|
|
346
|
+
`@sap/xotel-agent-ext-js` and added its span processor as a delegate. The optimizer logs a
|
|
347
|
+
warning if no OpenTelemetry metrics provider is registered.
|
|
348
|
+
- **Spans:** the optimizer's `chat <model>` spans are children of the CAP request spans and
|
|
349
|
+
carry token usage, savings and error category, never prompt text. They go through the same
|
|
350
|
+
pipeline as the CAP spans.
|
|
351
|
+
- **Not documented by SAP:** whether Cloud ALM shows **custom** OpenTelemetry metrics (like
|
|
352
|
+
`gen_ai.client.token.usage` or `cap_ai_optimizer.tokens.saved`) in its UIs. Its documented
|
|
353
|
+
use cases are Health, Exception, Job & Automation, Business Process and Real User Monitoring.
|
|
354
|
+
- **Not verified:** this setup hasn't been tested against a Cloud ALM tenant.
|
|
355
|
+
|
|
356
|
+
If you need token dashboards, use the `db` exporter with the admin service or your own
|
|
357
|
+
reporting service, or send the `otel` data to SAP Cloud Logging or Dynatrace via
|
|
358
|
+
`@cap-js/telemetry` (`cds.requires.telemetry.kind` = `to-cloud-logging` / `to-dynatrace`).
|
|
359
|
+
Note that with the agent present, `@cap-js/telemetry` also exports through its own
|
|
360
|
+
configured kind (console by default), so check the log volume after deploying.
|
|
361
|
+
|
|
362
|
+
## 10. Deploy and verify
|
|
363
|
+
|
|
364
|
+
```sh
|
|
365
|
+
npm install # updates package-lock.json
|
|
366
|
+
mbt build
|
|
367
|
+
cf deploy mta_archives/<your-app>_<version>.mtar -e secrets.mtaext
|
|
368
|
+
```
|
|
369
|
+
|
|
370
|
+
Then check:
|
|
371
|
+
|
|
372
|
+
1. **Logs:** `cf logs my-cap-app-srv --recent` shows
|
|
373
|
+
`autoInstrument: OrchestrationClient … calls are optimized and measured` at startup, and one
|
|
374
|
+
`"type":"ai-core-call"` line per AI call.
|
|
375
|
+
2. **Admin service:** with the `TokenOptimizerAdmin` role, open
|
|
376
|
+
`https://<app-url>/odata/v4/token-optimizer/summary()`.
|
|
377
|
+
3. **Cloud ALM:** the service appears under `SAP_CALM_SERVICE_NAME` in the activated
|
|
378
|
+
monitoring apps.
|
|
379
|
+
4. **Behavior:** your AI features answer exactly as before (shadow mode sends the original prompts).
|
|
380
|
+
|
|
381
|
+
## 11. Switch from shadow to active
|
|
382
|
+
|
|
383
|
+
After a few days in shadow mode, compare `savings.estimatedTokens` and the notes in the events
|
|
384
|
+
or the admin service per use case. Then switch one use case at a time:
|
|
385
|
+
|
|
386
|
+
```json
|
|
387
|
+
"tokenOptimizer": {
|
|
388
|
+
"mode": "shadow",
|
|
389
|
+
"useCases": {
|
|
390
|
+
"ticket-summary": { "mode": "active" }
|
|
391
|
+
}
|
|
392
|
+
}
|
|
393
|
+
```
|
|
394
|
+
|
|
395
|
+
In `active` mode the optimized prompts are sent, and identical calls with `temperature` ≤ 0.2
|
|
396
|
+
are answered from the cache, which is per app instance and per tenant. Check answer quality
|
|
397
|
+
for that use case before switching the next one. If an agent needs more context or tools,
|
|
398
|
+
raise `steps.history.maxTurns` or `steps.tools.maxTools`, or pin the tools it needs, for that
|
|
399
|
+
use case.
|
|
400
|
+
|
|
401
|
+
## 12. Turn it off or roll back
|
|
402
|
+
|
|
403
|
+
| Goal | How |
|
|
404
|
+
|---|---|
|
|
405
|
+
| Stop optimizing, keep measuring | `cds_requires_tokenOptimizer_mode=off`, then `cf restart` |
|
|
406
|
+
| Stop hooking the SDK | `cds_requires_tokenOptimizer_autoInstrument=false`, then `cf restart` |
|
|
407
|
+
| Turn off the Cloud ALM agent | `SAP_CALM_INSTRUMENTATION_ENABLED=false`, then `cf restart` |
|
|
408
|
+
| Remove the package | `npm rm @viveka10/cap-ai-token-optimizer`, delete the `tokenOptimizer` block, redeploy. If you used the `db` exporter, the `TokenUsageLog` table stays until you drop it. |
|
|
409
|
+
|
|
410
|
+
## Checklist
|
|
411
|
+
|
|
412
|
+
- [ ] Node ≥ 22.12 and `engines` set; `@sap-ai-sdk/*` ≥ 2.8, same version, no duplicates
|
|
413
|
+
- [ ] `npm add @viveka10/cap-ai-token-optimizer`
|
|
414
|
+
- [ ] `tokenOptimizer` block with `mode: "shadow"` and `autoInstrument: true`; `pricing` filled in
|
|
415
|
+
- [ ] Salt and other secrets in an `.mtaext` that is not committed
|
|
416
|
+
- [ ] Optional: `withUseCase` / `wrapLangChain` for named use cases
|
|
417
|
+
- [ ] Optional: `db` exporter (HANA table deployed with the app; MTX upgrade)
|
|
418
|
+
- [ ] Optional: admin service plus `TokenOptimizerAdmin` role collection
|
|
419
|
+
- [ ] Cloud ALM:
|
|
420
|
+
- [ ] RBSC user and `SAP_NPM_AUTH`
|
|
421
|
+
- [ ] agent ≥ 2.0.4 and `@cap-js/telemetry` ≥ 2.1.0
|
|
422
|
+
- [ ] `NODE_ARGS`, `SAP_CALM_*` properties and destination service in `mta.yaml`
|
|
423
|
+
- [ ] `CALM_datacollector` destination
|
|
424
|
+
- [ ] monitoring use cases activated in Cloud ALM
|
|
425
|
+
- [ ] `otel` exporter enabled
|
|
426
|
+
- [ ] Deploy, check logs and the admin service, then switch use cases to `active` one by one
|
|
@@ -130,7 +130,8 @@ for the same requests.
|
|
|
130
130
|
- **Scope:** the hooks apply to every caller in the process, including libraries. If you want
|
|
131
131
|
to choose exactly which calls are optimized, leave `autoInstrument` off and use `wrap()`,
|
|
132
132
|
`withUseCase()` and `wrapLangChain()` instead.
|
|
133
|
-
- **Streaming:** `stream()` calls are measured
|
|
133
|
+
- **Streaming:** `stream()` calls are measured, with token usage once the stream has been read
|
|
134
|
+
to the end, but not optimized.
|
|
134
135
|
- **Use cases:** without `withUseCase()`, all hooked calls share the use case set in
|
|
135
136
|
`autoInstrument.useCase` (default `default`), so per-use-case settings apply to all of them.
|
|
136
137
|
- **Start-up:** the hooks are installed when the CAP plugin loads. Outside CAP, call
|
|
@@ -7,6 +7,10 @@
|
|
|
7
7
|
* (orchestration: `{ error: { code, message, location } }`, Azure OpenAI: `{ error: { code, message, innererror } }`).
|
|
8
8
|
* Deployment-resolution and OAuth failures wrap further levels of `cause`. We therefore walk the
|
|
9
9
|
* whole cause chain (and AggregateError.errors) and collect every signal before deciding.
|
|
10
|
+
*
|
|
11
|
+
* Errors inside a stream are thrown by the SDK's SSE parser as
|
|
12
|
+
* `Error("Error received from the server.\n<JSON error>")`, wrapped in
|
|
13
|
+
* `ErrorWithCause("Error while iterating over SSE stream.")`; the JSON is read like a response body.
|
|
10
14
|
*/
|
|
11
15
|
|
|
12
16
|
const CATEGORIES = Object.freeze([
|
|
@@ -24,6 +28,7 @@ const RE_TIMEOUT = /timeout of \d+ms exceeded|timed? ?out|deadline exceeded/i
|
|
|
24
28
|
const RE_DEPLOYMENT = /no deployment matched|deployment not found|DeploymentNotFound|has no deployment url/i
|
|
25
29
|
const RE_AUTH = /unauthori[sz]ed|forbidden|invalid[_ ]client|could not fetch client credentials token/i
|
|
26
30
|
const RE_NETWORK = /socket hang up|network error|fetch failed|getaddrinfo|connect ECONN/i
|
|
31
|
+
const RE_SSE_ERROR = /^Error received from the server\.\n(\{[\s\S]*\})$/
|
|
27
32
|
|
|
28
33
|
const TIMEOUT_CODES = new Set(['ECONNABORTED', 'ETIMEDOUT', 'ESOCKETTIMEDOUT', 'ERR_TIMEOUT', 'UND_ERR_CONNECT_TIMEOUT', 'UND_ERR_HEADERS_TIMEOUT', 'UND_ERR_BODY_TIMEOUT'])
|
|
29
34
|
const NETWORK_CODES = new Set(['ECONNREFUSED', 'ECONNRESET', 'ENOTFOUND', 'EAI_AGAIN', 'EPIPE', 'EHOSTUNREACH', 'ENETUNREACH', 'ENETDOWN', 'ERR_NETWORK', 'UND_ERR_SOCKET'])
|
|
@@ -97,7 +102,7 @@ function collectSignals (root) {
|
|
|
97
102
|
if (typeof e.message === 'string' && e.message) s.texts.push(e.message)
|
|
98
103
|
s.retryAfter ??= header(e.response?.headers, 'retry-after') ?? header(e.headers, 'retry-after')
|
|
99
104
|
|
|
100
|
-
const body = parseBody(e.response?.data ?? e.body)
|
|
105
|
+
const body = parseBody(e.response?.data ?? e.body ?? sseErrorBody(e.message))
|
|
101
106
|
const errors = providerErrors(body)
|
|
102
107
|
// a non-JSON body (e.g. a gateway's "Payload Too Large") is the most useful message we have
|
|
103
108
|
if (!errors.length && typeof body?.message === 'string' && body.message) {
|
|
@@ -132,6 +137,13 @@ function providerErrors (body) {
|
|
|
132
137
|
return []
|
|
133
138
|
}
|
|
134
139
|
|
|
140
|
+
/** The `{ error }` body of an error event in a stream, from the SDK's error message. */
|
|
141
|
+
function sseErrorBody (message) {
|
|
142
|
+
const json = typeof message === 'string' ? RE_SSE_ERROR.exec(message)?.[1] : null
|
|
143
|
+
if (!json) return undefined
|
|
144
|
+
try { return { error: JSON.parse(json) } } catch { return undefined }
|
|
145
|
+
}
|
|
146
|
+
|
|
135
147
|
function parseBody (data) {
|
|
136
148
|
if (typeof data === 'string') {
|
|
137
149
|
try { return JSON.parse(data) } catch { return { message: data } }
|
package/lib/models.js
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
'use strict'
|
|
2
|
+
/**
|
|
3
|
+
* The CDS models this plugin adds to the app, depending on its configuration: the
|
|
4
|
+
* TokenUsageLog entity for the db exporter, the read-only admin service, and the UsageLog
|
|
5
|
+
* projection when both are on.
|
|
6
|
+
*/
|
|
7
|
+
const { name } = require('../package.json')
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* @param {object} [cfg] cds.requires.tokenOptimizer (raw, as in cds.env)
|
|
11
|
+
* @returns {string[]} model ids, resolved from the app root like `using from '<id>'`
|
|
12
|
+
*/
|
|
13
|
+
function modelsFor (cfg) {
|
|
14
|
+
const exporters = [].concat(cfg?.metrics?.exporters ?? [])
|
|
15
|
+
const dbLog = exporters.includes('db') || exporters.includes('hana')
|
|
16
|
+
const admin = cfg?.admin?.enabled === true
|
|
17
|
+
const models = []
|
|
18
|
+
if (dbLog) models.push(`${name}/db/token-usage-log`)
|
|
19
|
+
if (admin) models.push(`${name}/srv/admin-service`)
|
|
20
|
+
if (admin && dbLog) models.push(`${name}/srv/admin-usage-log`)
|
|
21
|
+
return models
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
module.exports = { modelsFor }
|
package/lib/optimizer.js
CHANGED
|
@@ -10,6 +10,10 @@
|
|
|
10
10
|
*
|
|
11
11
|
* Rule: nothing the optimizer does may break the AI call. If the pipeline fails, the original
|
|
12
12
|
* request is sent. Only errors thrown by the SDK itself reach the caller, unchanged.
|
|
13
|
+
*
|
|
14
|
+
* Streaming calls are recorded when the caller has read the stream to the end, so the event
|
|
15
|
+
* has AI Core's token usage, finish reason and the full latency. A stream that fails, is
|
|
16
|
+
* closed early or is not read within STREAM_RECORD_TIMEOUT_MS is recorded at that point.
|
|
13
17
|
*/
|
|
14
18
|
const { resolveConfig, forUseCase } = require('./config')
|
|
15
19
|
const { MetricsCollector } = require('./metrics/collector')
|
|
@@ -30,6 +34,7 @@ const WRAPPED = Symbol.for('cap-ai-token-optimizer.wrapped')
|
|
|
30
34
|
const LANGCHAIN_WRAPPED = Symbol.for('cap-ai-token-optimizer.langchain')
|
|
31
35
|
const MAX_PREFIXES = 1000
|
|
32
36
|
const ADVISORY_LOG_INTERVAL_MS = 10 * 60 * 1000
|
|
37
|
+
const STREAM_RECORD_TIMEOUT_MS = 10 * 60 * 1000
|
|
33
38
|
|
|
34
39
|
class TokenOptimizer {
|
|
35
40
|
/**
|
|
@@ -40,6 +45,7 @@ class TokenOptimizer {
|
|
|
40
45
|
this.collector = new MetricsCollector()
|
|
41
46
|
this._deps = deps
|
|
42
47
|
this._sleep = deps.sleep ?? ((ms) => new Promise(resolve => setTimeout(resolve, ms)))
|
|
48
|
+
this._streamRecordTimeoutMs = deps.streamRecordTimeoutMs ?? STREAM_RECORD_TIMEOUT_MS
|
|
43
49
|
this._configured = false
|
|
44
50
|
this.cache = null
|
|
45
51
|
// last prompt prefix (system + tools) per use case and model, for the prompt-caching advisory
|
|
@@ -253,10 +259,11 @@ class TokenOptimizer {
|
|
|
253
259
|
// the original method: a hooked one would route the call through an optimizer again
|
|
254
260
|
const response = await instrument.originalOf(plan.client[methodName]).apply(plan.client, callArgs)
|
|
255
261
|
if (ctx.cacheKey) this._cacheSet(ctx, response)
|
|
262
|
+
if (method.operation === 'stream' && safe(() => this._trackStream(ctx, { started, retries, response }), logRecordError)) return response
|
|
256
263
|
safe(() => this._record(ctx, { started, retries, response }), logRecordError)
|
|
257
264
|
return response
|
|
258
265
|
} catch (err) {
|
|
259
|
-
const classified = safe(() => classifyError(err)) ??
|
|
266
|
+
const classified = safe(() => classifyError(err)) ?? unknownError(err)
|
|
260
267
|
const delay = this._retryDelay(cfg, classified, retries)
|
|
261
268
|
if (delay != null) {
|
|
262
269
|
retries++
|
|
@@ -270,6 +277,44 @@ class TokenOptimizer {
|
|
|
270
277
|
}
|
|
271
278
|
}
|
|
272
279
|
|
|
280
|
+
/**
|
|
281
|
+
* Record a streaming call when its stream ends: replaces `response.stream` with a stream of
|
|
282
|
+
* the same class that passes every chunk through unchanged and records the event once, when
|
|
283
|
+
* the stream is read to the end (with AI Core's usage), fails (the error is classified and
|
|
284
|
+
* rethrown), is closed early by the caller, or is not read within the record timeout.
|
|
285
|
+
* @returns {boolean} false if the stream can't be tracked; the caller then records right away
|
|
286
|
+
*/
|
|
287
|
+
_trackStream (ctx, { started, retries, response }) {
|
|
288
|
+
const original = response?.stream
|
|
289
|
+
const Stream = original?.constructor
|
|
290
|
+
if (typeof original?.[Symbol.asyncIterator] !== 'function' || typeof Stream !== 'function' || Stream === Object) return false
|
|
291
|
+
|
|
292
|
+
let recorded = false
|
|
293
|
+
const record = (outcome = {}) => {
|
|
294
|
+
if (recorded) return
|
|
295
|
+
recorded = true
|
|
296
|
+
clearTimeout(timer)
|
|
297
|
+
safe(() => this._record(ctx, { started, retries, response, ...outcome }), logRecordError)
|
|
298
|
+
}
|
|
299
|
+
const timer = setTimeout(() => record({ note: 'stream: not read to the end within the record timeout' }), this._streamRecordTimeoutMs)
|
|
300
|
+
timer.unref?.()
|
|
301
|
+
|
|
302
|
+
async function * tracked () {
|
|
303
|
+
let ended = false
|
|
304
|
+
try {
|
|
305
|
+
for await (const chunk of original) yield chunk
|
|
306
|
+
ended = true
|
|
307
|
+
} catch (err) {
|
|
308
|
+
record({ error: safe(() => classifyError(err)) ?? unknownError(err) })
|
|
309
|
+
throw err
|
|
310
|
+
} finally {
|
|
311
|
+
record(ended ? {} : { note: 'stream: closed by the caller before the end' })
|
|
312
|
+
}
|
|
313
|
+
}
|
|
314
|
+
response.stream = new Stream(() => tracked(), original.controller)
|
|
315
|
+
return true
|
|
316
|
+
}
|
|
317
|
+
|
|
273
318
|
/**
|
|
274
319
|
* Everything known before the call: call info, estimates, pipeline result and cache key.
|
|
275
320
|
* Never throws: a failure leaves the call unoptimized.
|
|
@@ -375,7 +420,7 @@ class TokenOptimizer {
|
|
|
375
420
|
return Math.min(r.maxDelayMs, Math.round(backoff / 2 + Math.random() * backoff / 2))
|
|
376
421
|
}
|
|
377
422
|
|
|
378
|
-
_record (ctx, { started, retries, response, error, cacheEntry }) {
|
|
423
|
+
_record (ctx, { started, retries, response, error, cacheEntry, note }) {
|
|
379
424
|
const latencyMs = Math.round(performance.now() - started)
|
|
380
425
|
const cfg = ctx.cfg
|
|
381
426
|
const info = ctx.info ?? {}
|
|
@@ -398,6 +443,7 @@ class TokenOptimizer {
|
|
|
398
443
|
const model = info.model ?? null
|
|
399
444
|
const notes = [...(pipeline?.notes ?? [])]
|
|
400
445
|
if (ctx.pipelineFailed) notes.push('pipeline failed; original request sent')
|
|
446
|
+
if (note) notes.push(note)
|
|
401
447
|
const advisory = cacheEntry ? null : safe(() => this._checkPrefix(ctx))
|
|
402
448
|
|
|
403
449
|
const event = buildEvent({
|
|
@@ -453,6 +499,8 @@ function prefixOf (messages, tools) {
|
|
|
453
499
|
|
|
454
500
|
const logRecordError = (e) => log.warn('could not record call metrics:', e?.message)
|
|
455
501
|
|
|
502
|
+
const unknownError = (err) => ({ category: 'UNKNOWN', httpStatus: null, code: null, message: String(err?.message ?? err), retryAfterMs: null })
|
|
503
|
+
|
|
456
504
|
/**
|
|
457
505
|
* Run `fn`, returning undefined instead of throwing.
|
|
458
506
|
* @template T
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@viveka10/cap-ai-token-optimizer",
|
|
3
|
-
"version": "0.1.
|
|
4
|
-
"description": "
|
|
3
|
+
"version": "0.1.2",
|
|
4
|
+
"description": "SAP CAP plugin to reduce, measure and report SAP AI Core LLM token usage: prompt optimization, response cache, cost tracking, and metrics for SAP Cloud ALM / OpenTelemetry",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"sap",
|
|
7
7
|
"cap",
|
|
@@ -20,7 +20,17 @@
|
|
|
20
20
|
"mcp",
|
|
21
21
|
"agents",
|
|
22
22
|
"langchain",
|
|
23
|
-
"btp"
|
|
23
|
+
"btp",
|
|
24
|
+
"sap-cap",
|
|
25
|
+
"sap-btp",
|
|
26
|
+
"sap-ai",
|
|
27
|
+
"token-optimizer",
|
|
28
|
+
"ai-token-optimization",
|
|
29
|
+
"prompt-optimization",
|
|
30
|
+
"llm-cost",
|
|
31
|
+
"cost-optimization",
|
|
32
|
+
"openai",
|
|
33
|
+
"observability"
|
|
24
34
|
],
|
|
25
35
|
"homepage": "https://github.com/viveka10/npm_token_optimizer/tree/main/cap-ai-token-optimizer#readme",
|
|
26
36
|
"bugs": {
|
|
@@ -50,6 +60,7 @@
|
|
|
50
60
|
"index.js",
|
|
51
61
|
"index.d.ts",
|
|
52
62
|
"cds-plugin.js",
|
|
63
|
+
".cdsrc.js",
|
|
53
64
|
"lib/",
|
|
54
65
|
"srv/",
|
|
55
66
|
"db/",
|
package/srv/admin-service.js
CHANGED
|
@@ -13,11 +13,30 @@ const totals = (t) => ({
|
|
|
13
13
|
costEstimate: t.costEstimate
|
|
14
14
|
})
|
|
15
15
|
|
|
16
|
-
/**
|
|
16
|
+
/**
|
|
17
|
+
* OData exposes structured elements (`tokens`) as flat columns (`tokens_prompt`, …) unless the
|
|
18
|
+
* app uses structured OData (`odata.structs`, e.g. flavor x4). Return rows in the same shape.
|
|
19
|
+
*/
|
|
20
|
+
function shape (row) {
|
|
21
|
+
if (cds.env.effective?.odata?.structs || !row.tokens) return row
|
|
22
|
+
const { tokens, ...rest } = row
|
|
23
|
+
for (const [k, v] of Object.entries(tokens)) rest[`tokens_${k}`] = v
|
|
24
|
+
return rest
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* Rows for a READ, honouring a key lookup (e.g. UseCases('ask')). Lists carry `$count`, so
|
|
29
|
+
* `$count=true` (used by Fiori elements to size its tables) reports the number of rows.
|
|
30
|
+
*/
|
|
17
31
|
function rows (req, list, key) {
|
|
18
32
|
const wanted = req.data?.[key]
|
|
19
|
-
if (wanted
|
|
20
|
-
|
|
33
|
+
if (wanted !== undefined) {
|
|
34
|
+
const row = list.find(r => r[key] === wanted)
|
|
35
|
+
return row ? shape(row) : null
|
|
36
|
+
}
|
|
37
|
+
const result = list.map(shape)
|
|
38
|
+
result.$count = result.length
|
|
39
|
+
return result
|
|
21
40
|
}
|
|
22
41
|
|
|
23
42
|
module.exports = class TokenOptimizerService extends cds.ApplicationService {
|