@opensearch-project/agent-health 0.5.2 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -0
- package/cli/dist/index.js +1592 -620
- package/dist/assets/index-BfxtxmKc.css +1 -0
- package/dist/assets/index-CrjAfDHu.js +243 -0
- package/dist/index.html +2 -2
- package/docs/CLI.md +45 -2
- package/docs/CONFIGURATION.md +1 -1
- package/docs/INSTRUMENT_WITH_OTEL.md +7 -0
- package/docs/SDK.md +52 -3
- package/docs/STORAGE_INDEX_FIELD_LIMITS.md +216 -0
- package/docs/skills/AGENT_HEALTH.md +55 -1
- package/examples/eval-files/demo.eval.js +1 -1
- package/examples/eval-files/ops-rca-classification.eval.js +71 -0
- package/examples/eval-files/ops-rca-evaluator.json +15 -0
- package/examples/eval-files/sdk-demo.eval.js +72 -0
- package/examples/eval-files/sdk-describe-demo.eval.js +50 -0
- package/examples/eval-files/sdk-hooks-demo.eval.js +1 -1
- package/lib/dist/lib/bedrockCompat.d.ts +27 -0
- package/lib/dist/lib/bedrockCompat.d.ts.map +1 -0
- package/lib/dist/lib/bedrockCompat.js +83 -0
- package/lib/dist/lib/bedrockCompat.js.map +1 -0
- package/lib/dist/lib/benchmarkImage.d.ts +52 -0
- package/lib/dist/lib/benchmarkImage.d.ts.map +1 -0
- package/lib/dist/lib/benchmarkImage.js +113 -0
- package/lib/dist/lib/benchmarkImage.js.map +1 -0
- package/lib/dist/lib/benchmarkVersionUtils.d.ts +13 -0
- package/lib/dist/lib/benchmarkVersionUtils.d.ts.map +1 -1
- package/lib/dist/lib/benchmarkVersionUtils.js +21 -0
- package/lib/dist/lib/benchmarkVersionUtils.js.map +1 -1
- package/lib/dist/lib/chunkedFetch.d.ts +18 -0
- package/lib/dist/lib/chunkedFetch.d.ts.map +1 -0
- package/lib/dist/lib/chunkedFetch.js +40 -0
- package/lib/dist/lib/chunkedFetch.js.map +1 -0
- package/lib/dist/lib/comparisonInsights.d.ts +104 -0
- package/lib/dist/lib/comparisonInsights.d.ts.map +1 -0
- package/lib/dist/lib/comparisonInsights.js +212 -0
- package/lib/dist/lib/comparisonInsights.js.map +1 -0
- package/lib/dist/lib/config/loader.d.ts.map +1 -1
- package/lib/dist/lib/config/loader.js +5 -0
- package/lib/dist/lib/config/loader.js.map +1 -1
- package/lib/dist/lib/config/types.d.ts +14 -0
- package/lib/dist/lib/config/types.d.ts.map +1 -1
- package/lib/dist/lib/constants.d.ts +11 -0
- package/lib/dist/lib/constants.d.ts.map +1 -1
- package/lib/dist/lib/constants.js +10 -1
- package/lib/dist/lib/constants.js.map +1 -1
- package/lib/dist/lib/contextFormat.d.ts +26 -0
- package/lib/dist/lib/contextFormat.d.ts.map +1 -0
- package/lib/dist/lib/contextFormat.js +28 -0
- package/lib/dist/lib/contextFormat.js.map +1 -0
- package/lib/dist/lib/envCompat.d.ts.map +1 -1
- package/lib/dist/lib/envCompat.js +14 -5
- package/lib/dist/lib/envCompat.js.map +1 -1
- package/lib/dist/lib/evaluationRerun.d.ts +63 -0
- package/lib/dist/lib/evaluationRerun.d.ts.map +1 -0
- package/lib/dist/lib/evaluationRerun.js +85 -0
- package/lib/dist/lib/evaluationRerun.js.map +1 -0
- package/lib/dist/lib/matchers/traces.d.ts +17 -2
- package/lib/dist/lib/matchers/traces.d.ts.map +1 -1
- package/lib/dist/lib/matchers/traces.js +136 -16
- package/lib/dist/lib/matchers/traces.js.map +1 -1
- package/lib/dist/lib/matchers/tracesPricing.d.ts +38 -0
- package/lib/dist/lib/matchers/tracesPricing.d.ts.map +1 -0
- package/lib/dist/lib/matchers/tracesPricing.js +64 -0
- package/lib/dist/lib/matchers/tracesPricing.js.map +1 -0
- package/lib/dist/lib/runStats.d.ts +24 -0
- package/lib/dist/lib/runStats.d.ts.map +1 -1
- package/lib/dist/lib/runStats.js +32 -0
- package/lib/dist/lib/runStats.js.map +1 -1
- package/lib/dist/lib/testCases/loader.d.ts +11 -1
- package/lib/dist/lib/testCases/loader.d.ts.map +1 -1
- package/lib/dist/lib/testCases/loader.js +35 -5
- package/lib/dist/lib/testCases/loader.js.map +1 -1
- package/lib/dist/lib/utils.d.ts +15 -0
- package/lib/dist/lib/utils.d.ts.map +1 -1
- package/lib/dist/lib/utils.js +22 -0
- package/lib/dist/lib/utils.js.map +1 -1
- package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.d.ts +10 -0
- package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.d.ts.map +1 -1
- package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.js +17 -3
- package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.js.map +1 -1
- package/lib/dist/services/connectors/subprocess/SubprocessConnector.d.ts.map +1 -1
- package/lib/dist/services/connectors/subprocess/SubprocessConnector.js +8 -0
- package/lib/dist/services/connectors/subprocess/SubprocessConnector.js.map +1 -1
- package/lib/dist/services/evaluation/bedrockJudge.d.ts +7 -0
- package/lib/dist/services/evaluation/bedrockJudge.d.ts.map +1 -1
- package/lib/dist/services/evaluation/bedrockJudge.js +2 -0
- package/lib/dist/services/evaluation/bedrockJudge.js.map +1 -1
- package/lib/dist/services/storage/asyncBenchmarkStorage.d.ts +1 -1
- package/lib/dist/services/storage/asyncBenchmarkStorage.d.ts.map +1 -1
- package/lib/dist/services/storage/asyncBenchmarkStorage.js +14 -3
- package/lib/dist/services/storage/asyncBenchmarkStorage.js.map +1 -1
- package/lib/dist/services/storage/asyncRunStorage.d.ts +9 -1
- package/lib/dist/services/storage/asyncRunStorage.d.ts.map +1 -1
- package/lib/dist/services/storage/asyncRunStorage.js +77 -1
- package/lib/dist/services/storage/asyncRunStorage.js.map +1 -1
- package/lib/dist/services/storage/asyncTestCaseStorage.d.ts +18 -4
- package/lib/dist/services/storage/asyncTestCaseStorage.d.ts.map +1 -1
- package/lib/dist/services/storage/asyncTestCaseStorage.js +20 -4
- package/lib/dist/services/storage/asyncTestCaseStorage.js.map +1 -1
- package/lib/dist/services/storage/opensearchClient.d.ts +35 -12
- package/lib/dist/services/storage/opensearchClient.d.ts.map +1 -1
- package/lib/dist/services/storage/opensearchClient.js +12 -5
- package/lib/dist/services/storage/opensearchClient.js.map +1 -1
- package/lib/dist/services/traces/browserRecovery.d.ts +12 -1
- package/lib/dist/services/traces/browserRecovery.d.ts.map +1 -1
- package/lib/dist/services/traces/browserRecovery.js +32 -5
- package/lib/dist/services/traces/browserRecovery.js.map +1 -1
- package/lib/dist/services/traces/messageExtraction.d.ts.map +1 -1
- package/lib/dist/services/traces/messageExtraction.js +95 -33
- package/lib/dist/services/traces/messageExtraction.js.map +1 -1
- package/lib/dist/services/traces/spansToTrajectory.d.ts.map +1 -1
- package/lib/dist/services/traces/spansToTrajectory.js +47 -9
- package/lib/dist/services/traces/spansToTrajectory.js.map +1 -1
- package/lib/dist/services/traces/tracePoller.d.ts +27 -3
- package/lib/dist/services/traces/tracePoller.d.ts.map +1 -1
- package/lib/dist/services/traces/tracePoller.js +199 -33
- package/lib/dist/services/traces/tracePoller.js.map +1 -1
- package/lib/dist/types/index.d.ts +55 -1
- package/lib/dist/types/index.d.ts.map +1 -1
- package/lib/dist/types/index.js.map +1 -1
- package/package.json +3 -1
- package/server/dist/app.js +2238 -777
- package/server/dist/index.js +2238 -777
- package/dist/assets/index-CCQRDlO0.js +0 -243
- package/dist/assets/index-CNHQVbcj.css +0 -1
package/dist/index.html
CHANGED
|
@@ -13,8 +13,8 @@
|
|
|
13
13
|
<link rel="preconnect" href="https://fonts.googleapis.com">
|
|
14
14
|
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
|
|
15
15
|
<link href="https://fonts.googleapis.com/css2?family=Rubik:wght@300;400;500;600;700&family=Source+Code+Pro:wght@400;500;600&display=swap" rel="stylesheet">
|
|
16
|
-
<script type="module" crossorigin src="/assets/index-
|
|
17
|
-
<link rel="stylesheet" crossorigin href="/assets/index-
|
|
16
|
+
<script type="module" crossorigin src="/assets/index-CrjAfDHu.js"></script>
|
|
17
|
+
<link rel="stylesheet" crossorigin href="/assets/index-BfxtxmKc.css">
|
|
18
18
|
</head>
|
|
19
19
|
<body class="bg-background text-foreground antialiased h-full">
|
|
20
20
|
<div id="root" class="h-full"></div>
|
package/docs/CLI.md
CHANGED
|
@@ -61,11 +61,13 @@ agent-health list <resource> [-o table|json]
|
|
|
61
61
|
| `models` | | Available models |
|
|
62
62
|
| `test-cases` | `testcases`, `tc` | Stored test cases |
|
|
63
63
|
| `benchmarks` | `bench` | Stored benchmarks |
|
|
64
|
+
| `images` | `img` | Benchmark images (content-addressed evaluation-condition snapshots; runs sharing a digest are directly comparable) |
|
|
64
65
|
|
|
65
66
|
```bash
|
|
66
67
|
agent-health list agents
|
|
67
68
|
agent-health list tc -o json
|
|
68
69
|
agent-health list bench
|
|
70
|
+
agent-health list images
|
|
69
71
|
```
|
|
70
72
|
|
|
71
73
|
---
|
|
@@ -122,20 +124,61 @@ agent-health benchmark [options]
|
|
|
122
124
|
| `--stop-server` | Stop the server after benchmark completes | Keep running |
|
|
123
125
|
|
|
124
126
|
**Modes:**
|
|
125
|
-
- **Quick mode** (no `-n`, no `-f`):
|
|
127
|
+
- **Quick mode** (no `-n`, no `-f`): Runs all stored test cases as an **ad-hoc evaluation run** (no benchmark entity is created)
|
|
126
128
|
- **Named mode** (`-n <name>`): Runs a specific existing benchmark
|
|
127
129
|
- **File mode** (`-f <path>`): Imports test cases from a JSON file **or runs a code SDK file** (`.eval.js` / `.eval.ts` — see [SDK.md](./SDK.md)), creates a benchmark, and runs it
|
|
128
130
|
|
|
131
|
+
Every evaluation run is stamped with an **image digest** — a content hash of
|
|
132
|
+
its test-case contents + eval conditions (evaluator, judge model). Runs with
|
|
133
|
+
the same digest ran under identical conditions and are directly comparable;
|
|
134
|
+
re-running the same command converges on the same image instead of creating
|
|
135
|
+
new entities. See `agent-health list images`.
|
|
136
|
+
|
|
129
137
|
```bash
|
|
130
138
|
agent-health benchmark # quick mode
|
|
131
139
|
agent-health benchmark -n "Baseline" -a ml-commons # named mode
|
|
132
140
|
agent-health benchmark -f ./test-cases.json -a pulsar -v # file mode (JSON)
|
|
133
|
-
agent-health benchmark -f ./
|
|
141
|
+
agent-health benchmark -f ./examples/eval-files/demo.eval.js -a observio # file mode (code SDK)
|
|
134
142
|
agent-health benchmark -f ./test-cases.json -n "My Run" -a pulsar --export results.json
|
|
135
143
|
agent-health benchmark -n "Baseline" -e system-tool-usage -c 4 # custom evaluator, 4 in parallel
|
|
136
144
|
agent-health benchmark -n "Baseline" --export report.html --format html
|
|
137
145
|
```
|
|
138
146
|
|
|
147
|
+
#### benchmark doctor
|
|
148
|
+
|
|
149
|
+
Detect and clean up duplicated / debris benchmarks (dry-run by default).
|
|
150
|
+
|
|
151
|
+
```
|
|
152
|
+
agent-health benchmark doctor [--dry-run] [--apply] [--migrate-images] [--json]
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
| Option | Description |
|
|
156
|
+
|--------|-------------|
|
|
157
|
+
| `--dry-run` | Preview only — this is already the default; use --apply to execute |
|
|
158
|
+
| `--apply` | Execute the plan (default: dry-run report only) |
|
|
159
|
+
| `--migrate-images` | Convert remaining benchmarks into tagged benchmark images |
|
|
160
|
+
| `--json` | Output as JSON instead of the human-readable report |
|
|
161
|
+
|
|
162
|
+
What it detects:
|
|
163
|
+
- **Timestamped debris** — `quick-<ts>` / `*-<epoch-ms>` benchmarks with no
|
|
164
|
+
runs anywhere and older than 24h (deleted).
|
|
165
|
+
- **Content duplicates** — benchmarks with identical test-case sets. Embedded
|
|
166
|
+
runs are merged into the canonical (most runs › most references › oldest),
|
|
167
|
+
evaluation runs are re-pointed, duplicate shells deleted.
|
|
168
|
+
|
|
169
|
+
Runs and reports are **never** deleted. Sample data (`demo-*`) is never touched.
|
|
170
|
+
|
|
171
|
+
**Read-only mode**: When running without `--apply` or `--migrate-images` (pure dry-run),
|
|
172
|
+
the command safely reuses existing foreign servers in read-only mode. This allows
|
|
173
|
+
the diagnostic to run even when another agent-health instance is operating on a
|
|
174
|
+
different worktree/port, with a clear notice that no writes will be issued.
|
|
175
|
+
|
|
176
|
+
```bash
|
|
177
|
+
agent-health benchmark doctor # report what would change (read-only)
|
|
178
|
+
agent-health benchmark doctor --apply # clean up (strict server guard)
|
|
179
|
+
agent-health benchmark doctor --apply --migrate-images
|
|
180
|
+
```
|
|
181
|
+
|
|
139
182
|
---
|
|
140
183
|
|
|
141
184
|
### export
|
package/docs/CONFIGURATION.md
CHANGED
|
@@ -113,7 +113,7 @@ Rules of thumb:
|
|
|
113
113
|
`OPENSEARCH_STORAGE_*` or `storage` in your config to use a cluster instead).
|
|
114
114
|
- **`evals/`** is **your test source** — `.eval.js` / `.eval.ts` files you write
|
|
115
115
|
with the [code SDK](./SDK.md) and run with
|
|
116
|
-
`agent-health benchmark -f ./
|
|
116
|
+
`agent-health benchmark -f ./examples/eval-files/demo.eval.js`. They are **not** stored under
|
|
117
117
|
`.agent-health/data/`; *running* them produces run records that land there (or
|
|
118
118
|
in OpenSearch).
|
|
119
119
|
|
|
@@ -385,6 +385,13 @@ If you run an MCP server, expose this document as a resource at `agent-health://
|
|
|
385
385
|
|
|
386
386
|
## Reference
|
|
387
387
|
|
|
388
|
+
### New to OpenTelemetry?
|
|
389
|
+
|
|
390
|
+
- [What is OpenTelemetry?](https://opentelemetry.io/docs/what-is-opentelemetry/) — start here for the concepts (traces, spans, exporters, collectors).
|
|
391
|
+
- [OpenTelemetry GenAI semantic conventions (repo)](https://github.com/open-telemetry/semantic-conventions-genai) — the source-of-truth for the `gen_ai.*` attributes this guide uses.
|
|
392
|
+
|
|
393
|
+
### Specs used by this guide
|
|
394
|
+
|
|
388
395
|
- [OTel GenAI Semantic Conventions](https://opentelemetry.io/docs/specs/semconv/gen-ai/)
|
|
389
396
|
- [OTel GenAI Agent Spans](https://opentelemetry.io/docs/specs/semconv/gen-ai/gen-ai-agent-spans/)
|
|
390
397
|
- [Agent Health Telemetry Setup](./CLAUDE_CODE_TELEMETRY.md)
|
package/docs/SDK.md
CHANGED
|
@@ -166,7 +166,7 @@ Key points:
|
|
|
166
166
|
Hooks are a no-op when no test in the run uses them — the orchestrator
|
|
167
167
|
is short-circuited to a noop variant and existing tests pay zero cost.
|
|
168
168
|
|
|
169
|
-
See the demo at [`
|
|
169
|
+
See the demo at [`examples/eval-files/sdk-hooks-demo.eval.js`](../examples/eval-files/sdk-hooks-demo.eval.js).
|
|
170
170
|
|
|
171
171
|
### 5. Matchers record structured verdicts
|
|
172
172
|
|
|
@@ -483,8 +483,57 @@ enforced regardless of configuration.
|
|
|
483
483
|
|
|
484
484
|
---
|
|
485
485
|
|
|
486
|
+
## Weighted scoring across criteria
|
|
487
|
+
|
|
488
|
+
Each `expect(...)`, `judge(...)`, and `evaluate(...)` call records **one
|
|
489
|
+
per-matcher pass/fail** — they are not weighted against each other. When you
|
|
490
|
+
need a **single aggregate score** that weights criteria differently (e.g. root
|
|
491
|
+
cause 60%, SOP selection 20%, relevant-metrics check 10%, latency 10%), the
|
|
492
|
+
weighting lives in a **custom evaluator's `scoringConfig`**, not in the test
|
|
493
|
+
body. Define the weighted metrics once and attach the evaluator to the run via
|
|
494
|
+
`evaluatorId`:
|
|
495
|
+
|
|
496
|
+
```json
|
|
497
|
+
{
|
|
498
|
+
"name": "Ops RCA (weighted)",
|
|
499
|
+
"systemPrompt": "You evaluate an ops agent that triages a ticket. Score each metric 0-100. CRITICAL: root_cause_accuracy is the primary metric.",
|
|
500
|
+
"scoringConfig": {
|
|
501
|
+
"metrics": [
|
|
502
|
+
{ "name": "root_cause_accuracy", "weight": 0.6, "scale": 100 },
|
|
503
|
+
{ "name": "sop_selection", "weight": 0.2, "scale": 100 },
|
|
504
|
+
{ "name": "relevant_metrics", "weight": 0.1, "scale": 100 },
|
|
505
|
+
{ "name": "latency", "weight": 0.1, "scale": 100 }
|
|
506
|
+
],
|
|
507
|
+
"passThreshold": 80,
|
|
508
|
+
"scale": 100
|
|
509
|
+
}
|
|
510
|
+
}
|
|
511
|
+
```
|
|
512
|
+
|
|
513
|
+
The run's overall score is the weighted mean of the metrics the evaluator
|
|
514
|
+
emits, so you get one comparable number per run for progress tracking. Use the
|
|
515
|
+
deterministic `expect(...)` matchers for the hard checks (exact classification,
|
|
516
|
+
tool was called, budget cap) and the judge/evaluator + weighted `scoringConfig`
|
|
517
|
+
for the aggregate score. See the ops-RCA worked example at
|
|
518
|
+
[`examples/eval-files/ops-rca-classification.eval.js`](../examples/eval-files/ops-rca-classification.eval.js)
|
|
519
|
+
and its companion [`examples/eval-files/ops-rca-evaluator.json`](../examples/eval-files/ops-rca-evaluator.json),
|
|
520
|
+
and the evaluator reference in [docs/skills/AGENT_HEALTH.md](./skills/AGENT_HEALTH.md#custom-evaluators).
|
|
521
|
+
|
|
522
|
+
---
|
|
523
|
+
|
|
486
524
|
## Running the tests
|
|
487
525
|
|
|
526
|
+
> **Security note:** the ENTIRE contents of an imported `.eval.js`/`.eval.ts`
|
|
527
|
+
> file are persisted verbatim on the resulting test case (`sourceCode`) and
|
|
528
|
+
> rendered on the Test Case detail page as an IDE-style code view, so anyone
|
|
529
|
+
> who can view test cases in this deployment can read the full file --
|
|
530
|
+
> including any comments, hardcoded values, or internal URLs it contains.
|
|
531
|
+
> Treat eval files like any other source file that lands in your repo and
|
|
532
|
+
> gets deployed alongside the app: don't hardcode secrets, tokens, or
|
|
533
|
+
> customer data in them -- pull those from environment variables /
|
|
534
|
+
> `agent-health.config.ts` instead, same as you already would for the app's
|
|
535
|
+
> own credentials.
|
|
536
|
+
|
|
488
537
|
### Via the UI
|
|
489
538
|
|
|
490
539
|
`/evaluations/runs/new` → pick "Code import" → select your `.eval.js` files.
|
|
@@ -492,7 +541,7 @@ enforced regardless of configuration.
|
|
|
492
541
|
### Via the CLI
|
|
493
542
|
|
|
494
543
|
```bash
|
|
495
|
-
npx @opensearch-project/agent-health benchmark -f ./
|
|
544
|
+
npx @opensearch-project/agent-health benchmark -f ./examples/eval-files/demo.eval.js -a observio
|
|
496
545
|
```
|
|
497
546
|
|
|
498
547
|
### Via the HTTP API
|
|
@@ -504,7 +553,7 @@ curl -sN -X POST http://localhost:4001/api/storage/evaluation-runs \
|
|
|
504
553
|
"name": "Demo",
|
|
505
554
|
"sources": [{
|
|
506
555
|
"type": "code-import",
|
|
507
|
-
"filenames": ["
|
|
556
|
+
"filenames": ["examples/eval-files/demo.eval.js"],
|
|
508
557
|
"testCaseIds": []
|
|
509
558
|
}],
|
|
510
559
|
"agentKey": "observio",
|
|
@@ -0,0 +1,216 @@
|
|
|
1
|
+
# Storage: OpenSearch index field-limit growth (`evals_runs`)
|
|
2
|
+
|
|
3
|
+
## Incident
|
|
4
|
+
|
|
5
|
+
Owner-hit while running code-QA benchmarks: report/run persistence on the
|
|
6
|
+
shared cluster's `evals_runs` index failed with
|
|
7
|
+
|
|
8
|
+
```
|
|
9
|
+
illegal_argument_exception: Limit of total fields [5000] has been exceeded
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
Run **execution** succeeded — only the **write** errored out, i.e. data loss
|
|
13
|
+
(the completed report was never persisted).
|
|
14
|
+
|
|
15
|
+
This is the same *class* of bug PR #418 fixed for `evals_experiments`
|
|
16
|
+
(`EvaluationRun.results` / `testCaseSnapshots`): OpenSearch's default dynamic
|
|
17
|
+
mapping mints a new mapped field for every previously-unseen key under a
|
|
18
|
+
free-form object, and the field-count budget (`index.mapping.total_fields.limit`)
|
|
19
|
+
is **shared across every document in the index**, not per-document. #418 did
|
|
20
|
+
not cover `evals_runs` (the report/`TestCaseRun` index used by the code-SDK
|
|
21
|
+
path) — this fix does.
|
|
22
|
+
|
|
23
|
+
## Root cause: two unprotected growth vectors in `evals_runs`
|
|
24
|
+
|
|
25
|
+
Both are driven by the same source: `EvaluationMetrics`
|
|
26
|
+
(`types/index.ts`) is an open index signature (`[key: string]: number |
|
|
27
|
+
undefined`) by design — custom/system evaluators declare arbitrary metric
|
|
28
|
+
dimension names via `evaluator.scoringConfig.metrics`
|
|
29
|
+
(`server/services/judgeResponseParser.ts`'s `extractMetrics()`,
|
|
30
|
+
`services/storage/asyncRunStorage.ts`'s `storedMetricsToApp()` /
|
|
31
|
+
`toStorageFormat()` — see the comments in both, which explicitly call out
|
|
32
|
+
"preserve every metric the judge emitted, not just the four legacy keys").
|
|
33
|
+
Every *distinct* custom metric name, across every run/matcher ever written,
|
|
34
|
+
used to mint a brand-new mapped field, shared index-wide, forever.
|
|
35
|
+
|
|
36
|
+
| Field (in `evals_runs`) | Shape | Growth vector |
|
|
37
|
+
|---|---|---|
|
|
38
|
+
| `metrics` (report-level) | `Record<string, number>` | One set of dynamic names per run — one custom evaluator with N metric names adds ≤N new fields **the first time it's seen**, but a code-QA benchmark suite iterating on many custom evaluators over time accumulates without bound. |
|
|
39
|
+
| `matcherResults[].judgeMetrics` | `Record<string, number>`, nested inside a `nested`-typed array | Same growth, but **per SDK `judge()` call** — a single code-QA test case with many `expect`/`judge()` claims × many custom judge dimensions multiplies fast. This is the "code-SDK path" referenced in the incident — `matcherResults` is populated exclusively by the code-based test SDK (`docs/SDK.md`), not the legacy UI-driven runner. |
|
|
40
|
+
|
|
41
|
+
Everything else already flagged in the original bug report — matcher
|
|
42
|
+
`actual`/`expected`, `trajectory`, `logs`, `rawEvents`, `improvementStrategies`,
|
|
43
|
+
`spans` (span attributes) — was **already** `{ type: 'object', enabled: false
|
|
44
|
+
}` in `server/constants/indexMappings.ts` before this change (audited, not
|
|
45
|
+
touched). `llmJudgeResponse` (which itself has an open `extraFields`/
|
|
46
|
+
`parsedMetrics` shape) is **never persisted** to `evals_runs` at all
|
|
47
|
+
(`toStorageFormat()` doesn't include it) — confirmed via `git grep
|
|
48
|
+
llmJudgeResponse services/storage server/adapters`, no hits — so it isn't a
|
|
49
|
+
growth vector for this index either.
|
|
50
|
+
|
|
51
|
+
## Fix (mirrors #418's pattern)
|
|
52
|
+
|
|
53
|
+
`server/constants/indexMappings.ts`, `evals_runs` index:
|
|
54
|
+
|
|
55
|
+
```diff
|
|
56
|
+
metrics: {
|
|
57
|
+
+ dynamic: false,
|
|
58
|
+
properties: {
|
|
59
|
+
accuracy: { type: 'float' },
|
|
60
|
+
faithfulness: { type: 'float' },
|
|
61
|
+
latency_score: { type: 'float' },
|
|
62
|
+
trajectory_alignment_score: { type: 'float' },
|
|
63
|
+
},
|
|
64
|
+
},
|
|
65
|
+
...
|
|
66
|
+
matcherResults: {
|
|
67
|
+
type: 'nested',
|
|
68
|
+
properties: {
|
|
69
|
+
...
|
|
70
|
+
judgeMetrics: {
|
|
71
|
+
+ dynamic: false,
|
|
72
|
+
properties: {
|
|
73
|
+
accuracy: { type: 'float' },
|
|
74
|
+
faithfulness: { type: 'float' },
|
|
75
|
+
latency_score: { type: 'float' },
|
|
76
|
+
trajectory_alignment_score: { type: 'float' },
|
|
77
|
+
},
|
|
78
|
+
},
|
|
79
|
+
},
|
|
80
|
+
},
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
Unlike #418's `results`/`testCaseSnapshots` (`enabled: false`, fully opaque),
|
|
84
|
+
this uses `dynamic: false` **with explicit typed sub-properties** for the
|
|
85
|
+
four legacy metric names — they stay real, typed, queryable fields (nothing
|
|
86
|
+
queries them today — see the audit below — but it's free to keep them typed),
|
|
87
|
+
while every *other* metric/dimension name is stored in `_source` (readable,
|
|
88
|
+
unaffected) but never added to the mapping. `_source` is unaffected either
|
|
89
|
+
way — the choice between `enabled:false` and `dynamic:false` only changes
|
|
90
|
+
what OpenSearch can filter/sort/aggregate on, never what's persisted or
|
|
91
|
+
returned.
|
|
92
|
+
|
|
93
|
+
## Query audit — nothing queried becomes unsearchable
|
|
94
|
+
|
|
95
|
+
Every OpenSearch-level query/filter/sort/aggregation against `evals_runs`
|
|
96
|
+
(`server/adapters/opensearch/StorageModule.ts`'s `OpenSearchRunOperations`)
|
|
97
|
+
was enumerated. None touch `metrics.*` or `matcherResults[].judgeMetrics.*`
|
|
98
|
+
beyond the four legacy names, which stay mapped:
|
|
99
|
+
|
|
100
|
+
| Consumer | Query | Fields used | Affected by this fix? |
|
|
101
|
+
|---|---|---|---|
|
|
102
|
+
| `OpenSearchRunOperations.search()` | `term` filters | `experimentId`, `experimentRunId`, `testCaseId`, `agentId`, `modelId`, `status`, `passFailStatus` | No — untouched, still explicit `keyword` fields |
|
|
103
|
+
| `OpenSearchRunOperations.search()` | `range` filter | `createdAt` | No — untouched, still `date` |
|
|
104
|
+
| `OpenSearchRunOperations.getAll()` / `.search()` | `sort` | `createdAt` | No |
|
|
105
|
+
| `OpenSearchRunOperations.countsByTestCase()` | `terms` agg | `testCaseId` | No |
|
|
106
|
+
| `asyncRunStorage.ts` `SearchQuery.minAccuracy` | **application-level** `Array.filter()`, not an OpenSearch query (`reports.filter(r => r.metrics.accuracy >= ...)`) | `metrics.accuracy` (read from `_source` in JS) | No — reads the value out of `_source`, which is unaffected by `dynamic: false`. If this were ever converted to a server-side `range` query, it would still work: `accuracy` stays an explicitly mapped, queryable field. |
|
|
107
|
+
| UI (`MatcherResultsPanel.tsx`, `JudgeSection.tsx`, `RunDetailsContent.tsx`) | none — reads `matcherResults`/`judgeMetrics` out of the fetched JSON document, never issues its own OpenSearch query | n/a | No |
|
|
108
|
+
| `services/evaluation/index.ts`, `services/benchmarkRunner.ts`, `services/hookOrchestrator.ts` | none — same, in-process consumption of the already-fetched report | n/a | No |
|
|
109
|
+
|
|
110
|
+
Conclusion: **no consumer anywhere issues an OpenSearch-side query against a
|
|
111
|
+
non-legacy `metrics.*` or `judgeMetrics.*` name.** Both are read back via
|
|
112
|
+
`_source` wherever consumed (search, list, comparison, UI). This mirrors
|
|
113
|
+
exactly the trade-off #418 already made and documented for
|
|
114
|
+
`EvaluationRun.results`.
|
|
115
|
+
|
|
116
|
+
## Migration story — what to run, exactly
|
|
117
|
+
|
|
118
|
+
**Nothing runs automatically against the live cluster from this PR.**
|
|
119
|
+
|
|
120
|
+
### New / fresh indexes
|
|
121
|
+
|
|
122
|
+
No action needed. `ensureIndexes()` (`server/services/indexInitializer.ts`,
|
|
123
|
+
called on every server boot and on "attach new cluster") creates any missing
|
|
124
|
+
index straight from the updated `INDEX_MAPPINGS` — new deployments and any
|
|
125
|
+
environment that doesn't have `evals_runs` yet get the fix immediately.
|
|
126
|
+
|
|
127
|
+
### Existing, NOT-YET-poisoned `evals_runs` (most environments)
|
|
128
|
+
|
|
129
|
+
Also no action needed, but not immediate — `ensureIndexes()` also calls
|
|
130
|
+
`client.indices.putMapping()` on every boot for existing indexes, which is
|
|
131
|
+
how the `dynamic: false` fix reaches an already-existing-but-clean index: it
|
|
132
|
+
succeeds silently and the index is protected from the next write onward.
|
|
133
|
+
|
|
134
|
+
### The shared cluster's `evals_runs`, if already poisoned
|
|
135
|
+
|
|
136
|
+
If any code-QA benchmark run already wrote a custom evaluator metric name to
|
|
137
|
+
the shared cluster before this fix ships, `evals_runs.metrics` (and/or
|
|
138
|
+
`matcherResults.judgeMetrics`) already has real, dynamically-inferred
|
|
139
|
+
sub-properties. OpenSearch's `putMapping` **rejects** an `enabled`/`dynamic`
|
|
140
|
+
change on a field that already has sub-properties
|
|
141
|
+
(`mapper_exception: the [dynamic] parameter can't be updated for the object
|
|
142
|
+
mapping [metrics]`) — `ensureIndexes()` catches this, logs a warning, and
|
|
143
|
+
otherwise no-ops (no crash, no data loss, same as #418's documented
|
|
144
|
+
`mapper_exception` handling for `results`). **The index keeps growing** until
|
|
145
|
+
an explicit reindex is run.
|
|
146
|
+
|
|
147
|
+
**This is not new migration code** — the existing generic reindex mechanism
|
|
148
|
+
(`reindexSingleIndex()`, `server/services/mappingFixer.ts`, already shipped
|
|
149
|
+
and already exposed at `POST /api/storage/reindex`, `server/routes/storage/admin.ts`)
|
|
150
|
+
already recreates any `INDEX_MAPPINGS`-registered index from scratch and
|
|
151
|
+
copies every document across, which sheds a poisoned mapping's dynamically-
|
|
152
|
+
inferred sub-fields while preserving 100% of the underlying `_source` data
|
|
153
|
+
(proven in `tests/integration/services/storage/evalsRunsMappingMigrationRecipe.integration.test.ts`,
|
|
154
|
+
run against a real OpenSearch container with a deliberately-poisoned index).
|
|
155
|
+
|
|
156
|
+
**The owner's exact recipe, when ready to run it against the shared cluster:**
|
|
157
|
+
|
|
158
|
+
```bash
|
|
159
|
+
# 1. Confirm the target index actually needs it (optional sanity check):
|
|
160
|
+
curl -s -X GET "$OPENSEARCH_STORAGE_ENDPOINT/evals_runs/_mapping" \
|
|
161
|
+
-u "$OPENSEARCH_STORAGE_USERNAME:$OPENSEARCH_STORAGE_PASSWORD" \
|
|
162
|
+
| jq '.evals_runs.mappings.properties.metrics'
|
|
163
|
+
# If this prints a `dynamic` key, it's already fixed. If it prints only
|
|
164
|
+
# `properties` with more than the 4 legacy metric names, it's poisoned.
|
|
165
|
+
|
|
166
|
+
# 2. Run the reindex via the running server's admin API (recreates the
|
|
167
|
+
# index from the current INDEX_MAPPINGS and copies every document
|
|
168
|
+
# across; the same doc-count-validated, recovery-safe path
|
|
169
|
+
# reindexSingleIndex() has always used for keyword-type mismatch fixes):
|
|
170
|
+
curl -s -X POST "http://localhost:4001/api/storage/reindex" \
|
|
171
|
+
-H 'Content-Type: application/json' \
|
|
172
|
+
-d '{"index": "evals_runs"}'
|
|
173
|
+
```
|
|
174
|
+
|
|
175
|
+
Caveats to read before running this against the shared cluster:
|
|
176
|
+
|
|
177
|
+
- **No write lock during a manual `/api/storage/reindex` call.** The
|
|
178
|
+
auto-fix boot path (`fixIndexMappings()`) acquires a process-local
|
|
179
|
+
migration lock around the reindex; the manual admin route calls
|
|
180
|
+
`reindexSingleIndex()` directly and does **not** (pre-existing gap in
|
|
181
|
+
`server/routes/storage/admin.ts`, not introduced by this PR — flagged here,
|
|
182
|
+
not fixed, since it's out of scope for this change). Run it during a quiet
|
|
183
|
+
window (no in-flight evaluation runs writing reports) to avoid a write
|
|
184
|
+
racing the index delete/recreate step.
|
|
185
|
+
- It touches only `evals_runs`. The already-known-poisoned `evals_experiments`
|
|
186
|
+
(800+ stale `results.*` fields, per the incident notes) uses the identical
|
|
187
|
+
recipe (`{"index": "evals_experiments"}`) — that cleanup is separately
|
|
188
|
+
planned by ops; this PR does not touch or schedule it.
|
|
189
|
+
- Document count is validated before the temporary index is deleted; if the
|
|
190
|
+
copy-back count doesn't match, the error message names the surviving temp
|
|
191
|
+
index (`evals_runs_reindex_temp`) for manual recovery — nothing is deleted
|
|
192
|
+
until the counts are confirmed equal.
|
|
193
|
+
|
|
194
|
+
## Tests
|
|
195
|
+
|
|
196
|
+
- Unit (`tests/unit/server/constants/indexMappings.test.ts`): mapping-shape
|
|
197
|
+
assertions — `dynamic: false` + typed legacy properties on both `metrics`
|
|
198
|
+
and `matcherResults.judgeMetrics`; pre-existing `enabled:false` fields stay
|
|
199
|
+
disabled; every field `OpenSearchRunOperations.search()` queries stays
|
|
200
|
+
explicitly mapped.
|
|
201
|
+
- Integration, real OpenSearch
|
|
202
|
+
(`tests/integration/services/storage/testCaseRunMetricsMappingGrowth.integration.test.ts`):
|
|
203
|
+
writes one report with 1000+ distinct custom `metrics`/`judgeMetrics` names
|
|
204
|
+
(500 report-level + 125 `judge()` calls × 4 dimensions), asserts it
|
|
205
|
+
round-trips correctly and the index's total mapped-field count does not
|
|
206
|
+
grow; asserts the query-audit fields stay queryable.
|
|
207
|
+
- Integration, real OpenSearch, migration recipe
|
|
208
|
+
(`tests/integration/services/storage/evalsRunsMappingMigrationRecipe.integration.test.ts`):
|
|
209
|
+
deliberately poisons a throwaway index the old way, runs the *existing*
|
|
210
|
+
`reindexSingleIndex()`, asserts the mapping resets to `dynamic: false` and
|
|
211
|
+
all document data survives byte-for-byte.
|
|
212
|
+
|
|
213
|
+
Both integration suites skip gracefully (with a console warning) if no
|
|
214
|
+
OpenSearch cluster is reachable at `TEST_OPENSEARCH_ENDPOINT` (default
|
|
215
|
+
`http://localhost:9200`) — the unit suite covers the mapping-shape assertions
|
|
216
|
+
unconditionally.
|
|
@@ -167,7 +167,7 @@ run a JSON file:
|
|
|
167
167
|
|
|
168
168
|
```bash
|
|
169
169
|
# `-f` accepts BOTH JSON test-case files and code SDK (.eval.js / .eval.ts) files
|
|
170
|
-
npx @opensearch-project/agent-health benchmark -f ./
|
|
170
|
+
npx @opensearch-project/agent-health benchmark -f ./examples/eval-files/demo.eval.js -a my-agent
|
|
171
171
|
```
|
|
172
172
|
|
|
173
173
|
They produce **per-matcher results** (`matcherResults[]`) instead of a single
|
|
@@ -303,6 +303,60 @@ Repeat until all high-priority issues are resolved.
|
|
|
303
303
|
| Server / config issues | `npx agent-health doctor` (checks config + connectivity) |
|
|
304
304
|
| "OpenSearch storage not configured" | Fine for local use — file-based storage is the default. Set `OPENSEARCH_STORAGE_*` only for shared / production persistence. |
|
|
305
305
|
|
|
306
|
+
> **Note (benchmark execution vs. reads):** file-based storage covers reads,
|
|
307
|
+
> single `run`s, and sample data, but **executing a multi-case `benchmark`
|
|
308
|
+
> currently requires an OpenSearch storage client** — without one the execute
|
|
309
|
+
> path returns `Cannot execute in sample-only mode`. Point `OPENSEARCH_STORAGE_*`
|
|
310
|
+
> at a cluster (a local security-disabled Docker OpenSearch on plain HTTP with
|
|
311
|
+
> `authType=none` is enough) before running `benchmark`.
|
|
312
|
+
|
|
313
|
+
---
|
|
314
|
+
|
|
315
|
+
## Benchmarking a local (subprocess) agent — gotchas
|
|
316
|
+
|
|
317
|
+
When you wrap a local CLI agent (Kiro, Claude Code, Pi, or your own script) as a
|
|
318
|
+
**subprocess** agent and run a `benchmark`, these are the traps that bite first
|
|
319
|
+
— check them before blaming the agent:
|
|
320
|
+
|
|
321
|
+
1. **Pin a real judge model — don't leave it on `demo`.** The `demo` provider is
|
|
322
|
+
a mock judge that returns high pass rates without calling an LLM, so a run
|
|
323
|
+
can look like "100% pass" while nothing was actually judged. Set the run's
|
|
324
|
+
judge model to a real Bedrock model, e.g.
|
|
325
|
+
`us.anthropic.claude-sonnet-4-5-20250929-v1:0`, and confirm it's invocable in
|
|
326
|
+
your account (`aws bedrock ... ` / `doctor`). If your pass rate looks too
|
|
327
|
+
good, check the judge provider first.
|
|
328
|
+
2. **Raise the subprocess timeout for slow agents.** The subprocess connector
|
|
329
|
+
defaults to a 5-minute (`300000` ms) timeout. A real ops/RCA agent can run
|
|
330
|
+
~10 min. Set it in your agent's `connectorConfig`:
|
|
331
|
+
```ts
|
|
332
|
+
{ key: 'my-ops-agent', connectorType: 'subprocess',
|
|
333
|
+
connectorConfig: { timeout: 1200000 /* 20 min */, /* ... */ } }
|
|
334
|
+
```
|
|
335
|
+
3. **Fail loud on a wrong agent name.** If your wrapper points at an agent key
|
|
336
|
+
that doesn't exist, the underlying CLI may silently fall back to a default
|
|
337
|
+
agent — so you benchmark the wrong thing. Echo the resolved agent name in
|
|
338
|
+
your wrapper and eyeball the first trajectory.
|
|
339
|
+
4. **Long runs + CLI SSE disconnects.** On runs longer than a few minutes the
|
|
340
|
+
CLI's streaming connection can drop and report `0/0` / `fetch failed` **while
|
|
341
|
+
the server keeps going**. The results are still persisted — read them back
|
|
342
|
+
from storage (`list runs` / the UI run inspector / `--export`) rather than
|
|
343
|
+
trusting the CLI summary.
|
|
344
|
+
5. **Watch for a stale server on the port.** If you patch/upgrade the package
|
|
345
|
+
but a previously-started `npx` server is still bound to port 4001, your runs
|
|
346
|
+
are served by the old code. Confirm which process owns the port
|
|
347
|
+
(`lsof -i :4001` / a `/proc` sweep) and restart the one you actually patched
|
|
348
|
+
(the server loads `server/dist/app.js`, not `index.js`).
|
|
349
|
+
6. **Judge `CredentialsProviderError` after a session rotates.** If your AWS
|
|
350
|
+
sandbox session rotates, `AWS_SHARED_CREDENTIALS_FILE` can point at a stale/
|
|
351
|
+
empty file and the judge fails. Re-vend credentials and restart the server
|
|
352
|
+
with the AWS env explicit (`aws sts get-caller-identity` to confirm first).
|
|
353
|
+
7. **Local OpenSearch dying (exit 255) = memory pressure.** Give the container
|
|
354
|
+
enough heap and run it with `--restart=unless-stopped`.
|
|
355
|
+
8. **Case schema is strict.** Test cases need `expectedOutcomes` (a
|
|
356
|
+
**string[]**, not a singular `expectedOutcome` string) and a capitalized
|
|
357
|
+
`difficulty` (`Easy` | `Medium` | `Hard`). Validate your converter output
|
|
358
|
+
against a known-good case before importing 16 of them.
|
|
359
|
+
|
|
306
360
|
---
|
|
307
361
|
|
|
308
362
|
## Server API Reference
|
|
@@ -18,7 +18,7 @@
|
|
|
18
18
|
* -H 'Content-Type: application/json' \
|
|
19
19
|
* -d '{
|
|
20
20
|
* "name":"SDK Demo",
|
|
21
|
-
* "sources":[{"type":"code-import","filenames":["
|
|
21
|
+
* "sources":[{"type":"code-import","filenames":["examples/eval-files/demo.eval.js"],"testCaseIds":[]}],
|
|
22
22
|
* "agentKey":"observio",
|
|
23
23
|
* "modelId":"claude-sonnet"
|
|
24
24
|
* }'
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Copyright OpenSearch Contributors
|
|
3
|
+
* SPDX-License-Identifier: Apache-2.0
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* Worked example — evaluating an ops / RCA agent that triages a ticket.
|
|
8
|
+
*
|
|
9
|
+
* The agent under test reads a ticket and emits a structured classification:
|
|
10
|
+
* - ticketType: "latency" | "fault" (deterministic — exact match)
|
|
11
|
+
* - rootCause: one of a fixed category set (deterministic — exact match)
|
|
12
|
+
* - sop: a recommended runbook (non-deterministic — LLM judge)
|
|
13
|
+
*
|
|
14
|
+
* This shows the two check styles side by side:
|
|
15
|
+
* • deterministic → expect(...).to.equal(...) on the parsed output
|
|
16
|
+
* • non-deterministic → judge(result, '<natural-language claim>')
|
|
17
|
+
*
|
|
18
|
+
* WEIGHTED SCORING (root-cause 60%, SOP, metrics, latency 10%):
|
|
19
|
+
* per-matcher pass/fail lives here, but the *single weighted aggregate score*
|
|
20
|
+
* across criteria is defined once in a custom EVALUATOR, not in the test body.
|
|
21
|
+
* Attach it at run time with `evaluatorId` — see ops-rca-evaluator.json next to
|
|
22
|
+
* this file. Run:
|
|
23
|
+
*
|
|
24
|
+
* npx @opensearch-project/agent-health benchmark \
|
|
25
|
+
* -f ./examples/eval-files/ops-rca-classification.eval.js -a my-ops-agent
|
|
26
|
+
*
|
|
27
|
+
* Docs: ../../docs/SDK.md Instrumentation: ../../docs/INSTRUMENT_WITH_OTEL.md
|
|
28
|
+
*/
|
|
29
|
+
|
|
30
|
+
const { test, expect } = require('@opensearch-project/agent-health');
|
|
31
|
+
|
|
32
|
+
// The agent's allowed root-cause categories — deterministic ground truth.
|
|
33
|
+
const ROOT_CAUSE_CATEGORIES = [
|
|
34
|
+
'dependency_outage',
|
|
35
|
+
'resource_exhaustion',
|
|
36
|
+
'config_error',
|
|
37
|
+
'code_regression',
|
|
38
|
+
'network',
|
|
39
|
+
];
|
|
40
|
+
|
|
41
|
+
test('ticket-123-db-outage', {
|
|
42
|
+
prompt: 'Triage ticket TICKET-123 and return your classification table.',
|
|
43
|
+
description: 'DB dependency outage — must classify as fault + dependency_outage',
|
|
44
|
+
context: [
|
|
45
|
+
{
|
|
46
|
+
description: 'Ticket body',
|
|
47
|
+
value:
|
|
48
|
+
'TICKET-123: payment-service returning 500s since 10:30. ' +
|
|
49
|
+
'Logs: "Connection refused to database-primary:5432". p99 latency normal until errors began.',
|
|
50
|
+
},
|
|
51
|
+
],
|
|
52
|
+
labels: ['category:RCA', 'difficulty:Medium', 'agent:ops', 'type:fault'],
|
|
53
|
+
}, async function ({ agent, judge }) {
|
|
54
|
+
const result = await agent.run();
|
|
55
|
+
|
|
56
|
+
// ── Deterministic checks — exact classification, no LLM, $0 ──────────────
|
|
57
|
+
const out = result.parsedOutput() || {}; // agent emits JSON classification
|
|
58
|
+
expect(out.ticketType).to.equal('fault');
|
|
59
|
+
expect(ROOT_CAUSE_CATEGORIES).to.include(out.rootCause);
|
|
60
|
+
expect(out.rootCause).to.equal('dependency_outage');
|
|
61
|
+
|
|
62
|
+
// Prove it actually investigated rather than guessing.
|
|
63
|
+
expect(result.trajectory).to.haveStepsOfType('action');
|
|
64
|
+
|
|
65
|
+
// ── Non-deterministic checks — LLM judge on the free-text SOP ────────────
|
|
66
|
+
await judge(result, 'Recommends a runbook appropriate for a database dependency outage');
|
|
67
|
+
await judge(result, 'Explains that payment-service cannot reach database-primary as the root cause');
|
|
68
|
+
|
|
69
|
+
// ── Budget guard (feeds the "latency 10%" weight via the evaluator) ──────
|
|
70
|
+
expect(result).to.haveCompletedWithin(120_000);
|
|
71
|
+
});
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "Ops RCA (weighted)",
|
|
3
|
+
"description": "Weighted scoring for a ticket-triage / RCA ops agent. Root-cause accuracy dominates; latency is a small tie-breaker. Attach to a run via evaluatorId.",
|
|
4
|
+
"systemPrompt": "You are evaluating an ops agent that triages a support ticket. The agent must (1) classify the ticket type (latency vs fault), (2) identify the root cause from a fixed category set, (3) recommend an appropriate SOP/runbook, and (4) check the relevant metrics. Score each metric from 0 to 100.\n\nCRITICAL CRITERIA:\n- root_cause_accuracy is the PRIMARY metric. A wrong root cause is a failed triage regardless of everything else.\n- sop_selection: did the agent recommend a runbook appropriate for the identified root cause?\n- relevant_metrics: did the agent inspect the metrics/logs that actually matter for this failure mode?\n- latency: score higher when the agent reaches a correct answer in fewer steps / less wall-clock time.\n\nReturn pass_fail_status, reasoning, a metrics object with the four numeric scores, and improvement_strategies.",
|
|
5
|
+
"scoringConfig": {
|
|
6
|
+
"metrics": [
|
|
7
|
+
{ "name": "root_cause_accuracy", "weight": 0.6, "scale": 100 },
|
|
8
|
+
{ "name": "sop_selection", "weight": 0.2, "scale": 100 },
|
|
9
|
+
{ "name": "relevant_metrics", "weight": 0.1, "scale": 100 },
|
|
10
|
+
{ "name": "latency", "weight": 0.1, "scale": 100 }
|
|
11
|
+
],
|
|
12
|
+
"passThreshold": 80,
|
|
13
|
+
"scale": 100
|
|
14
|
+
}
|
|
15
|
+
}
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Copyright OpenSearch Contributors
|
|
3
|
+
* SPDX-License-Identifier: Apache-2.0
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* SDK demo eval — three test cases that show the spectrum of evaluation
|
|
8
|
+
* methods the code-based SDK supports:
|
|
9
|
+
*
|
|
10
|
+
* 1. mock-says-hello (deterministic) — agent invoked, only chai matchers
|
|
11
|
+
* 2. mock-rca-judged (agentic) — agent invoked + LLM judge matcher
|
|
12
|
+
* 3. data-only-no-prompt (deterministic) — no agent call at all
|
|
13
|
+
*
|
|
14
|
+
* Run with:
|
|
15
|
+
* AH_PORT=4002 npx @opensearch-project/agent-health benchmark \
|
|
16
|
+
* -f examples/eval-files/sdk-demo.eval.js -a demo
|
|
17
|
+
*/
|
|
18
|
+
|
|
19
|
+
const { test, expect } = require('@opensearch-project/agent-health');
|
|
20
|
+
|
|
21
|
+
// ─────────────────────────────────────────────────────────────────────────────
|
|
22
|
+
// 1. Deterministic — agent runs, all assertions are local chai matchers
|
|
23
|
+
// ─────────────────────────────────────────────────────────────────────────────
|
|
24
|
+
|
|
25
|
+
test('mock-says-hello', {
|
|
26
|
+
prompt: 'Say hello in one short sentence.',
|
|
27
|
+
description: 'Mock agent must produce a non-empty response within 30s',
|
|
28
|
+
labels: ['category:Smoke', 'difficulty:Easy', 'method:deterministic'],
|
|
29
|
+
}, async function ({ agent }) {
|
|
30
|
+
const result = await agent.run();
|
|
31
|
+
expect(result.trajectory).to.have.length.greaterThan(0);
|
|
32
|
+
expect(result.agentOutput.trim()).to.have.length.greaterThan(0);
|
|
33
|
+
expect(result).to.haveCompletedWithin(30_000);
|
|
34
|
+
});
|
|
35
|
+
|
|
36
|
+
// ─────────────────────────────────────────────────────────────────────────────
|
|
37
|
+
// 2. Agentic (hybrid) — deterministic preflight + LLM judge for semantic claim
|
|
38
|
+
// ─────────────────────────────────────────────────────────────────────────────
|
|
39
|
+
|
|
40
|
+
test('mock-rca-judged', {
|
|
41
|
+
prompt: 'Diagnose why the payment service is failing and explain the root cause.',
|
|
42
|
+
description: 'Hybrid: structural checks first, then LLM judge for semantic correctness',
|
|
43
|
+
context: [
|
|
44
|
+
{
|
|
45
|
+
description: 'Error log',
|
|
46
|
+
value: 'ERROR 2026-05-20 10:31:22 [payment-service] Connection refused to db-primary:5432',
|
|
47
|
+
},
|
|
48
|
+
],
|
|
49
|
+
labels: ['category:RCA', 'difficulty:Medium', 'method:agentic'],
|
|
50
|
+
}, async function ({ agent, judge }) {
|
|
51
|
+
const result = await agent.run();
|
|
52
|
+
|
|
53
|
+
// Cheap deterministic preflight — fail fast before spending $ on the judge
|
|
54
|
+
expect(result.trajectory).to.have.length.greaterThan(0);
|
|
55
|
+
expect(result).to.haveCompletedWithin(60_000);
|
|
56
|
+
|
|
57
|
+
// LLM judge — produces a structured matcher verdict with score + reasoning
|
|
58
|
+
await judge(result, 'Mentions the payment service or its database connection failure');
|
|
59
|
+
});
|
|
60
|
+
|
|
61
|
+
// ─────────────────────────────────────────────────────────────────────────────
|
|
62
|
+
// 3. Deterministic, no prompt — agent never invoked, $0 / 0ms agent step
|
|
63
|
+
// ─────────────────────────────────────────────────────────────────────────────
|
|
64
|
+
|
|
65
|
+
test('data-only-no-prompt', {
|
|
66
|
+
description: 'Pure data check; agent invocation skipped entirely',
|
|
67
|
+
labels: ['category:Data Quality', 'difficulty:Easy', 'method:deterministic'],
|
|
68
|
+
}, function ({ result }) {
|
|
69
|
+
expect(result.durationMs).to.equal(0);
|
|
70
|
+
expect(result.trajectory).to.have.length(0);
|
|
71
|
+
expect(2 + 2).to.equal(4);
|
|
72
|
+
});
|