@opensearch-project/agent-health 0.5.2 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (126) hide show
  1. package/README.md +1 -0
  2. package/cli/dist/index.js +1592 -620
  3. package/dist/assets/index-BfxtxmKc.css +1 -0
  4. package/dist/assets/index-CrjAfDHu.js +243 -0
  5. package/dist/index.html +2 -2
  6. package/docs/CLI.md +45 -2
  7. package/docs/CONFIGURATION.md +1 -1
  8. package/docs/INSTRUMENT_WITH_OTEL.md +7 -0
  9. package/docs/SDK.md +52 -3
  10. package/docs/STORAGE_INDEX_FIELD_LIMITS.md +216 -0
  11. package/docs/skills/AGENT_HEALTH.md +55 -1
  12. package/examples/eval-files/demo.eval.js +1 -1
  13. package/examples/eval-files/ops-rca-classification.eval.js +71 -0
  14. package/examples/eval-files/ops-rca-evaluator.json +15 -0
  15. package/examples/eval-files/sdk-demo.eval.js +72 -0
  16. package/examples/eval-files/sdk-describe-demo.eval.js +50 -0
  17. package/examples/eval-files/sdk-hooks-demo.eval.js +1 -1
  18. package/lib/dist/lib/bedrockCompat.d.ts +27 -0
  19. package/lib/dist/lib/bedrockCompat.d.ts.map +1 -0
  20. package/lib/dist/lib/bedrockCompat.js +83 -0
  21. package/lib/dist/lib/bedrockCompat.js.map +1 -0
  22. package/lib/dist/lib/benchmarkImage.d.ts +52 -0
  23. package/lib/dist/lib/benchmarkImage.d.ts.map +1 -0
  24. package/lib/dist/lib/benchmarkImage.js +113 -0
  25. package/lib/dist/lib/benchmarkImage.js.map +1 -0
  26. package/lib/dist/lib/benchmarkVersionUtils.d.ts +13 -0
  27. package/lib/dist/lib/benchmarkVersionUtils.d.ts.map +1 -1
  28. package/lib/dist/lib/benchmarkVersionUtils.js +21 -0
  29. package/lib/dist/lib/benchmarkVersionUtils.js.map +1 -1
  30. package/lib/dist/lib/chunkedFetch.d.ts +18 -0
  31. package/lib/dist/lib/chunkedFetch.d.ts.map +1 -0
  32. package/lib/dist/lib/chunkedFetch.js +40 -0
  33. package/lib/dist/lib/chunkedFetch.js.map +1 -0
  34. package/lib/dist/lib/comparisonInsights.d.ts +104 -0
  35. package/lib/dist/lib/comparisonInsights.d.ts.map +1 -0
  36. package/lib/dist/lib/comparisonInsights.js +212 -0
  37. package/lib/dist/lib/comparisonInsights.js.map +1 -0
  38. package/lib/dist/lib/config/loader.d.ts.map +1 -1
  39. package/lib/dist/lib/config/loader.js +5 -0
  40. package/lib/dist/lib/config/loader.js.map +1 -1
  41. package/lib/dist/lib/config/types.d.ts +14 -0
  42. package/lib/dist/lib/config/types.d.ts.map +1 -1
  43. package/lib/dist/lib/constants.d.ts +11 -0
  44. package/lib/dist/lib/constants.d.ts.map +1 -1
  45. package/lib/dist/lib/constants.js +10 -1
  46. package/lib/dist/lib/constants.js.map +1 -1
  47. package/lib/dist/lib/contextFormat.d.ts +26 -0
  48. package/lib/dist/lib/contextFormat.d.ts.map +1 -0
  49. package/lib/dist/lib/contextFormat.js +28 -0
  50. package/lib/dist/lib/contextFormat.js.map +1 -0
  51. package/lib/dist/lib/envCompat.d.ts.map +1 -1
  52. package/lib/dist/lib/envCompat.js +14 -5
  53. package/lib/dist/lib/envCompat.js.map +1 -1
  54. package/lib/dist/lib/evaluationRerun.d.ts +63 -0
  55. package/lib/dist/lib/evaluationRerun.d.ts.map +1 -0
  56. package/lib/dist/lib/evaluationRerun.js +85 -0
  57. package/lib/dist/lib/evaluationRerun.js.map +1 -0
  58. package/lib/dist/lib/matchers/traces.d.ts +17 -2
  59. package/lib/dist/lib/matchers/traces.d.ts.map +1 -1
  60. package/lib/dist/lib/matchers/traces.js +136 -16
  61. package/lib/dist/lib/matchers/traces.js.map +1 -1
  62. package/lib/dist/lib/matchers/tracesPricing.d.ts +38 -0
  63. package/lib/dist/lib/matchers/tracesPricing.d.ts.map +1 -0
  64. package/lib/dist/lib/matchers/tracesPricing.js +64 -0
  65. package/lib/dist/lib/matchers/tracesPricing.js.map +1 -0
  66. package/lib/dist/lib/runStats.d.ts +24 -0
  67. package/lib/dist/lib/runStats.d.ts.map +1 -1
  68. package/lib/dist/lib/runStats.js +32 -0
  69. package/lib/dist/lib/runStats.js.map +1 -1
  70. package/lib/dist/lib/testCases/loader.d.ts +11 -1
  71. package/lib/dist/lib/testCases/loader.d.ts.map +1 -1
  72. package/lib/dist/lib/testCases/loader.js +35 -5
  73. package/lib/dist/lib/testCases/loader.js.map +1 -1
  74. package/lib/dist/lib/utils.d.ts +15 -0
  75. package/lib/dist/lib/utils.d.ts.map +1 -1
  76. package/lib/dist/lib/utils.js +22 -0
  77. package/lib/dist/lib/utils.js.map +1 -1
  78. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.d.ts +10 -0
  79. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.d.ts.map +1 -1
  80. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.js +17 -3
  81. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.js.map +1 -1
  82. package/lib/dist/services/connectors/subprocess/SubprocessConnector.d.ts.map +1 -1
  83. package/lib/dist/services/connectors/subprocess/SubprocessConnector.js +8 -0
  84. package/lib/dist/services/connectors/subprocess/SubprocessConnector.js.map +1 -1
  85. package/lib/dist/services/evaluation/bedrockJudge.d.ts +7 -0
  86. package/lib/dist/services/evaluation/bedrockJudge.d.ts.map +1 -1
  87. package/lib/dist/services/evaluation/bedrockJudge.js +2 -0
  88. package/lib/dist/services/evaluation/bedrockJudge.js.map +1 -1
  89. package/lib/dist/services/storage/asyncBenchmarkStorage.d.ts +1 -1
  90. package/lib/dist/services/storage/asyncBenchmarkStorage.d.ts.map +1 -1
  91. package/lib/dist/services/storage/asyncBenchmarkStorage.js +14 -3
  92. package/lib/dist/services/storage/asyncBenchmarkStorage.js.map +1 -1
  93. package/lib/dist/services/storage/asyncRunStorage.d.ts +9 -1
  94. package/lib/dist/services/storage/asyncRunStorage.d.ts.map +1 -1
  95. package/lib/dist/services/storage/asyncRunStorage.js +77 -1
  96. package/lib/dist/services/storage/asyncRunStorage.js.map +1 -1
  97. package/lib/dist/services/storage/asyncTestCaseStorage.d.ts +18 -4
  98. package/lib/dist/services/storage/asyncTestCaseStorage.d.ts.map +1 -1
  99. package/lib/dist/services/storage/asyncTestCaseStorage.js +20 -4
  100. package/lib/dist/services/storage/asyncTestCaseStorage.js.map +1 -1
  101. package/lib/dist/services/storage/opensearchClient.d.ts +35 -12
  102. package/lib/dist/services/storage/opensearchClient.d.ts.map +1 -1
  103. package/lib/dist/services/storage/opensearchClient.js +12 -5
  104. package/lib/dist/services/storage/opensearchClient.js.map +1 -1
  105. package/lib/dist/services/traces/browserRecovery.d.ts +12 -1
  106. package/lib/dist/services/traces/browserRecovery.d.ts.map +1 -1
  107. package/lib/dist/services/traces/browserRecovery.js +32 -5
  108. package/lib/dist/services/traces/browserRecovery.js.map +1 -1
  109. package/lib/dist/services/traces/messageExtraction.d.ts.map +1 -1
  110. package/lib/dist/services/traces/messageExtraction.js +95 -33
  111. package/lib/dist/services/traces/messageExtraction.js.map +1 -1
  112. package/lib/dist/services/traces/spansToTrajectory.d.ts.map +1 -1
  113. package/lib/dist/services/traces/spansToTrajectory.js +47 -9
  114. package/lib/dist/services/traces/spansToTrajectory.js.map +1 -1
  115. package/lib/dist/services/traces/tracePoller.d.ts +27 -3
  116. package/lib/dist/services/traces/tracePoller.d.ts.map +1 -1
  117. package/lib/dist/services/traces/tracePoller.js +199 -33
  118. package/lib/dist/services/traces/tracePoller.js.map +1 -1
  119. package/lib/dist/types/index.d.ts +55 -1
  120. package/lib/dist/types/index.d.ts.map +1 -1
  121. package/lib/dist/types/index.js.map +1 -1
  122. package/package.json +3 -1
  123. package/server/dist/app.js +2238 -777
  124. package/server/dist/index.js +2238 -777
  125. package/dist/assets/index-CCQRDlO0.js +0 -243
  126. package/dist/assets/index-CNHQVbcj.css +0 -1
package/dist/index.html CHANGED
@@ -13,8 +13,8 @@
13
13
  <link rel="preconnect" href="https://fonts.googleapis.com">
14
14
  <link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
15
15
  <link href="https://fonts.googleapis.com/css2?family=Rubik:wght@300;400;500;600;700&family=Source+Code+Pro:wght@400;500;600&display=swap" rel="stylesheet">
16
- <script type="module" crossorigin src="/assets/index-CCQRDlO0.js"></script>
17
- <link rel="stylesheet" crossorigin href="/assets/index-CNHQVbcj.css">
16
+ <script type="module" crossorigin src="/assets/index-CrjAfDHu.js"></script>
17
+ <link rel="stylesheet" crossorigin href="/assets/index-BfxtxmKc.css">
18
18
  </head>
19
19
  <body class="bg-background text-foreground antialiased h-full">
20
20
  <div id="root" class="h-full"></div>
package/docs/CLI.md CHANGED
@@ -61,11 +61,13 @@ agent-health list <resource> [-o table|json]
61
61
  | `models` | | Available models |
62
62
  | `test-cases` | `testcases`, `tc` | Stored test cases |
63
63
  | `benchmarks` | `bench` | Stored benchmarks |
64
+ | `images` | `img` | Benchmark images (content-addressed evaluation-condition snapshots; runs sharing a digest are directly comparable) |
64
65
 
65
66
  ```bash
66
67
  agent-health list agents
67
68
  agent-health list tc -o json
68
69
  agent-health list bench
70
+ agent-health list images
69
71
  ```
70
72
 
71
73
  ---
@@ -122,20 +124,61 @@ agent-health benchmark [options]
122
124
  | `--stop-server` | Stop the server after benchmark completes | Keep running |
123
125
 
124
126
  **Modes:**
125
- - **Quick mode** (no `-n`, no `-f`): Auto-creates a benchmark from all stored test cases
127
+ - **Quick mode** (no `-n`, no `-f`): Runs all stored test cases as an **ad-hoc evaluation run** (no benchmark entity is created)
126
128
  - **Named mode** (`-n <name>`): Runs a specific existing benchmark
127
129
  - **File mode** (`-f <path>`): Imports test cases from a JSON file **or runs a code SDK file** (`.eval.js` / `.eval.ts` — see [SDK.md](./SDK.md)), creates a benchmark, and runs it
128
130
 
131
+ Every evaluation run is stamped with an **image digest** — a content hash of
132
+ its test-case contents + eval conditions (evaluator, judge model). Runs with
133
+ the same digest ran under identical conditions and are directly comparable;
134
+ re-running the same command converges on the same image instead of creating
135
+ new entities. See `agent-health list images`.
136
+
129
137
  ```bash
130
138
  agent-health benchmark # quick mode
131
139
  agent-health benchmark -n "Baseline" -a ml-commons # named mode
132
140
  agent-health benchmark -f ./test-cases.json -a pulsar -v # file mode (JSON)
133
- agent-health benchmark -f ./evals/demo.eval.js -a observio # file mode (code SDK)
141
+ agent-health benchmark -f ./examples/eval-files/demo.eval.js -a observio # file mode (code SDK)
134
142
  agent-health benchmark -f ./test-cases.json -n "My Run" -a pulsar --export results.json
135
143
  agent-health benchmark -n "Baseline" -e system-tool-usage -c 4 # custom evaluator, 4 in parallel
136
144
  agent-health benchmark -n "Baseline" --export report.html --format html
137
145
  ```
138
146
 
147
+ #### benchmark doctor
148
+
149
+ Detect and clean up duplicated / debris benchmarks (dry-run by default).
150
+
151
+ ```
152
+ agent-health benchmark doctor [--dry-run] [--apply] [--migrate-images] [--json]
153
+ ```
154
+
155
+ | Option | Description |
156
+ |--------|-------------|
157
+ | `--dry-run` | Preview only — this is already the default; use --apply to execute |
158
+ | `--apply` | Execute the plan (default: dry-run report only) |
159
+ | `--migrate-images` | Convert remaining benchmarks into tagged benchmark images |
160
+ | `--json` | Output as JSON instead of the human-readable report |
161
+
162
+ What it detects:
163
+ - **Timestamped debris** — `quick-<ts>` / `*-<epoch-ms>` benchmarks with no
164
+ runs anywhere and older than 24h (deleted).
165
+ - **Content duplicates** — benchmarks with identical test-case sets. Embedded
166
+ runs are merged into the canonical (most runs › most references › oldest),
167
+ evaluation runs are re-pointed, duplicate shells deleted.
168
+
169
+ Runs and reports are **never** deleted. Sample data (`demo-*`) is never touched.
170
+
171
+ **Read-only mode**: When running without `--apply` or `--migrate-images` (pure dry-run),
172
+ the command safely reuses existing foreign servers in read-only mode. This allows
173
+ the diagnostic to run even when another agent-health instance is operating on a
174
+ different worktree/port, with a clear notice that no writes will be issued.
175
+
176
+ ```bash
177
+ agent-health benchmark doctor # report what would change (read-only)
178
+ agent-health benchmark doctor --apply # clean up (strict server guard)
179
+ agent-health benchmark doctor --apply --migrate-images
180
+ ```
181
+
139
182
  ---
140
183
 
141
184
  ### export
@@ -113,7 +113,7 @@ Rules of thumb:
113
113
  `OPENSEARCH_STORAGE_*` or `storage` in your config to use a cluster instead).
114
114
  - **`evals/`** is **your test source** — `.eval.js` / `.eval.ts` files you write
115
115
  with the [code SDK](./SDK.md) and run with
116
- `agent-health benchmark -f ./evals/demo.eval.js`. They are **not** stored under
116
+ `agent-health benchmark -f ./examples/eval-files/demo.eval.js`. They are **not** stored under
117
117
  `.agent-health/data/`; *running* them produces run records that land there (or
118
118
  in OpenSearch).
119
119
 
@@ -385,6 +385,13 @@ If you run an MCP server, expose this document as a resource at `agent-health://
385
385
 
386
386
  ## Reference
387
387
 
388
+ ### New to OpenTelemetry?
389
+
390
+ - [What is OpenTelemetry?](https://opentelemetry.io/docs/what-is-opentelemetry/) — start here for the concepts (traces, spans, exporters, collectors).
391
+ - [OpenTelemetry GenAI semantic conventions (repo)](https://github.com/open-telemetry/semantic-conventions-genai) — the source-of-truth for the `gen_ai.*` attributes this guide uses.
392
+
393
+ ### Specs used by this guide
394
+
388
395
  - [OTel GenAI Semantic Conventions](https://opentelemetry.io/docs/specs/semconv/gen-ai/)
389
396
  - [OTel GenAI Agent Spans](https://opentelemetry.io/docs/specs/semconv/gen-ai/gen-ai-agent-spans/)
390
397
  - [Agent Health Telemetry Setup](./CLAUDE_CODE_TELEMETRY.md)
package/docs/SDK.md CHANGED
@@ -166,7 +166,7 @@ Key points:
166
166
  Hooks are a no-op when no test in the run uses them — the orchestrator
167
167
  is short-circuited to a noop variant and existing tests pay zero cost.
168
168
 
169
- See the demo at [`evals/sdk-hooks-demo.eval.js`](../evals/sdk-hooks-demo.eval.js).
169
+ See the demo at [`examples/eval-files/sdk-hooks-demo.eval.js`](../examples/eval-files/sdk-hooks-demo.eval.js).
170
170
 
171
171
  ### 5. Matchers record structured verdicts
172
172
 
@@ -483,8 +483,57 @@ enforced regardless of configuration.
483
483
 
484
484
  ---
485
485
 
486
+ ## Weighted scoring across criteria
487
+
488
+ Each `expect(...)`, `judge(...)`, and `evaluate(...)` call records **one
489
+ per-matcher pass/fail** — they are not weighted against each other. When you
490
+ need a **single aggregate score** that weights criteria differently (e.g. root
491
+ cause 60%, SOP selection 20%, relevant-metrics check 10%, latency 10%), the
492
+ weighting lives in a **custom evaluator's `scoringConfig`**, not in the test
493
+ body. Define the weighted metrics once and attach the evaluator to the run via
494
+ `evaluatorId`:
495
+
496
+ ```json
497
+ {
498
+ "name": "Ops RCA (weighted)",
499
+ "systemPrompt": "You evaluate an ops agent that triages a ticket. Score each metric 0-100. CRITICAL: root_cause_accuracy is the primary metric.",
500
+ "scoringConfig": {
501
+ "metrics": [
502
+ { "name": "root_cause_accuracy", "weight": 0.6, "scale": 100 },
503
+ { "name": "sop_selection", "weight": 0.2, "scale": 100 },
504
+ { "name": "relevant_metrics", "weight": 0.1, "scale": 100 },
505
+ { "name": "latency", "weight": 0.1, "scale": 100 }
506
+ ],
507
+ "passThreshold": 80,
508
+ "scale": 100
509
+ }
510
+ }
511
+ ```
512
+
513
+ The run's overall score is the weighted mean of the metrics the evaluator
514
+ emits, so you get one comparable number per run for progress tracking. Use the
515
+ deterministic `expect(...)` matchers for the hard checks (exact classification,
516
+ tool was called, budget cap) and the judge/evaluator + weighted `scoringConfig`
517
+ for the aggregate score. See the ops-RCA worked example at
518
+ [`examples/eval-files/ops-rca-classification.eval.js`](../examples/eval-files/ops-rca-classification.eval.js)
519
+ and its companion [`examples/eval-files/ops-rca-evaluator.json`](../examples/eval-files/ops-rca-evaluator.json),
520
+ and the evaluator reference in [docs/skills/AGENT_HEALTH.md](./skills/AGENT_HEALTH.md#custom-evaluators).
521
+
522
+ ---
523
+
486
524
  ## Running the tests
487
525
 
526
+ > **Security note:** the ENTIRE contents of an imported `.eval.js`/`.eval.ts`
527
+ > file are persisted verbatim on the resulting test case (`sourceCode`) and
528
+ > rendered on the Test Case detail page as an IDE-style code view, so anyone
529
+ > who can view test cases in this deployment can read the full file --
530
+ > including any comments, hardcoded values, or internal URLs it contains.
531
+ > Treat eval files like any other source file that lands in your repo and
532
+ > gets deployed alongside the app: don't hardcode secrets, tokens, or
533
+ > customer data in them -- pull those from environment variables /
534
+ > `agent-health.config.ts` instead, same as you already would for the app's
535
+ > own credentials.
536
+
488
537
  ### Via the UI
489
538
 
490
539
  `/evaluations/runs/new` → pick "Code import" → select your `.eval.js` files.
@@ -492,7 +541,7 @@ enforced regardless of configuration.
492
541
  ### Via the CLI
493
542
 
494
543
  ```bash
495
- npx @opensearch-project/agent-health benchmark -f ./evals/demo.eval.js -a observio
544
+ npx @opensearch-project/agent-health benchmark -f ./examples/eval-files/demo.eval.js -a observio
496
545
  ```
497
546
 
498
547
  ### Via the HTTP API
@@ -504,7 +553,7 @@ curl -sN -X POST http://localhost:4001/api/storage/evaluation-runs \
504
553
  "name": "Demo",
505
554
  "sources": [{
506
555
  "type": "code-import",
507
- "filenames": ["evals/demo.eval.js"],
556
+ "filenames": ["examples/eval-files/demo.eval.js"],
508
557
  "testCaseIds": []
509
558
  }],
510
559
  "agentKey": "observio",
@@ -0,0 +1,216 @@
1
+ # Storage: OpenSearch index field-limit growth (`evals_runs`)
2
+
3
+ ## Incident
4
+
5
+ Owner-hit while running code-QA benchmarks: report/run persistence on the
6
+ shared cluster's `evals_runs` index failed with
7
+
8
+ ```
9
+ illegal_argument_exception: Limit of total fields [5000] has been exceeded
10
+ ```
11
+
12
+ Run **execution** succeeded — only the **write** errored out, i.e. data loss
13
+ (the completed report was never persisted).
14
+
15
+ This is the same *class* of bug PR #418 fixed for `evals_experiments`
16
+ (`EvaluationRun.results` / `testCaseSnapshots`): OpenSearch's default dynamic
17
+ mapping mints a new mapped field for every previously-unseen key under a
18
+ free-form object, and the field-count budget (`index.mapping.total_fields.limit`)
19
+ is **shared across every document in the index**, not per-document. #418 did
20
+ not cover `evals_runs` (the report/`TestCaseRun` index used by the code-SDK
21
+ path) — this fix does.
22
+
23
+ ## Root cause: two unprotected growth vectors in `evals_runs`
24
+
25
+ Both are driven by the same source: `EvaluationMetrics`
26
+ (`types/index.ts`) is an open index signature (`[key: string]: number |
27
+ undefined`) by design — custom/system evaluators declare arbitrary metric
28
+ dimension names via `evaluator.scoringConfig.metrics`
29
+ (`server/services/judgeResponseParser.ts`'s `extractMetrics()`,
30
+ `services/storage/asyncRunStorage.ts`'s `storedMetricsToApp()` /
31
+ `toStorageFormat()` — see the comments in both, which explicitly call out
32
+ "preserve every metric the judge emitted, not just the four legacy keys").
33
+ Every *distinct* custom metric name, across every run/matcher ever written,
34
+ used to mint a brand-new mapped field, shared index-wide, forever.
35
+
36
+ | Field (in `evals_runs`) | Shape | Growth vector |
37
+ |---|---|---|
38
+ | `metrics` (report-level) | `Record<string, number>` | One set of dynamic names per run — one custom evaluator with N metric names adds ≤N new fields **the first time it's seen**, but a code-QA benchmark suite iterating on many custom evaluators over time accumulates without bound. |
39
+ | `matcherResults[].judgeMetrics` | `Record<string, number>`, nested inside a `nested`-typed array | Same growth, but **per SDK `judge()` call** — a single code-QA test case with many `expect`/`judge()` claims × many custom judge dimensions multiplies fast. This is the "code-SDK path" referenced in the incident — `matcherResults` is populated exclusively by the code-based test SDK (`docs/SDK.md`), not the legacy UI-driven runner. |
40
+
41
+ Everything else already flagged in the original bug report — matcher
42
+ `actual`/`expected`, `trajectory`, `logs`, `rawEvents`, `improvementStrategies`,
43
+ `spans` (span attributes) — was **already** `{ type: 'object', enabled: false
44
+ }` in `server/constants/indexMappings.ts` before this change (audited, not
45
+ touched). `llmJudgeResponse` (which itself has an open `extraFields`/
46
+ `parsedMetrics` shape) is **never persisted** to `evals_runs` at all
47
+ (`toStorageFormat()` doesn't include it) — confirmed via `git grep
48
+ llmJudgeResponse services/storage server/adapters`, no hits — so it isn't a
49
+ growth vector for this index either.
50
+
51
+ ## Fix (mirrors #418's pattern)
52
+
53
+ `server/constants/indexMappings.ts`, `evals_runs` index:
54
+
55
+ ```diff
56
+ metrics: {
57
+ + dynamic: false,
58
+ properties: {
59
+ accuracy: { type: 'float' },
60
+ faithfulness: { type: 'float' },
61
+ latency_score: { type: 'float' },
62
+ trajectory_alignment_score: { type: 'float' },
63
+ },
64
+ },
65
+ ...
66
+ matcherResults: {
67
+ type: 'nested',
68
+ properties: {
69
+ ...
70
+ judgeMetrics: {
71
+ + dynamic: false,
72
+ properties: {
73
+ accuracy: { type: 'float' },
74
+ faithfulness: { type: 'float' },
75
+ latency_score: { type: 'float' },
76
+ trajectory_alignment_score: { type: 'float' },
77
+ },
78
+ },
79
+ },
80
+ },
81
+ ```
82
+
83
+ Unlike #418's `results`/`testCaseSnapshots` (`enabled: false`, fully opaque),
84
+ this uses `dynamic: false` **with explicit typed sub-properties** for the
85
+ four legacy metric names — they stay real, typed, queryable fields (nothing
86
+ queries them today — see the audit below — but it's free to keep them typed),
87
+ while every *other* metric/dimension name is stored in `_source` (readable,
88
+ unaffected) but never added to the mapping. `_source` is unaffected either
89
+ way — the choice between `enabled:false` and `dynamic:false` only changes
90
+ what OpenSearch can filter/sort/aggregate on, never what's persisted or
91
+ returned.
92
+
93
+ ## Query audit — nothing queried becomes unsearchable
94
+
95
+ Every OpenSearch-level query/filter/sort/aggregation against `evals_runs`
96
+ (`server/adapters/opensearch/StorageModule.ts`'s `OpenSearchRunOperations`)
97
+ was enumerated. None touch `metrics.*` or `matcherResults[].judgeMetrics.*`
98
+ beyond the four legacy names, which stay mapped:
99
+
100
+ | Consumer | Query | Fields used | Affected by this fix? |
101
+ |---|---|---|---|
102
+ | `OpenSearchRunOperations.search()` | `term` filters | `experimentId`, `experimentRunId`, `testCaseId`, `agentId`, `modelId`, `status`, `passFailStatus` | No — untouched, still explicit `keyword` fields |
103
+ | `OpenSearchRunOperations.search()` | `range` filter | `createdAt` | No — untouched, still `date` |
104
+ | `OpenSearchRunOperations.getAll()` / `.search()` | `sort` | `createdAt` | No |
105
+ | `OpenSearchRunOperations.countsByTestCase()` | `terms` agg | `testCaseId` | No |
106
+ | `asyncRunStorage.ts` `SearchQuery.minAccuracy` | **application-level** `Array.filter()`, not an OpenSearch query (`reports.filter(r => r.metrics.accuracy >= ...)`) | `metrics.accuracy` (read from `_source` in JS) | No — reads the value out of `_source`, which is unaffected by `dynamic: false`. If this were ever converted to a server-side `range` query, it would still work: `accuracy` stays an explicitly mapped, queryable field. |
107
+ | UI (`MatcherResultsPanel.tsx`, `JudgeSection.tsx`, `RunDetailsContent.tsx`) | none — reads `matcherResults`/`judgeMetrics` out of the fetched JSON document, never issues its own OpenSearch query | n/a | No |
108
+ | `services/evaluation/index.ts`, `services/benchmarkRunner.ts`, `services/hookOrchestrator.ts` | none — same, in-process consumption of the already-fetched report | n/a | No |
109
+
110
+ Conclusion: **no consumer anywhere issues an OpenSearch-side query against a
111
+ non-legacy `metrics.*` or `judgeMetrics.*` name.** Both are read back via
112
+ `_source` wherever consumed (search, list, comparison, UI). This mirrors
113
+ exactly the trade-off #418 already made and documented for
114
+ `EvaluationRun.results`.
115
+
116
+ ## Migration story — what to run, exactly
117
+
118
+ **Nothing runs automatically against the live cluster from this PR.**
119
+
120
+ ### New / fresh indexes
121
+
122
+ No action needed. `ensureIndexes()` (`server/services/indexInitializer.ts`,
123
+ called on every server boot and on "attach new cluster") creates any missing
124
+ index straight from the updated `INDEX_MAPPINGS` — new deployments and any
125
+ environment that doesn't have `evals_runs` yet get the fix immediately.
126
+
127
+ ### Existing, NOT-YET-poisoned `evals_runs` (most environments)
128
+
129
+ Also no action needed, but not immediate — `ensureIndexes()` also calls
130
+ `client.indices.putMapping()` on every boot for existing indexes, which is
131
+ how the `dynamic: false` fix reaches an already-existing-but-clean index: it
132
+ succeeds silently and the index is protected from the next write onward.
133
+
134
+ ### The shared cluster's `evals_runs`, if already poisoned
135
+
136
+ If any code-QA benchmark run already wrote a custom evaluator metric name to
137
+ the shared cluster before this fix ships, `evals_runs.metrics` (and/or
138
+ `matcherResults.judgeMetrics`) already has real, dynamically-inferred
139
+ sub-properties. OpenSearch's `putMapping` **rejects** an `enabled`/`dynamic`
140
+ change on a field that already has sub-properties
141
+ (`mapper_exception: the [dynamic] parameter can't be updated for the object
142
+ mapping [metrics]`) — `ensureIndexes()` catches this, logs a warning, and
143
+ otherwise no-ops (no crash, no data loss, same as #418's documented
144
+ `mapper_exception` handling for `results`). **The index keeps growing** until
145
+ an explicit reindex is run.
146
+
147
+ **This is not new migration code** — the existing generic reindex mechanism
148
+ (`reindexSingleIndex()`, `server/services/mappingFixer.ts`, already shipped
149
+ and already exposed at `POST /api/storage/reindex`, `server/routes/storage/admin.ts`)
150
+ already recreates any `INDEX_MAPPINGS`-registered index from scratch and
151
+ copies every document across, which sheds a poisoned mapping's dynamically-
152
+ inferred sub-fields while preserving 100% of the underlying `_source` data
153
+ (proven in `tests/integration/services/storage/evalsRunsMappingMigrationRecipe.integration.test.ts`,
154
+ run against a real OpenSearch container with a deliberately-poisoned index).
155
+
156
+ **The owner's exact recipe, when ready to run it against the shared cluster:**
157
+
158
+ ```bash
159
+ # 1. Confirm the target index actually needs it (optional sanity check):
160
+ curl -s -X GET "$OPENSEARCH_STORAGE_ENDPOINT/evals_runs/_mapping" \
161
+ -u "$OPENSEARCH_STORAGE_USERNAME:$OPENSEARCH_STORAGE_PASSWORD" \
162
+ | jq '.evals_runs.mappings.properties.metrics'
163
+ # If this prints a `dynamic` key, it's already fixed. If it prints only
164
+ # `properties` with more than the 4 legacy metric names, it's poisoned.
165
+
166
+ # 2. Run the reindex via the running server's admin API (recreates the
167
+ # index from the current INDEX_MAPPINGS and copies every document
168
+ # across; the same doc-count-validated, recovery-safe path
169
+ # reindexSingleIndex() has always used for keyword-type mismatch fixes):
170
+ curl -s -X POST "http://localhost:4001/api/storage/reindex" \
171
+ -H 'Content-Type: application/json' \
172
+ -d '{"index": "evals_runs"}'
173
+ ```
174
+
175
+ Caveats to read before running this against the shared cluster:
176
+
177
+ - **No write lock during a manual `/api/storage/reindex` call.** The
178
+ auto-fix boot path (`fixIndexMappings()`) acquires a process-local
179
+ migration lock around the reindex; the manual admin route calls
180
+ `reindexSingleIndex()` directly and does **not** (pre-existing gap in
181
+ `server/routes/storage/admin.ts`, not introduced by this PR — flagged here,
182
+ not fixed, since it's out of scope for this change). Run it during a quiet
183
+ window (no in-flight evaluation runs writing reports) to avoid a write
184
+ racing the index delete/recreate step.
185
+ - It touches only `evals_runs`. The already-known-poisoned `evals_experiments`
186
+ (800+ stale `results.*` fields, per the incident notes) uses the identical
187
+ recipe (`{"index": "evals_experiments"}`) — that cleanup is separately
188
+ planned by ops; this PR does not touch or schedule it.
189
+ - Document count is validated before the temporary index is deleted; if the
190
+ copy-back count doesn't match, the error message names the surviving temp
191
+ index (`evals_runs_reindex_temp`) for manual recovery — nothing is deleted
192
+ until the counts are confirmed equal.
193
+
194
+ ## Tests
195
+
196
+ - Unit (`tests/unit/server/constants/indexMappings.test.ts`): mapping-shape
197
+ assertions — `dynamic: false` + typed legacy properties on both `metrics`
198
+ and `matcherResults.judgeMetrics`; pre-existing `enabled:false` fields stay
199
+ disabled; every field `OpenSearchRunOperations.search()` queries stays
200
+ explicitly mapped.
201
+ - Integration, real OpenSearch
202
+ (`tests/integration/services/storage/testCaseRunMetricsMappingGrowth.integration.test.ts`):
203
+ writes one report with 1000+ distinct custom `metrics`/`judgeMetrics` names
204
+ (500 report-level + 125 `judge()` calls × 4 dimensions), asserts it
205
+ round-trips correctly and the index's total mapped-field count does not
206
+ grow; asserts the query-audit fields stay queryable.
207
+ - Integration, real OpenSearch, migration recipe
208
+ (`tests/integration/services/storage/evalsRunsMappingMigrationRecipe.integration.test.ts`):
209
+ deliberately poisons a throwaway index the old way, runs the *existing*
210
+ `reindexSingleIndex()`, asserts the mapping resets to `dynamic: false` and
211
+ all document data survives byte-for-byte.
212
+
213
+ Both integration suites skip gracefully (with a console warning) if no
214
+ OpenSearch cluster is reachable at `TEST_OPENSEARCH_ENDPOINT` (default
215
+ `http://localhost:9200`) — the unit suite covers the mapping-shape assertions
216
+ unconditionally.
@@ -167,7 +167,7 @@ run a JSON file:
167
167
 
168
168
  ```bash
169
169
  # `-f` accepts BOTH JSON test-case files and code SDK (.eval.js / .eval.ts) files
170
- npx @opensearch-project/agent-health benchmark -f ./evals/demo.eval.js -a my-agent
170
+ npx @opensearch-project/agent-health benchmark -f ./examples/eval-files/demo.eval.js -a my-agent
171
171
  ```
172
172
 
173
173
  They produce **per-matcher results** (`matcherResults[]`) instead of a single
@@ -303,6 +303,60 @@ Repeat until all high-priority issues are resolved.
303
303
  | Server / config issues | `npx agent-health doctor` (checks config + connectivity) |
304
304
  | "OpenSearch storage not configured" | Fine for local use — file-based storage is the default. Set `OPENSEARCH_STORAGE_*` only for shared / production persistence. |
305
305
 
306
+ > **Note (benchmark execution vs. reads):** file-based storage covers reads,
307
+ > single `run`s, and sample data, but **executing a multi-case `benchmark`
308
+ > currently requires an OpenSearch storage client** — without one the execute
309
+ > path returns `Cannot execute in sample-only mode`. Point `OPENSEARCH_STORAGE_*`
310
+ > at a cluster (a local security-disabled Docker OpenSearch on plain HTTP with
311
+ > `authType=none` is enough) before running `benchmark`.
312
+
313
+ ---
314
+
315
+ ## Benchmarking a local (subprocess) agent — gotchas
316
+
317
+ When you wrap a local CLI agent (Kiro, Claude Code, Pi, or your own script) as a
318
+ **subprocess** agent and run a `benchmark`, these are the traps that bite first
319
+ — check them before blaming the agent:
320
+
321
+ 1. **Pin a real judge model — don't leave it on `demo`.** The `demo` provider is
322
+ a mock judge that returns high pass rates without calling an LLM, so a run
323
+ can look like "100% pass" while nothing was actually judged. Set the run's
324
+ judge model to a real Bedrock model, e.g.
325
+ `us.anthropic.claude-sonnet-4-5-20250929-v1:0`, and confirm it's invocable in
326
+ your account (`aws bedrock ... ` / `doctor`). If your pass rate looks too
327
+ good, check the judge provider first.
328
+ 2. **Raise the subprocess timeout for slow agents.** The subprocess connector
329
+ defaults to a 5-minute (`300000` ms) timeout. A real ops/RCA agent can run
330
+ ~10 min. Set it in your agent's `connectorConfig`:
331
+ ```ts
332
+ { key: 'my-ops-agent', connectorType: 'subprocess',
333
+ connectorConfig: { timeout: 1200000 /* 20 min */, /* ... */ } }
334
+ ```
335
+ 3. **Fail loud on a wrong agent name.** If your wrapper points at an agent key
336
+ that doesn't exist, the underlying CLI may silently fall back to a default
337
+ agent — so you benchmark the wrong thing. Echo the resolved agent name in
338
+ your wrapper and eyeball the first trajectory.
339
+ 4. **Long runs + CLI SSE disconnects.** On runs longer than a few minutes the
340
+ CLI's streaming connection can drop and report `0/0` / `fetch failed` **while
341
+ the server keeps going**. The results are still persisted — read them back
342
+ from storage (`list runs` / the UI run inspector / `--export`) rather than
343
+ trusting the CLI summary.
344
+ 5. **Watch for a stale server on the port.** If you patch/upgrade the package
345
+ but a previously-started `npx` server is still bound to port 4001, your runs
346
+ are served by the old code. Confirm which process owns the port
347
+ (`lsof -i :4001` / a `/proc` sweep) and restart the one you actually patched
348
+ (the server loads `server/dist/app.js`, not `index.js`).
349
+ 6. **Judge `CredentialsProviderError` after a session rotates.** If your AWS
350
+ sandbox session rotates, `AWS_SHARED_CREDENTIALS_FILE` can point at a stale/
351
+ empty file and the judge fails. Re-vend credentials and restart the server
352
+ with the AWS env explicit (`aws sts get-caller-identity` to confirm first).
353
+ 7. **Local OpenSearch dying (exit 255) = memory pressure.** Give the container
354
+ enough heap and run it with `--restart=unless-stopped`.
355
+ 8. **Case schema is strict.** Test cases need `expectedOutcomes` (a
356
+ **string[]**, not a singular `expectedOutcome` string) and a capitalized
357
+ `difficulty` (`Easy` | `Medium` | `Hard`). Validate your converter output
358
+ against a known-good case before importing 16 of them.
359
+
306
360
  ---
307
361
 
308
362
  ## Server API Reference
@@ -18,7 +18,7 @@
18
18
  * -H 'Content-Type: application/json' \
19
19
  * -d '{
20
20
  * "name":"SDK Demo",
21
- * "sources":[{"type":"code-import","filenames":["evals/demo.eval.js"],"testCaseIds":[]}],
21
+ * "sources":[{"type":"code-import","filenames":["examples/eval-files/demo.eval.js"],"testCaseIds":[]}],
22
22
  * "agentKey":"observio",
23
23
  * "modelId":"claude-sonnet"
24
24
  * }'
@@ -0,0 +1,71 @@
1
+ /*
2
+ * Copyright OpenSearch Contributors
3
+ * SPDX-License-Identifier: Apache-2.0
4
+ */
5
+
6
+ /**
7
+ * Worked example — evaluating an ops / RCA agent that triages a ticket.
8
+ *
9
+ * The agent under test reads a ticket and emits a structured classification:
10
+ * - ticketType: "latency" | "fault" (deterministic — exact match)
11
+ * - rootCause: one of a fixed category set (deterministic — exact match)
12
+ * - sop: a recommended runbook (non-deterministic — LLM judge)
13
+ *
14
+ * This shows the two check styles side by side:
15
+ * • deterministic → expect(...).to.equal(...) on the parsed output
16
+ * • non-deterministic → judge(result, '<natural-language claim>')
17
+ *
18
+ * WEIGHTED SCORING (root-cause 60%, SOP, metrics, latency 10%):
19
+ * per-matcher pass/fail lives here, but the *single weighted aggregate score*
20
+ * across criteria is defined once in a custom EVALUATOR, not in the test body.
21
+ * Attach it at run time with `evaluatorId` — see ops-rca-evaluator.json next to
22
+ * this file. Run:
23
+ *
24
+ * npx @opensearch-project/agent-health benchmark \
25
+ * -f ./examples/eval-files/ops-rca-classification.eval.js -a my-ops-agent
26
+ *
27
+ * Docs: ../../docs/SDK.md Instrumentation: ../../docs/INSTRUMENT_WITH_OTEL.md
28
+ */
29
+
30
+ const { test, expect } = require('@opensearch-project/agent-health');
31
+
32
+ // The agent's allowed root-cause categories — deterministic ground truth.
33
+ const ROOT_CAUSE_CATEGORIES = [
34
+ 'dependency_outage',
35
+ 'resource_exhaustion',
36
+ 'config_error',
37
+ 'code_regression',
38
+ 'network',
39
+ ];
40
+
41
+ test('ticket-123-db-outage', {
42
+ prompt: 'Triage ticket TICKET-123 and return your classification table.',
43
+ description: 'DB dependency outage — must classify as fault + dependency_outage',
44
+ context: [
45
+ {
46
+ description: 'Ticket body',
47
+ value:
48
+ 'TICKET-123: payment-service returning 500s since 10:30. ' +
49
+ 'Logs: "Connection refused to database-primary:5432". p99 latency normal until errors began.',
50
+ },
51
+ ],
52
+ labels: ['category:RCA', 'difficulty:Medium', 'agent:ops', 'type:fault'],
53
+ }, async function ({ agent, judge }) {
54
+ const result = await agent.run();
55
+
56
+ // ── Deterministic checks — exact classification, no LLM, $0 ──────────────
57
+ const out = result.parsedOutput() || {}; // agent emits JSON classification
58
+ expect(out.ticketType).to.equal('fault');
59
+ expect(ROOT_CAUSE_CATEGORIES).to.include(out.rootCause);
60
+ expect(out.rootCause).to.equal('dependency_outage');
61
+
62
+ // Prove it actually investigated rather than guessing.
63
+ expect(result.trajectory).to.haveStepsOfType('action');
64
+
65
+ // ── Non-deterministic checks — LLM judge on the free-text SOP ────────────
66
+ await judge(result, 'Recommends a runbook appropriate for a database dependency outage');
67
+ await judge(result, 'Explains that payment-service cannot reach database-primary as the root cause');
68
+
69
+ // ── Budget guard (feeds the "latency 10%" weight via the evaluator) ──────
70
+ expect(result).to.haveCompletedWithin(120_000);
71
+ });
@@ -0,0 +1,15 @@
1
+ {
2
+ "name": "Ops RCA (weighted)",
3
+ "description": "Weighted scoring for a ticket-triage / RCA ops agent. Root-cause accuracy dominates; latency is a small tie-breaker. Attach to a run via evaluatorId.",
4
+ "systemPrompt": "You are evaluating an ops agent that triages a support ticket. The agent must (1) classify the ticket type (latency vs fault), (2) identify the root cause from a fixed category set, (3) recommend an appropriate SOP/runbook, and (4) check the relevant metrics. Score each metric from 0 to 100.\n\nCRITICAL CRITERIA:\n- root_cause_accuracy is the PRIMARY metric. A wrong root cause is a failed triage regardless of everything else.\n- sop_selection: did the agent recommend a runbook appropriate for the identified root cause?\n- relevant_metrics: did the agent inspect the metrics/logs that actually matter for this failure mode?\n- latency: score higher when the agent reaches a correct answer in fewer steps / less wall-clock time.\n\nReturn pass_fail_status, reasoning, a metrics object with the four numeric scores, and improvement_strategies.",
5
+ "scoringConfig": {
6
+ "metrics": [
7
+ { "name": "root_cause_accuracy", "weight": 0.6, "scale": 100 },
8
+ { "name": "sop_selection", "weight": 0.2, "scale": 100 },
9
+ { "name": "relevant_metrics", "weight": 0.1, "scale": 100 },
10
+ { "name": "latency", "weight": 0.1, "scale": 100 }
11
+ ],
12
+ "passThreshold": 80,
13
+ "scale": 100
14
+ }
15
+ }
@@ -0,0 +1,72 @@
1
+ /*
2
+ * Copyright OpenSearch Contributors
3
+ * SPDX-License-Identifier: Apache-2.0
4
+ */
5
+
6
+ /**
7
+ * SDK demo eval — three test cases that show the spectrum of evaluation
8
+ * methods the code-based SDK supports:
9
+ *
10
+ * 1. mock-says-hello (deterministic) — agent invoked, only chai matchers
11
+ * 2. mock-rca-judged (agentic) — agent invoked + LLM judge matcher
12
+ * 3. data-only-no-prompt (deterministic) — no agent call at all
13
+ *
14
+ * Run with:
15
+ * AH_PORT=4002 npx @opensearch-project/agent-health benchmark \
16
+ * -f examples/eval-files/sdk-demo.eval.js -a demo
17
+ */
18
+
19
+ const { test, expect } = require('@opensearch-project/agent-health');
20
+
21
+ // ─────────────────────────────────────────────────────────────────────────────
22
+ // 1. Deterministic — agent runs, all assertions are local chai matchers
23
+ // ─────────────────────────────────────────────────────────────────────────────
24
+
25
+ test('mock-says-hello', {
26
+ prompt: 'Say hello in one short sentence.',
27
+ description: 'Mock agent must produce a non-empty response within 30s',
28
+ labels: ['category:Smoke', 'difficulty:Easy', 'method:deterministic'],
29
+ }, async function ({ agent }) {
30
+ const result = await agent.run();
31
+ expect(result.trajectory).to.have.length.greaterThan(0);
32
+ expect(result.agentOutput.trim()).to.have.length.greaterThan(0);
33
+ expect(result).to.haveCompletedWithin(30_000);
34
+ });
35
+
36
+ // ─────────────────────────────────────────────────────────────────────────────
37
+ // 2. Agentic (hybrid) — deterministic preflight + LLM judge for semantic claim
38
+ // ─────────────────────────────────────────────────────────────────────────────
39
+
40
+ test('mock-rca-judged', {
41
+ prompt: 'Diagnose why the payment service is failing and explain the root cause.',
42
+ description: 'Hybrid: structural checks first, then LLM judge for semantic correctness',
43
+ context: [
44
+ {
45
+ description: 'Error log',
46
+ value: 'ERROR 2026-05-20 10:31:22 [payment-service] Connection refused to db-primary:5432',
47
+ },
48
+ ],
49
+ labels: ['category:RCA', 'difficulty:Medium', 'method:agentic'],
50
+ }, async function ({ agent, judge }) {
51
+ const result = await agent.run();
52
+
53
+ // Cheap deterministic preflight — fail fast before spending $ on the judge
54
+ expect(result.trajectory).to.have.length.greaterThan(0);
55
+ expect(result).to.haveCompletedWithin(60_000);
56
+
57
+ // LLM judge — produces a structured matcher verdict with score + reasoning
58
+ await judge(result, 'Mentions the payment service or its database connection failure');
59
+ });
60
+
61
+ // ─────────────────────────────────────────────────────────────────────────────
62
+ // 3. Deterministic, no prompt — agent never invoked, $0 / 0ms agent step
63
+ // ─────────────────────────────────────────────────────────────────────────────
64
+
65
+ test('data-only-no-prompt', {
66
+ description: 'Pure data check; agent invocation skipped entirely',
67
+ labels: ['category:Data Quality', 'difficulty:Easy', 'method:deterministic'],
68
+ }, function ({ result }) {
69
+ expect(result.durationMs).to.equal(0);
70
+ expect(result.trajectory).to.have.length(0);
71
+ expect(2 + 2).to.equal(4);
72
+ });