@opensearch-project/agent-health 0.6.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. package/cli/dist/index.js +722 -330
  2. package/dist/assets/index-D-Np_l_T.js +246 -0
  3. package/dist/assets/index-vZt9QZKf.css +1 -0
  4. package/dist/index.html +2 -2
  5. package/docs/CLI.md +138 -1
  6. package/docs/CONNECTORS.md +1 -1
  7. package/docs/SDK.md +126 -2
  8. package/docs/skills/add-connector/SKILL.md +5 -1
  9. package/lib/dist/lib/agentTrends.d.ts +210 -0
  10. package/lib/dist/lib/agentTrends.d.ts.map +1 -0
  11. package/lib/dist/lib/agentTrends.js +360 -0
  12. package/lib/dist/lib/agentTrends.js.map +1 -0
  13. package/lib/dist/lib/benchmarkCaseReview.d.ts +114 -0
  14. package/lib/dist/lib/benchmarkCaseReview.d.ts.map +1 -0
  15. package/lib/dist/lib/benchmarkCaseReview.js +177 -0
  16. package/lib/dist/lib/benchmarkCaseReview.js.map +1 -0
  17. package/lib/dist/lib/benchmarkRunsTable.d.ts +109 -0
  18. package/lib/dist/lib/benchmarkRunsTable.d.ts.map +1 -0
  19. package/lib/dist/lib/benchmarkRunsTable.js +212 -0
  20. package/lib/dist/lib/benchmarkRunsTable.js.map +1 -0
  21. package/lib/dist/lib/comparisonInsights.d.ts +49 -2
  22. package/lib/dist/lib/comparisonInsights.d.ts.map +1 -1
  23. package/lib/dist/lib/comparisonInsights.js +65 -7
  24. package/lib/dist/lib/comparisonInsights.js.map +1 -1
  25. package/lib/dist/lib/config/loader.d.ts.map +1 -1
  26. package/lib/dist/lib/config/loader.js +11 -1
  27. package/lib/dist/lib/config/loader.js.map +1 -1
  28. package/lib/dist/lib/dashboardMetrics.d.ts +11 -2
  29. package/lib/dist/lib/dashboardMetrics.d.ts.map +1 -1
  30. package/lib/dist/lib/dashboardMetrics.js +38 -3
  31. package/lib/dist/lib/dashboardMetrics.js.map +1 -1
  32. package/lib/dist/lib/evaluationRerun.d.ts +39 -0
  33. package/lib/dist/lib/evaluationRerun.d.ts.map +1 -1
  34. package/lib/dist/lib/evaluationRerun.js +49 -0
  35. package/lib/dist/lib/evaluationRerun.js.map +1 -1
  36. package/lib/dist/lib/judgeFailureSummary.d.ts +66 -0
  37. package/lib/dist/lib/judgeFailureSummary.d.ts.map +1 -0
  38. package/lib/dist/lib/judgeFailureSummary.js +68 -0
  39. package/lib/dist/lib/judgeFailureSummary.js.map +1 -0
  40. package/lib/dist/lib/judgeStrategies.d.ts +108 -0
  41. package/lib/dist/lib/judgeStrategies.d.ts.map +1 -0
  42. package/lib/dist/lib/judgeStrategies.js +135 -0
  43. package/lib/dist/lib/judgeStrategies.js.map +1 -0
  44. package/lib/dist/lib/matchers/expect.d.ts +21 -1
  45. package/lib/dist/lib/matchers/expect.d.ts.map +1 -1
  46. package/lib/dist/lib/matchers/expect.js +51 -0
  47. package/lib/dist/lib/matchers/expect.js.map +1 -1
  48. package/lib/dist/lib/matchers/judgeAccessor.d.ts +4 -0
  49. package/lib/dist/lib/matchers/judgeAccessor.d.ts.map +1 -1
  50. package/lib/dist/lib/matchers/judgeAccessor.js +13 -2
  51. package/lib/dist/lib/matchers/judgeAccessor.js.map +1 -1
  52. package/lib/dist/lib/matchers/judgeReasoningParse.d.ts +49 -0
  53. package/lib/dist/lib/matchers/judgeReasoningParse.d.ts.map +1 -0
  54. package/lib/dist/lib/matchers/judgeReasoningParse.js +128 -0
  55. package/lib/dist/lib/matchers/judgeReasoningParse.js.map +1 -0
  56. package/lib/dist/lib/matchers/types.d.ts +25 -0
  57. package/lib/dist/lib/matchers/types.d.ts.map +1 -1
  58. package/lib/dist/lib/resolveCanonicalRun.d.ts +23 -0
  59. package/lib/dist/lib/resolveCanonicalRun.d.ts.map +1 -0
  60. package/lib/dist/lib/resolveCanonicalRun.js +26 -0
  61. package/lib/dist/lib/resolveCanonicalRun.js.map +1 -0
  62. package/lib/dist/lib/runActions.d.ts +120 -0
  63. package/lib/dist/lib/runActions.d.ts.map +1 -0
  64. package/lib/dist/lib/runActions.js +130 -0
  65. package/lib/dist/lib/runActions.js.map +1 -0
  66. package/lib/dist/lib/runInsights.d.ts +86 -0
  67. package/lib/dist/lib/runInsights.d.ts.map +1 -0
  68. package/lib/dist/lib/runInsights.js +185 -0
  69. package/lib/dist/lib/runInsights.js.map +1 -0
  70. package/lib/dist/lib/runName.d.ts +29 -0
  71. package/lib/dist/lib/runName.d.ts.map +1 -0
  72. package/lib/dist/lib/runName.js +38 -0
  73. package/lib/dist/lib/runName.js.map +1 -0
  74. package/lib/dist/lib/runReportPath.d.ts +16 -0
  75. package/lib/dist/lib/runReportPath.d.ts.map +1 -0
  76. package/lib/dist/lib/runReportPath.js +22 -0
  77. package/lib/dist/lib/runReportPath.js.map +1 -0
  78. package/lib/dist/lib/runSort.d.ts +27 -0
  79. package/lib/dist/lib/runSort.d.ts.map +1 -0
  80. package/lib/dist/lib/runSort.js +31 -0
  81. package/lib/dist/lib/runSort.js.map +1 -0
  82. package/lib/dist/lib/runStats.d.ts +86 -6
  83. package/lib/dist/lib/runStats.d.ts.map +1 -1
  84. package/lib/dist/lib/runStats.js +170 -19
  85. package/lib/dist/lib/runStats.js.map +1 -1
  86. package/lib/dist/lib/testCases/define.d.ts.map +1 -1
  87. package/lib/dist/lib/testCases/define.js +93 -46
  88. package/lib/dist/lib/testCases/define.js.map +1 -1
  89. package/lib/dist/lib/testCases/judge.d.ts.map +1 -1
  90. package/lib/dist/lib/testCases/judge.js +22 -4
  91. package/lib/dist/lib/testCases/judge.js.map +1 -1
  92. package/lib/dist/lib/testCases/loader.d.ts +24 -0
  93. package/lib/dist/lib/testCases/loader.d.ts.map +1 -1
  94. package/lib/dist/lib/testCases/loader.js +253 -36
  95. package/lib/dist/lib/testCases/loader.js.map +1 -1
  96. package/lib/dist/lib/trajectoryStepDisplay.d.ts +25 -0
  97. package/lib/dist/lib/trajectoryStepDisplay.d.ts.map +1 -0
  98. package/lib/dist/lib/trajectoryStepDisplay.js +42 -0
  99. package/lib/dist/lib/trajectoryStepDisplay.js.map +1 -0
  100. package/lib/dist/lib/utils.d.ts +19 -0
  101. package/lib/dist/lib/utils.d.ts.map +1 -1
  102. package/lib/dist/lib/utils.js +27 -0
  103. package/lib/dist/lib/utils.js.map +1 -1
  104. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.d.ts +68 -32
  105. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.d.ts.map +1 -1
  106. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.js +148 -125
  107. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.js.map +1 -1
  108. package/lib/dist/services/connectors/kiro/KiroConnector.d.ts +21 -13
  109. package/lib/dist/services/connectors/kiro/KiroConnector.d.ts.map +1 -1
  110. package/lib/dist/services/connectors/kiro/KiroConnector.js +22 -25
  111. package/lib/dist/services/connectors/kiro/KiroConnector.js.map +1 -1
  112. package/lib/dist/services/connectors/pi/PiConnector.d.ts +49 -10
  113. package/lib/dist/services/connectors/pi/PiConnector.d.ts.map +1 -1
  114. package/lib/dist/services/connectors/pi/PiConnector.js +102 -86
  115. package/lib/dist/services/connectors/pi/PiConnector.js.map +1 -1
  116. package/lib/dist/services/connectors/subprocess/SubprocessConnector.d.ts +59 -14
  117. package/lib/dist/services/connectors/subprocess/SubprocessConnector.d.ts.map +1 -1
  118. package/lib/dist/services/connectors/subprocess/SubprocessConnector.js +87 -61
  119. package/lib/dist/services/connectors/subprocess/SubprocessConnector.js.map +1 -1
  120. package/lib/dist/services/evaluation/bedrockJudge.d.ts +11 -0
  121. package/lib/dist/services/evaluation/bedrockJudge.d.ts.map +1 -1
  122. package/lib/dist/services/evaluation/bedrockJudge.js +2 -0
  123. package/lib/dist/services/evaluation/bedrockJudge.js.map +1 -1
  124. package/lib/dist/services/evaluation/index.d.ts +18 -1
  125. package/lib/dist/services/evaluation/index.d.ts.map +1 -1
  126. package/lib/dist/services/evaluation/index.js +156 -19
  127. package/lib/dist/services/evaluation/index.js.map +1 -1
  128. package/lib/dist/services/metrics.d.ts +55 -0
  129. package/lib/dist/services/metrics.d.ts.map +1 -0
  130. package/lib/dist/services/metrics.js +89 -0
  131. package/lib/dist/services/metrics.js.map +1 -0
  132. package/lib/dist/services/storage/asyncBenchmarkStorage.d.ts.map +1 -1
  133. package/lib/dist/services/storage/asyncBenchmarkStorage.js +16 -0
  134. package/lib/dist/services/storage/asyncBenchmarkStorage.js.map +1 -1
  135. package/lib/dist/services/storage/asyncRunStorage.d.ts +11 -0
  136. package/lib/dist/services/storage/asyncRunStorage.d.ts.map +1 -1
  137. package/lib/dist/services/storage/asyncRunStorage.js +49 -2
  138. package/lib/dist/services/storage/asyncRunStorage.js.map +1 -1
  139. package/lib/dist/services/storage/asyncTestCaseStorage.d.ts +1 -1
  140. package/lib/dist/services/storage/asyncTestCaseStorage.d.ts.map +1 -1
  141. package/lib/dist/services/storage/asyncTestCaseStorage.js +2 -1
  142. package/lib/dist/services/storage/asyncTestCaseStorage.js.map +1 -1
  143. package/lib/dist/services/storage/opensearchClient.d.ts +12 -0
  144. package/lib/dist/services/storage/opensearchClient.d.ts.map +1 -1
  145. package/lib/dist/services/storage/opensearchClient.js.map +1 -1
  146. package/lib/dist/services/traces/browserRecovery.d.ts.map +1 -1
  147. package/lib/dist/services/traces/browserRecovery.js +3 -0
  148. package/lib/dist/services/traces/browserRecovery.js.map +1 -1
  149. package/lib/dist/services/traces/index.d.ts +8 -1
  150. package/lib/dist/services/traces/index.d.ts.map +1 -1
  151. package/lib/dist/services/traces/index.js +33 -12
  152. package/lib/dist/services/traces/index.js.map +1 -1
  153. package/lib/dist/services/traces/judgeAgentsHints.d.ts +98 -3
  154. package/lib/dist/services/traces/judgeAgentsHints.d.ts.map +1 -1
  155. package/lib/dist/services/traces/judgeAgentsHints.js +144 -3
  156. package/lib/dist/services/traces/judgeAgentsHints.js.map +1 -1
  157. package/lib/dist/services/traces/spansToTrajectory.js +4 -4
  158. package/lib/dist/services/traces/spansToTrajectory.js.map +1 -1
  159. package/lib/dist/services/traces/tracePoller.d.ts.map +1 -1
  160. package/lib/dist/services/traces/tracePoller.js +17 -20
  161. package/lib/dist/services/traces/tracePoller.js.map +1 -1
  162. package/lib/dist/services/traces/trajectoryMerge.d.ts +79 -0
  163. package/lib/dist/services/traces/trajectoryMerge.d.ts.map +1 -0
  164. package/lib/dist/services/traces/trajectoryMerge.js +109 -0
  165. package/lib/dist/services/traces/trajectoryMerge.js.map +1 -0
  166. package/lib/dist/types/index.d.ts +195 -2
  167. package/lib/dist/types/index.d.ts.map +1 -1
  168. package/lib/dist/types/index.js +24 -0
  169. package/lib/dist/types/index.js.map +1 -1
  170. package/package.json +6 -5
  171. package/server/dist/app.js +3398 -926
  172. package/server/dist/index.js +3401 -929
  173. package/dist/assets/index-BfxtxmKc.css +0 -1
  174. package/dist/assets/index-CrjAfDHu.js +0 -243
@@ -529,7 +529,7 @@ See `services/connectors/claude-code/ClaudeCodeConnector.ts` for a complete exam
529
529
  ### Kiro Connector
530
530
 
531
531
  See `services/connectors/kiro/KiroConnector.ts` for a `SubprocessConnector`
532
- subclass that overrides `parseStderrChunk()` to convert Kiro's stderr-borne
532
+ subclass that overrides `parseStderrChunk(chunk, trajectory, onProgress, state)` to convert Kiro's stderr-borne
533
533
  `[tool] Running:` / `[tool] status:` markers into structured `action` +
534
534
  `tool_result` steps. The base `SubprocessConnector` also persists `stderr` to
535
535
  `rawOutput` and honors per-request `connectorConfig` overrides (`args` /
package/docs/SDK.md CHANGED
@@ -64,7 +64,7 @@ async function ({ agent, judge, evaluate, expect, testInfo, provisioned }) { ...
64
64
  | `agent` | `AgentFixture` | `await agent.run(prompt?, options?)` — invoke the agent **once** (control inversion). Returns an `EvalResult`. |
65
65
  | `judge` | `JudgeFn` | Non-throwing LLM judge. `judge(...)` gates; `judge.observe(...)` is observational. Returns a `Verdict`. |
66
66
  | `evaluate` | `EvaluateFn` | Run a custom programmatic evaluator registered with `defineEvaluator()`. |
67
- | `expect` | chai's `expect` with our recording plugin | Synchronous matcher entry-point |
67
+ | `expect` | chai's `expect` with our recording plugin | Synchronous matcher entry-point. `expect.soft(...)` records a failing assertion instead of throwing — see [Always-record guarantee](#5-matchers-record-structured-verdicts). |
68
68
  | `testInfo` | `TestInfo` (read-only) | `{ name, benchmarkPath, sourceFile, testCaseId }` |
69
69
  | `provisioned` | `Readonly<Record<string, unknown>>` | Values set by `beforeEach` via `provide()` |
70
70
  | `result` | `EvalResult` *(legacy eager path)* | Pre-populated agent result — present when the runner invokes before the body. Prefer `await agent.run()`. |
@@ -187,6 +187,79 @@ Matchers (4/5 passed)
187
187
  This is the major upgrade over throw-and-fail: every assertion gets its own
188
188
  row, status, and detail block.
189
189
 
190
+ #### Always-record guarantee
191
+
192
+ chai's `expect()` is fail-fast — the first failing assertion throws, and the
193
+ rest of the test body (any later `expect()`/`judge()`/`evaluate()` calls)
194
+ never executes. For a pass/fail verdict that's the intended Playwright-style
195
+ contract, but a report consumed by an optimizer (or by you, comparing runs)
196
+ needs the objective actuals — duration, token usage, USD cost — from every
197
+ axis, even when one gate fails partway through. A token-budget gate failing
198
+ shouldn't erase the cost figure that would otherwise have been on the report.
199
+
200
+ So regardless of where (or whether) the body throws, the runner **always**
201
+ stamps `performanceMetrics.durationMs` / `.agentDurationMs` /
202
+ `.totalTokens` / `.totalCostUsd` onto the report the moment `agent.run()`
203
+ resolves — reading them straight from the same OTel-derived data
204
+ `result.traces`/the `traces` fixture exposes, independent of whether the
205
+ body's own code ever reached a matcher that asserted on them. (`totalTokens`/
206
+ `totalCostUsd` are `undefined`, not `0`, when `useTraces: true` but spans
207
+ never arrived — see the loud-failure accessor in "Traces fixture" below;
208
+ they're real `0`s when `useTraces: false`.)
209
+
210
+ This guarantee has one hard limit: a `judge()`/`evaluate()` call itself is
211
+ already non-throwing and records its score/verdict the instant it's called —
212
+ so a REACHED judge call is always on the report today, throw-or-not. But a
213
+ judge call placed textually *after* a failing `expect()` never executes at
214
+ all — that's a pure source-order problem the runner cannot retroactively fix
215
+ (it can't record a call that never ran). Use `expect.soft(...)` (below) when
216
+ you need later judge/evaluate calls to run even after an earlier assertion
217
+ fails.
218
+
219
+ The matcher panel also gets a distinct **"not reached"** row appended
220
+ whenever the body threw — so "this axis never ran" (grey, not-reached) reads
221
+ differently from "this axis ran and failed" (red). It's excluded from the
222
+ passed/failed counts and gets its own tally: `(3/4 passed, 1 not reached)`.
223
+
224
+ #### `expect.soft(...)` — keep going instead of bailing on the first failure
225
+
226
+ `expect.soft(value)` is the same chai assertion surface as `expect(value)`
227
+ — every built-in BDD matcher plus the custom ones below — except a failing
228
+ assertion **records** the MatcherResult and returns instead of throwing:
229
+
230
+ ```javascript
231
+ test('budget-aware RCA', { prompt: '...' }, async ({ agent, expect, judge }) => {
232
+ const result = await agent.run();
233
+ expect.soft(result.traces.totalTokens).to.be.lessThan(10_000); // fails — recorded, body continues
234
+ expect.soft(result.traces.totalCost).to.be.lessThan(0.05); // still runs
235
+ await judge(result, 'identifies the root cause'); // still runs — the whole point
236
+ expect(result.agentOutput).to.contain('root cause'); // a HARD expect still bails on ITS OWN failure
237
+ });
238
+ ```
239
+
240
+ The runner's overall verdict is unaffected by soft vs. hard: a test already
241
+ fails when ANY recorded gate matcher has `pass: false` (that's exactly how
242
+ non-throwing `judge()`/`evaluate()` gates have always worked) — `expect.soft`
243
+ just extends that same non-throwing contract to chai assertions. Mix soft and
244
+ hard freely in one body; a hard `expect()` after a soft failure still bails
245
+ at that point (source order still matters for the calls *after* a HARD
246
+ failure). Reach for `.soft` on measurement/budget-style axis checks where you
247
+ want every number on the report regardless of which gates failed; keep hard
248
+ `expect()` for a precondition that makes the rest of the body meaningless if
249
+ it's false (e.g. "the agent produced any output at all").
250
+
251
+ > **Known limitation: multi-step property chains can record a derivative
252
+ > second failure.** `expect.soft(obj).to.have.property('x').that.equals(5)`
253
+ > is really two chai assertions run back-to-back on the same chain (`property`
254
+ > checks existence, `equals` checks the value) — with a HARD `expect()`, a
255
+ > missing property throws immediately and `.that.equals(5)` never runs. In
256
+ > soft mode nothing throws, so `.that.equals(5)` *does* run against
257
+ > `undefined` and records its own "expected undefined to equal 5" failure —
258
+ > real, but a symptom of the first failure, not independent information.
259
+ > Prefer single-step matchers on primitive/simple values with `.soft` (as in
260
+ > every example above); for a value that might not exist, use a hard
261
+ > `expect()` to check existence first, then `.soft` on the value itself.
262
+
190
263
  ---
191
264
 
192
265
  ## Test options
@@ -575,6 +648,57 @@ The `.eval.js`, `.eval.ts`, and `.eval.mjs` loaders all run through a single
575
648
  code-import execution path, so `benchmark -f <file>` executes the SDK body
576
649
  directly.
577
650
 
651
+ > **`.eval.js` and `.eval.ts` are both executed as synthetic CJS — only
652
+ > `.eval.mjs` needs the package to be really resolvable.** `.eval.js` and
653
+ > `.eval.ts` files are both executed in a synthetic context where the
654
+ > `require(...)` call for `@opensearch-project/agent-health` is intercepted
655
+ > directly, so they work from anywhere on disk regardless of whether the
656
+ > package is actually installed at that location. (`.eval.ts` is
657
+ > transpiled to CommonJS with `esbuild` — a required runtime dependency of
658
+ > this package — before running through that exact same synthetic-CJS
659
+ > path; it deliberately does NOT use a native `import()`, both to avoid two
660
+ > different Node versions giving the same file two different
661
+ > module-execution semantics, and because `import()` caches by URL, which
662
+ > would make a `.eval.ts` file unloadable a second time in one process
663
+ > without a restart. One consequence: an `.eval.ts` fixture can't use
664
+ > `import.meta` or a top-level `await` — real-ESM-only features with no
665
+ > CommonJS equivalent.) `.eval.mjs` files use a plain native `import()`, so
666
+ > `import { test } from '@opensearch-project/agent-health'` in an `.mjs`
667
+ > file must resolve through Node's normal module resolution — the package
668
+ > needs to be a real dependency reachable from the file's location
669
+ > (installed in `node_modules`, a workspace link, or a `node_modules`
670
+ > symlink to a local checkout). Test registrations are shared process-wide
671
+ > (keyed off `globalThis`) regardless of loader, so it doesn't matter
672
+ > *which* physical copy of the package a given import resolves to — dist
673
+ > vs source, symlinked vs installed all register into the same registry
674
+ > the loader reads from.
675
+
676
+ > **`.eval.ts` can `import` sibling `.ts` helper files, but not ESM-only
677
+ > packages.** A relative/absolute `import` inside an `.eval.ts` file that
678
+ > resolves to another `.ts` file (e.g. `import { helper } from
679
+ > './helper.ts'`) is transpiled and executed through the same synthetic-CJS
680
+ > mechanism recursively — multi-file `.eval.ts` fixtures work, and a helper
681
+ > required more than once within one load executes exactly once (cached
682
+ > for the duration of that load only, never across separate loads/reloads).
683
+ > An `import` of a package that ships **no CommonJS entry point** (ESM-only,
684
+ > e.g. modern `chalk`) fails with an actionable error — `.eval.ts` executes
685
+ > as synthetic CJS, so under require()'s own rules that package can't load
686
+ > there — unless the running Node version has native `require(esm)`
687
+ > interop (stable since Node 22.12; check `process.features.require_module`),
688
+ > in which case it works transparently via Node's own mechanism. If you hit
689
+ > the error on an older Node, switch the fixture to `.eval.mjs` (real ESM,
690
+ > can import ESM-only packages directly) or pre-compile to `.eval.js`.
691
+
692
+ > **`.eval.mjs` reloads are cache-busted, not cached.** Every `import()` of
693
+ > an `.eval.mjs` file is given a unique query string
694
+ > (`?ah-reload=<timestamp>-<random>`) so re-loading the same file a second
695
+ > time in one process (e.g. re-running the CLI against an edited fixture)
696
+ > re-executes its top-level code instead of silently returning Node's
697
+ > already-cached module for that URL. Accepted cost: each reload leaves one
698
+ > module instance cached under a never-reused query string for the
699
+ > lifetime of the process — negligible for CLI/server process lifetimes
700
+ > (a handful to a few hundred loads, not an unbounded long-running loop).
701
+
578
702
  ---
579
703
 
580
704
  ## Dev tips
@@ -623,4 +747,4 @@ keeping it user-supplied lets you opt in without breaking anyone else.
623
747
  - [x] Non-throwing run-scoped `judge` with `gate` / `observe` roles + `skip` + `orThrow()`
624
748
  - [x] `defineEvaluator()` / `evaluate()` for mechanical / external verification ([#244](https://github.com/opensearch-project/agent-health/issues/244))
625
749
  - [x] Single code-import execution path (`benchmark -f *.eval.js` runs the SDK body) + unified `.js` / `.ts` / `.mjs` loaders + `agent-health migrate sdk-v2` codemod
626
- - [ ] `expect.soft()` to collect-all-failures instead of bail-on-first
750
+ - [x] `expect.soft()` to collect-all-failures instead of bail-on-first
@@ -62,7 +62,11 @@ connectorRegistry.register(new CustomConnector());
62
62
  - `propagateHeader: true` → inject a `traceparent` HTTP header into HTTP/SSE agents (`injectTraceparentHeaders()`).
63
63
  - `serviceName: '<otel-service-name>'` → service-name + time-window fallback. Defaults: `claude-code-agent`, `kiro-agent`, `pi-agent`, `observio-sample-agent`. See the "Trace correlation conventions" section in `AGENTS.md`.
64
64
  - **Subprocess connectors** (`SubprocessConnector` subclasses) can override
65
- `parseStderrChunk(chunk)` to turn stderr markers into trajectory steps (how
65
+ `parseStderrChunk(chunk, trajectory, onProgress, state)` to turn stderr markers into trajectory steps (how
66
66
  `kiro` surfaces `[tool] Running:` / `[tool] status:` as `action` +
67
67
  `tool_result` steps). The base class persists `stderr` to `rawOutput` and
68
68
  honors per-request `connectorConfig` overrides (`args` / `inputMode` / `timeout`).
69
+ Keep ALL streaming state (partial-line buffers, pending tool names, captured
70
+ ids) on the per-invocation `state` object (extend `SubprocessExecutionState`
71
+ via `createExecutionState()`), never on `this` — the registry shares one
72
+ connector instance across concurrent runs.
@@ -0,0 +1,210 @@
1
+ /**
2
+ * Agent Trends Band — Aggregation Utilities
3
+ *
4
+ * Builds the per-run, per-agent data model behind the landing-page
5
+ * "Agent trends" band: v3 replaces the all-agents overlay chart (one
6
+ * ECharts line+scatter series per agent, sharing a 10-color palette) with
7
+ * a small-multiples sparkline-table — one row per agent, sorted by latest
8
+ * score or biggest drop, each row's sparkline + a single-agent focused
9
+ * detail chart built from the SAME gap-broken series shape.
10
+ *
11
+ * Deliberately kept UI-framework-free so it's cheaply unit testable and
12
+ * so it can be reused by the row list, the drawer, and the detail chart.
13
+ *
14
+ * Data-source note: accuracy/pass counts come straight from
15
+ * `Benchmark.runs[]` (already loaded by the Dashboard for the runs list —
16
+ * no extra fetch). Cost/tokens are trace-derived and only available for
17
+ * runs whose reports were resolved in the caller's `metricsMap` (see
18
+ * services/metrics.ts#fetchBatchMetrics); a run reports `costUsd`/`tokens`
19
+ * only when EVERY one of its reports resolved a match — a partial sum
20
+ * would understate the run's true cost while looking complete, which is
21
+ * worse than admitting "unknown" — so partially- or fully-unmatched runs
22
+ * get `null` rather than a fabricated or partial total, and callers render
23
+ * an honest "—" instead.
24
+ */
25
+ import type { Benchmark, BenchmarkRun, EvaluationReport } from '../types/index.js';
26
+ export type TrendMetricKey = 'accuracy' | 'cost' | 'tokens';
27
+ export type TrendsTimeRange = '7d' | '30d' | '90d';
28
+ export type TrendSortMode = 'latest' | 'biggestDrop';
29
+ export interface RunMetricsLookup {
30
+ costUsd: number;
31
+ tokens: number;
32
+ }
33
+ /** One dot on the trend chart: a single BenchmarkRun for one agent. */
34
+ export interface AgentRunPoint {
35
+ runDocId: string;
36
+ benchmarkId: string;
37
+ benchmarkName: string;
38
+ agentKey: string;
39
+ agentName: string;
40
+ modelId: string;
41
+ createdAt: string;
42
+ timestamp: number;
43
+ passed: number;
44
+ failed: number;
45
+ total: number;
46
+ /**
47
+ * 0-100 pass rate over the evaluable set, or `null` when the run had
48
+ * ZERO evaluable test cases (every case errored — no judge verdict at
49
+ * all). Owner-reported bug: this used to default to `0`, which is
50
+ * visually indistinguishable from a genuine "the agent passed nothing"
51
+ * result — a run with no signal must never render as a real data point.
52
+ */
53
+ accuracyPct: number | null;
54
+ costUsd: number | null;
55
+ tokens: number | null;
56
+ }
57
+ /**
58
+ * Deterministic, visually distinct color palette for the chart lines / chip
59
+ * left-borders. Assigned by sorted agentKey so colors stay stable across
60
+ * re-renders and time-range/benchmark filtering (not by first-seen order,
61
+ * which would shuffle as the visible run set changes).
62
+ *
63
+ * v3 note: the small-multiples redesign no longer overlays every agent's
64
+ * line on one chart, so per-agent hue no longer needs to scale to N agents
65
+ * (the old 10-color palette repeated after the 10th agent — "indistinguishable
66
+ * colors" in the owner's feedback with 14 agents in scope). Each row/detail
67
+ * chart is already labeled by name, so a single, theme-aware accent color
68
+ * plus semantic (up/down) delta coloring is used everywhere instead — see
69
+ * `TREND_LINE_COLOR` in AgentTrendSparkline/AgentTrendDetailChart.
70
+ */
71
+ /**
72
+ * Compute passed/failed/total/accuracyPct for a run, preferring the
73
+ * denormalized `run.stats` (fast path) and falling back to bucketing
74
+ * `run.results` directly (same canonical logic used by the runs list) —
75
+ * no report fetch required either way.
76
+ */
77
+ export declare function getRunAccuracy(run: BenchmarkRun): {
78
+ passed: number;
79
+ failed: number;
80
+ total: number;
81
+ accuracyPct: number | null;
82
+ };
83
+ export declare function timeRangeToSinceMs(range: TrendsTimeRange, nowMs?: number): number;
84
+ export interface BuildAgentRunPointsOptions {
85
+ benchmarkId?: string | null;
86
+ sinceMs?: number | null;
87
+ agentDisplayName?: (agentKey: string) => string;
88
+ }
89
+ /**
90
+ * Build one AgentRunPoint per BenchmarkRun (skipping runs with zero test
91
+ * cases — e.g. still-provisioning or cancelled-before-start runs), sorted
92
+ * ascending by time.
93
+ */
94
+ export declare function buildAgentRunPoints(benchmarks: Benchmark[], reports: EvaluationReport[], metricsMap: Map<string, RunMetricsLookup>, options?: BuildAgentRunPointsOptions): AgentRunPoint[];
95
+ /** Group already time-sorted points by agentKey, preserving relative order. */
96
+ export declare function groupPointsByAgent(points: AgentRunPoint[]): Map<string, AgentRunPoint[]>;
97
+ export declare function metricValue(point: AgentRunPoint, metric: TrendMetricKey): number | null;
98
+ export declare function formatMetricValue(metric: TrendMetricKey, value: number): string;
99
+ /** One entry in a gap-broken trend series. `value: null` marks a synthetic break, never a real run. */
100
+ export interface TrendSeriesEntry {
101
+ timestamp: number;
102
+ value: number | null;
103
+ point: AgentRunPoint | null;
104
+ }
105
+ /**
106
+ * Build one agent's gap-broken series for `metric`: points whose metric
107
+ * value is null (no evaluable accuracy, or no trace-resolved cost/tokens)
108
+ * are dropped — never plotted as a fake zero. A synthetic `{ value: null }`
109
+ * marker is inserted at the midpoint between any two consecutive PLOTTED
110
+ * points whose time gap exceeds `gapBreakMs` (default 7 days).
111
+ *
112
+ * Why: a chart line drawn between two data-array entries connects them
113
+ * with a straight (or smoothed) segment regardless of how far apart their
114
+ * x-values are — so two runs 25 days apart look, visually, like a
115
+ * continuous trend across those 25 days ("sparse runs produce misleading
116
+ * long interpolated lines" — owner feedback). Inserting a null-valued
117
+ * entry between them gives recharts' `<Line connectNulls={false}>` (the
118
+ * default) an explicit place to break the path, while the surrounding
119
+ * dead space still renders proportionally on a time-scaled x-axis.
120
+ */
121
+ export declare function buildGapBrokenSeries(agentPoints: AgentRunPoint[], // already time-sorted, single agent
122
+ metric: TrendMetricKey, gapBreakMs?: number): TrendSeriesEntry[];
123
+ /** Latest-vs-previous delta for one agent's metric series (not week-over-week — the immediately prior run). */
124
+ export interface TrendDelta {
125
+ latestValue: number | null;
126
+ previousValue: number | null;
127
+ /** latestValue - previousValue; null unless BOTH resolved to a real value. */
128
+ delta: number | null;
129
+ }
130
+ export declare function computeLatestDelta(agentPoints: AgentRunPoint[], metric: TrendMetricKey): TrendDelta;
131
+ export declare function formatDelta(metric: TrendMetricKey, delta: number | null): string;
132
+ /** One row in the sparkline-table: one agent, its gap-broken series for the current metric, latest value + delta. */
133
+ export interface AgentTrendRow {
134
+ agentKey: string;
135
+ agentName: string;
136
+ runCount: number;
137
+ latestRunAt: string | null;
138
+ latestRunDocId: string | null;
139
+ latestBenchmarkId: string | null;
140
+ /** All of this agent's points, time-sorted (unfiltered by metric) — the detail chart re-derives per-metric series from this on focus. */
141
+ points: AgentRunPoint[];
142
+ /** Gap-broken series for the CURRENTLY selected metric — what the row's sparkline draws. */
143
+ series: TrendSeriesEntry[];
144
+ latestValue: number | null;
145
+ previousValue: number | null;
146
+ delta: number | null;
147
+ }
148
+ /**
149
+ * Group `points` by agent and shape each into a sparkline-table row for
150
+ * the given metric. One row per agent that has at least one point in
151
+ * scope — a single run is a perfectly valid row (a lone dot, `delta: null`
152
+ * rendered as "n/a"); there is no "not enough data" floor at the row
153
+ * level (unlike the old all-agents chart, which needed >=2 points across
154
+ * ALL agents just to avoid an empty canvas).
155
+ */
156
+ export declare function buildAgentTrendRows(points: AgentRunPoint[], metric: TrendMetricKey, gapBreakMs?: number): AgentTrendRow[];
157
+ /**
158
+ * Sort rows by latest value (highest first; rows with no value for the
159
+ * current metric sort last) or by biggest drop (most negative delta
160
+ * first; rows with no delta sort last, ordered among themselves by latest
161
+ * value so a still-informative "no signal yet" row isn't buried below a
162
+ * literal `undefined`-sorts-arbitrarily bucket).
163
+ */
164
+ export declare function sortAgentTrendRows(rows: AgentTrendRow[], sortMode: TrendSortMode): AgentTrendRow[];
165
+ /** Default benchmark selector value: the benchmark with the most recent run. */
166
+ export declare function getMostRecentlyActiveBenchmarkId(benchmarks: Benchmark[]): string | null;
167
+ /** One plottable dot: a run + its resolved metric value (metric-null points are never included). */
168
+ export interface DotPlotEntry {
169
+ point: AgentRunPoint;
170
+ value: number;
171
+ }
172
+ /** One row of the ranked dot plot: one agent, its latest dot (solid) + earlier dots (faded). */
173
+ export interface BenchmarkDotPlotRow {
174
+ agentKey: string;
175
+ agentName: string;
176
+ /** null when this agent has no run with a resolved value for the current metric (e.g. no trace-derived cost yet). */
177
+ latest: DotPlotEntry | null;
178
+ /** Earlier runs with a resolved value, time-ascending, EXCLUDING `latest`. */
179
+ history: DotPlotEntry[];
180
+ /** 0-based rank after `rankDotPlotRows` (0 = top/best); -1 until ranked. */
181
+ rank: number;
182
+ }
183
+ /**
184
+ * Group `points` (already scoped to one benchmark by the caller) by agent
185
+ * and split each agent's valued runs into `latest` (most recent with a
186
+ * resolved value) + `history` (earlier ones). An agent with zero valued
187
+ * runs for this metric still gets a row (`latest: null`) so it's visible
188
+ * in the list rather than silently disappearing when switching metrics.
189
+ */
190
+ export declare function buildBenchmarkDotPlotRows(points: AgentRunPoint[], metric: TrendMetricKey): BenchmarkDotPlotRow[];
191
+ /**
192
+ * Rank rows by latest value, best (highest) first — rows with no value for
193
+ * the current metric sort last (still listed, just unranked among
194
+ * themselves by name for a stable order). Stamps `rank` (0 = top) on the
195
+ * returned copies; does not mutate the input.
196
+ */
197
+ export declare function rankDotPlotRows(rows: BenchmarkDotPlotRow[]): BenchmarkDotPlotRow[];
198
+ /**
199
+ * Shared X-domain for a dot plot: accuracy is always anchored to the true
200
+ * [0, 100] scale (auto-scaling to the observed min/max would visually
201
+ * exaggerate small real-world differences, e.g. an 88-95% cluster would
202
+ * stretch to fill the whole width). Cost/tokens are zero-anchored (0 is a
203
+ * meaningful reference point for both) up to the observed max plus a
204
+ * little headroom so the top dot isn't flush against the edge. Returns
205
+ * `null` when there is nothing to plot at all.
206
+ */
207
+ export declare function metricDomain(rows: BenchmarkDotPlotRow[], metric: TrendMetricKey): [number, number] | null;
208
+ /** Map a value into a 0-100 position within `domain`, clamped — usable directly as a CSS `left`/`width` percentage. */
209
+ export declare function valueToPercent(value: number, domain: [number, number]): number;
210
+ //# sourceMappingURL=agentTrends.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"agentTrends.d.ts","sourceRoot":"","sources":["../../agentTrends.ts"],"names":[],"mappings":"AAKA;;;;;;;;;;;;;;;;;;;;;;;GAuBG;AAEH,OAAO,KAAK,EAAE,SAAS,EAAE,YAAY,EAAE,gBAAgB,EAAE,MAAM,SAAS,CAAC;AAIzE,MAAM,MAAM,cAAc,GAAG,UAAU,GAAG,MAAM,GAAG,QAAQ,CAAC;AAE5D,MAAM,MAAM,eAAe,GAAG,IAAI,GAAG,KAAK,GAAG,KAAK,CAAC;AAEnD,MAAM,MAAM,aAAa,GAAG,QAAQ,GAAG,aAAa,CAAC;AAErD,MAAM,WAAW,gBAAgB;IAC/B,OAAO,EAAE,MAAM,CAAC;IAChB,MAAM,EAAE,MAAM,CAAC;CAChB;AAED,uEAAuE;AACvE,MAAM,WAAW,aAAa;IAC5B,QAAQ,EAAE,MAAM,CAAC;IACjB,WAAW,EAAE,MAAM,CAAC;IACpB,aAAa,EAAE,MAAM,CAAC;IACtB,QAAQ,EAAE,MAAM,CAAC;IACjB,SAAS,EAAE,MAAM,CAAC;IAClB,OAAO,EAAE,MAAM,CAAC;IAChB,SAAS,EAAE,MAAM,CAAC;IAClB,SAAS,EAAE,MAAM,CAAC;IAClB,MAAM,EAAE,MAAM,CAAC;IACf,MAAM,EAAE,MAAM,CAAC;IACf,KAAK,EAAE,MAAM,CAAC;IACd;;;;;;OAMG;IACH,WAAW,EAAE,MAAM,GAAG,IAAI,CAAC;IAC3B,OAAO,EAAE,MAAM,GAAG,IAAI,CAAC;IACvB,MAAM,EAAE,MAAM,GAAG,IAAI,CAAC;CACvB;AAID;;;;;;;;;;;;;GAaG;AAEH;;;;;GAKG;AACH,wBAAgB,cAAc,CAAC,GAAG,EAAE,YAAY,GAAG;IACjD,MAAM,EAAE,MAAM,CAAC;IACf,MAAM,EAAE,MAAM,CAAC;IACf,KAAK,EAAE,MAAM,CAAC;IACd,WAAW,EAAE,MAAM,GAAG,IAAI,CAAC;CAC5B,CAUA;AAED,wBAAgB,kBAAkB,CAAC,KAAK,EAAE,eAAe,EAAE,KAAK,GAAE,MAAmB,GAAG,MAAM,CAG7F;AAED,MAAM,WAAW,0BAA0B;IACzC,WAAW,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAC5B,OAAO,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IACxB,gBAAgB,CAAC,EAAE,CAAC,QAAQ,EAAE,MAAM,KAAK,MAAM,CAAC;CACjD;AAED;;;;GAIG;AACH,wBAAgB,mBAAmB,CACjC,UAAU,EAAE,SAAS,EAAE,EACvB,OAAO,EAAE,gBAAgB,EAAE,EAC3B,UAAU,EAAE,GAAG,CAAC,MAAM,EAAE,gBAAgB,CAAC,EACzC,OAAO,GAAE,0BAA+B,GACvC,aAAa,EAAE,CAgEjB;AAED,+EAA+E;AAC/E,wBAAgB,kBAAkB,CAAC,MAAM,EAAE,aAAa,EAAE,GAAG,GAAG,CAAC,MAAM,EAAE,aAAa,EAAE,CAAC,CAQxF;AAED,wBAAgB,WAAW,CAAC,KAAK,EAAE,aAAa,EAAE,MAAM,EAAE,cAAc,GAAG,MAAM,GAAG,IAAI,CAWvF;AAED,wBAAgB,iBAAiB,CAAC,MAAM,EAAE,cAAc,EAAE,KAAK,EAAE,MAAM,GAAG,MAAM,CAW/E;AAED,uGAAuG;AACvG,MAAM,WAAW,gBAAgB;IAC/B,SAAS,EAAE,MAAM,CAAC;IAClB,KAAK,EAAE,MAAM,GAAG,IAAI,CAAC;IACrB,KAAK,EAAE,aAAa,GAAG,IAAI,CAAC;CAC7B;AAED;;;;;;;;;;;;;;;GAeG;AACH,wBAAgB,oBAAoB,CAClC,WAAW,EAAE,aAAa,EAAE,EAAE,oCAAoC;AAClE,MAAM,EAAE,cAAc,EACtB,UAAU,GAAE,MAA6B,GACxC,gBAAgB,EAAE,CAiBpB;AAED,+GAA+G;AAC/G,MAAM,WAAW,UAAU;IACzB,WAAW,EAAE,MAAM,GAAG,IAAI,CAAC;IAC3B,aAAa,EAAE,MAAM,GAAG,IAAI,CAAC;IAC7B,8EAA8E;IAC9E,KAAK,EAAE,MAAM,GAAG,IAAI,CAAC;CACtB;AAED,wBAAgB,kBAAkB,CAAC,WAAW,EAAE,aAAa,EAAE,EAAE,MAAM,EAAE,cAAc,GAAG,UAAU,CASnG;AAED,wBAAgB,WAAW,CAAC,MAAM,EAAE,cAAc,EAAE,KAAK,EAAE,MAAM,GAAG,IAAI,GAAG,MAAM,CAShF;AAED,qHAAqH;AACrH,MAAM,WAAW,aAAa;IAC5B,QAAQ,EAAE,MAAM,CAAC;IACjB,SAAS,EAAE,MAAM,CAAC;IAClB,QAAQ,EAAE,MAAM,CAAC;IACjB,WAAW,EAAE,MAAM,GAAG,IAAI,CAAC;IAC3B,cAAc,EAAE,MAAM,GAAG,IAAI,CAAC;IAC9B,iBAAiB,EAAE,MAAM,GAAG,IAAI,CAAC;IACjC,yIAAyI;IACzI,MAAM,EAAE,aAAa,EAAE,CAAC;IACxB,4FAA4F;IAC5F,MAAM,EAAE,gBAAgB,EAAE,CAAC;IAC3B,WAAW,EAAE,MAAM,GAAG,IAAI,CAAC;IAC3B,aAAa,EAAE,MAAM,GAAG,IAAI,CAAC;IAC7B,KAAK,EAAE,MAAM,GAAG,IAAI,CAAC;CACtB;AAED;;;;;;;GAOG;AACH,wBAAgB,mBAAmB,CACjC,MAAM,EAAE,aAAa,EAAE,EACvB,MAAM,EAAE,cAAc,EACtB,UAAU,GAAE,MAA6B,GACxC,aAAa,EAAE,CAsBjB;AAED;;;;;;GAMG;AACH,wBAAgB,kBAAkB,CAAC,IAAI,EAAE,aAAa,EAAE,EAAE,QAAQ,EAAE,aAAa,GAAG,aAAa,EAAE,CAkBlG;AAED,gFAAgF;AAChF,wBAAgB,gCAAgC,CAAC,UAAU,EAAE,SAAS,EAAE,GAAG,MAAM,GAAG,IAAI,CAavF;AAWD,oGAAoG;AACpG,MAAM,WAAW,YAAY;IAC3B,KAAK,EAAE,aAAa,CAAC;IACrB,KAAK,EAAE,MAAM,CAAC;CACf;AAED,gGAAgG;AAChG,MAAM,WAAW,mBAAmB;IAClC,QAAQ,EAAE,MAAM,CAAC;IACjB,SAAS,EAAE,MAAM,CAAC;IAClB,qHAAqH;IACrH,MAAM,EAAE,YAAY,GAAG,IAAI,CAAC;IAC5B,8EAA8E;IAC9E,OAAO,EAAE,YAAY,EAAE,CAAC;IACxB,4EAA4E;IAC5E,IAAI,EAAE,MAAM,CAAC;CACd;AAED;;;;;;GAMG;AACH,wBAAgB,yBAAyB,CACvC,MAAM,EAAE,aAAa,EAAE,EACvB,MAAM,EAAE,cAAc,GACrB,mBAAmB,EAAE,CAkBvB;AAED;;;;;GAKG;AACH,wBAAgB,eAAe,CAAC,IAAI,EAAE,mBAAmB,EAAE,GAAG,mBAAmB,EAAE,CAQlF;AAED;;;;;;;;GAQG;AACH,wBAAgB,YAAY,CAAC,IAAI,EAAE,mBAAmB,EAAE,EAAE,MAAM,EAAE,cAAc,GAAG,CAAC,MAAM,EAAE,MAAM,CAAC,GAAG,IAAI,CAUzG;AAED,uHAAuH;AACvH,wBAAgB,cAAc,CAAC,KAAK,EAAE,MAAM,EAAE,MAAM,EAAE,CAAC,MAAM,EAAE,MAAM,CAAC,GAAG,MAAM,CAK9E"}