@opensearch-project/agent-health 0.5.2 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (230) hide show
  1. package/README.md +1 -0
  2. package/cli/dist/index.js +2325 -961
  3. package/dist/assets/index-D-Np_l_T.js +246 -0
  4. package/dist/assets/index-vZt9QZKf.css +1 -0
  5. package/dist/index.html +2 -2
  6. package/docs/CLI.md +183 -3
  7. package/docs/CONFIGURATION.md +1 -1
  8. package/docs/CONNECTORS.md +1 -1
  9. package/docs/INSTRUMENT_WITH_OTEL.md +7 -0
  10. package/docs/SDK.md +178 -5
  11. package/docs/STORAGE_INDEX_FIELD_LIMITS.md +216 -0
  12. package/docs/skills/AGENT_HEALTH.md +55 -1
  13. package/docs/skills/add-connector/SKILL.md +5 -1
  14. package/examples/eval-files/demo.eval.js +1 -1
  15. package/examples/eval-files/ops-rca-classification.eval.js +71 -0
  16. package/examples/eval-files/ops-rca-evaluator.json +15 -0
  17. package/examples/eval-files/sdk-demo.eval.js +72 -0
  18. package/examples/eval-files/sdk-describe-demo.eval.js +50 -0
  19. package/examples/eval-files/sdk-hooks-demo.eval.js +1 -1
  20. package/lib/dist/lib/agentTrends.d.ts +210 -0
  21. package/lib/dist/lib/agentTrends.d.ts.map +1 -0
  22. package/lib/dist/lib/agentTrends.js +360 -0
  23. package/lib/dist/lib/agentTrends.js.map +1 -0
  24. package/lib/dist/lib/bedrockCompat.d.ts +27 -0
  25. package/lib/dist/lib/bedrockCompat.d.ts.map +1 -0
  26. package/lib/dist/lib/bedrockCompat.js +83 -0
  27. package/lib/dist/lib/bedrockCompat.js.map +1 -0
  28. package/lib/dist/lib/benchmarkCaseReview.d.ts +114 -0
  29. package/lib/dist/lib/benchmarkCaseReview.d.ts.map +1 -0
  30. package/lib/dist/lib/benchmarkCaseReview.js +177 -0
  31. package/lib/dist/lib/benchmarkCaseReview.js.map +1 -0
  32. package/lib/dist/lib/benchmarkImage.d.ts +52 -0
  33. package/lib/dist/lib/benchmarkImage.d.ts.map +1 -0
  34. package/lib/dist/lib/benchmarkImage.js +113 -0
  35. package/lib/dist/lib/benchmarkImage.js.map +1 -0
  36. package/lib/dist/lib/benchmarkRunsTable.d.ts +109 -0
  37. package/lib/dist/lib/benchmarkRunsTable.d.ts.map +1 -0
  38. package/lib/dist/lib/benchmarkRunsTable.js +212 -0
  39. package/lib/dist/lib/benchmarkRunsTable.js.map +1 -0
  40. package/lib/dist/lib/benchmarkVersionUtils.d.ts +13 -0
  41. package/lib/dist/lib/benchmarkVersionUtils.d.ts.map +1 -1
  42. package/lib/dist/lib/benchmarkVersionUtils.js +21 -0
  43. package/lib/dist/lib/benchmarkVersionUtils.js.map +1 -1
  44. package/lib/dist/lib/chunkedFetch.d.ts +18 -0
  45. package/lib/dist/lib/chunkedFetch.d.ts.map +1 -0
  46. package/lib/dist/lib/chunkedFetch.js +40 -0
  47. package/lib/dist/lib/chunkedFetch.js.map +1 -0
  48. package/lib/dist/lib/comparisonInsights.d.ts +151 -0
  49. package/lib/dist/lib/comparisonInsights.d.ts.map +1 -0
  50. package/lib/dist/lib/comparisonInsights.js +270 -0
  51. package/lib/dist/lib/comparisonInsights.js.map +1 -0
  52. package/lib/dist/lib/config/loader.d.ts.map +1 -1
  53. package/lib/dist/lib/config/loader.js +16 -1
  54. package/lib/dist/lib/config/loader.js.map +1 -1
  55. package/lib/dist/lib/config/types.d.ts +14 -0
  56. package/lib/dist/lib/config/types.d.ts.map +1 -1
  57. package/lib/dist/lib/constants.d.ts +11 -0
  58. package/lib/dist/lib/constants.d.ts.map +1 -1
  59. package/lib/dist/lib/constants.js +10 -1
  60. package/lib/dist/lib/constants.js.map +1 -1
  61. package/lib/dist/lib/contextFormat.d.ts +26 -0
  62. package/lib/dist/lib/contextFormat.d.ts.map +1 -0
  63. package/lib/dist/lib/contextFormat.js +28 -0
  64. package/lib/dist/lib/contextFormat.js.map +1 -0
  65. package/lib/dist/lib/dashboardMetrics.d.ts +11 -2
  66. package/lib/dist/lib/dashboardMetrics.d.ts.map +1 -1
  67. package/lib/dist/lib/dashboardMetrics.js +38 -3
  68. package/lib/dist/lib/dashboardMetrics.js.map +1 -1
  69. package/lib/dist/lib/envCompat.d.ts.map +1 -1
  70. package/lib/dist/lib/envCompat.js +14 -5
  71. package/lib/dist/lib/envCompat.js.map +1 -1
  72. package/lib/dist/lib/evaluationRerun.d.ts +102 -0
  73. package/lib/dist/lib/evaluationRerun.d.ts.map +1 -0
  74. package/lib/dist/lib/evaluationRerun.js +134 -0
  75. package/lib/dist/lib/evaluationRerun.js.map +1 -0
  76. package/lib/dist/lib/judgeFailureSummary.d.ts +66 -0
  77. package/lib/dist/lib/judgeFailureSummary.d.ts.map +1 -0
  78. package/lib/dist/lib/judgeFailureSummary.js +68 -0
  79. package/lib/dist/lib/judgeFailureSummary.js.map +1 -0
  80. package/lib/dist/lib/judgeStrategies.d.ts +108 -0
  81. package/lib/dist/lib/judgeStrategies.d.ts.map +1 -0
  82. package/lib/dist/lib/judgeStrategies.js +135 -0
  83. package/lib/dist/lib/judgeStrategies.js.map +1 -0
  84. package/lib/dist/lib/matchers/expect.d.ts +21 -1
  85. package/lib/dist/lib/matchers/expect.d.ts.map +1 -1
  86. package/lib/dist/lib/matchers/expect.js +51 -0
  87. package/lib/dist/lib/matchers/expect.js.map +1 -1
  88. package/lib/dist/lib/matchers/judgeAccessor.d.ts +4 -0
  89. package/lib/dist/lib/matchers/judgeAccessor.d.ts.map +1 -1
  90. package/lib/dist/lib/matchers/judgeAccessor.js +13 -2
  91. package/lib/dist/lib/matchers/judgeAccessor.js.map +1 -1
  92. package/lib/dist/lib/matchers/judgeReasoningParse.d.ts +49 -0
  93. package/lib/dist/lib/matchers/judgeReasoningParse.d.ts.map +1 -0
  94. package/lib/dist/lib/matchers/judgeReasoningParse.js +128 -0
  95. package/lib/dist/lib/matchers/judgeReasoningParse.js.map +1 -0
  96. package/lib/dist/lib/matchers/traces.d.ts +17 -2
  97. package/lib/dist/lib/matchers/traces.d.ts.map +1 -1
  98. package/lib/dist/lib/matchers/traces.js +136 -16
  99. package/lib/dist/lib/matchers/traces.js.map +1 -1
  100. package/lib/dist/lib/matchers/tracesPricing.d.ts +38 -0
  101. package/lib/dist/lib/matchers/tracesPricing.d.ts.map +1 -0
  102. package/lib/dist/lib/matchers/tracesPricing.js +64 -0
  103. package/lib/dist/lib/matchers/tracesPricing.js.map +1 -0
  104. package/lib/dist/lib/matchers/types.d.ts +25 -0
  105. package/lib/dist/lib/matchers/types.d.ts.map +1 -1
  106. package/lib/dist/lib/resolveCanonicalRun.d.ts +23 -0
  107. package/lib/dist/lib/resolveCanonicalRun.d.ts.map +1 -0
  108. package/lib/dist/lib/resolveCanonicalRun.js +26 -0
  109. package/lib/dist/lib/resolveCanonicalRun.js.map +1 -0
  110. package/lib/dist/lib/runActions.d.ts +120 -0
  111. package/lib/dist/lib/runActions.d.ts.map +1 -0
  112. package/lib/dist/lib/runActions.js +130 -0
  113. package/lib/dist/lib/runActions.js.map +1 -0
  114. package/lib/dist/lib/runInsights.d.ts +86 -0
  115. package/lib/dist/lib/runInsights.d.ts.map +1 -0
  116. package/lib/dist/lib/runInsights.js +185 -0
  117. package/lib/dist/lib/runInsights.js.map +1 -0
  118. package/lib/dist/lib/runName.d.ts +29 -0
  119. package/lib/dist/lib/runName.d.ts.map +1 -0
  120. package/lib/dist/lib/runName.js +38 -0
  121. package/lib/dist/lib/runName.js.map +1 -0
  122. package/lib/dist/lib/runReportPath.d.ts +16 -0
  123. package/lib/dist/lib/runReportPath.d.ts.map +1 -0
  124. package/lib/dist/lib/runReportPath.js +22 -0
  125. package/lib/dist/lib/runReportPath.js.map +1 -0
  126. package/lib/dist/lib/runSort.d.ts +27 -0
  127. package/lib/dist/lib/runSort.d.ts.map +1 -0
  128. package/lib/dist/lib/runSort.js +31 -0
  129. package/lib/dist/lib/runSort.js.map +1 -0
  130. package/lib/dist/lib/runStats.d.ts +109 -5
  131. package/lib/dist/lib/runStats.d.ts.map +1 -1
  132. package/lib/dist/lib/runStats.js +193 -10
  133. package/lib/dist/lib/runStats.js.map +1 -1
  134. package/lib/dist/lib/testCases/define.d.ts.map +1 -1
  135. package/lib/dist/lib/testCases/define.js +93 -46
  136. package/lib/dist/lib/testCases/define.js.map +1 -1
  137. package/lib/dist/lib/testCases/judge.d.ts.map +1 -1
  138. package/lib/dist/lib/testCases/judge.js +22 -4
  139. package/lib/dist/lib/testCases/judge.js.map +1 -1
  140. package/lib/dist/lib/testCases/loader.d.ts +35 -1
  141. package/lib/dist/lib/testCases/loader.d.ts.map +1 -1
  142. package/lib/dist/lib/testCases/loader.js +278 -31
  143. package/lib/dist/lib/testCases/loader.js.map +1 -1
  144. package/lib/dist/lib/trajectoryStepDisplay.d.ts +25 -0
  145. package/lib/dist/lib/trajectoryStepDisplay.d.ts.map +1 -0
  146. package/lib/dist/lib/trajectoryStepDisplay.js +42 -0
  147. package/lib/dist/lib/trajectoryStepDisplay.js.map +1 -0
  148. package/lib/dist/lib/utils.d.ts +34 -0
  149. package/lib/dist/lib/utils.d.ts.map +1 -1
  150. package/lib/dist/lib/utils.js +49 -0
  151. package/lib/dist/lib/utils.js.map +1 -1
  152. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.d.ts +68 -22
  153. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.d.ts.map +1 -1
  154. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.js +148 -111
  155. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.js.map +1 -1
  156. package/lib/dist/services/connectors/kiro/KiroConnector.d.ts +21 -13
  157. package/lib/dist/services/connectors/kiro/KiroConnector.d.ts.map +1 -1
  158. package/lib/dist/services/connectors/kiro/KiroConnector.js +22 -25
  159. package/lib/dist/services/connectors/kiro/KiroConnector.js.map +1 -1
  160. package/lib/dist/services/connectors/pi/PiConnector.d.ts +49 -10
  161. package/lib/dist/services/connectors/pi/PiConnector.d.ts.map +1 -1
  162. package/lib/dist/services/connectors/pi/PiConnector.js +102 -86
  163. package/lib/dist/services/connectors/pi/PiConnector.js.map +1 -1
  164. package/lib/dist/services/connectors/subprocess/SubprocessConnector.d.ts +59 -14
  165. package/lib/dist/services/connectors/subprocess/SubprocessConnector.d.ts.map +1 -1
  166. package/lib/dist/services/connectors/subprocess/SubprocessConnector.js +95 -61
  167. package/lib/dist/services/connectors/subprocess/SubprocessConnector.js.map +1 -1
  168. package/lib/dist/services/evaluation/bedrockJudge.d.ts +18 -0
  169. package/lib/dist/services/evaluation/bedrockJudge.d.ts.map +1 -1
  170. package/lib/dist/services/evaluation/bedrockJudge.js +4 -0
  171. package/lib/dist/services/evaluation/bedrockJudge.js.map +1 -1
  172. package/lib/dist/services/evaluation/index.d.ts +18 -1
  173. package/lib/dist/services/evaluation/index.d.ts.map +1 -1
  174. package/lib/dist/services/evaluation/index.js +156 -19
  175. package/lib/dist/services/evaluation/index.js.map +1 -1
  176. package/lib/dist/services/metrics.d.ts +55 -0
  177. package/lib/dist/services/metrics.d.ts.map +1 -0
  178. package/lib/dist/services/metrics.js +89 -0
  179. package/lib/dist/services/metrics.js.map +1 -0
  180. package/lib/dist/services/storage/asyncBenchmarkStorage.d.ts +1 -1
  181. package/lib/dist/services/storage/asyncBenchmarkStorage.d.ts.map +1 -1
  182. package/lib/dist/services/storage/asyncBenchmarkStorage.js +30 -3
  183. package/lib/dist/services/storage/asyncBenchmarkStorage.js.map +1 -1
  184. package/lib/dist/services/storage/asyncRunStorage.d.ts +20 -1
  185. package/lib/dist/services/storage/asyncRunStorage.d.ts.map +1 -1
  186. package/lib/dist/services/storage/asyncRunStorage.js +125 -2
  187. package/lib/dist/services/storage/asyncRunStorage.js.map +1 -1
  188. package/lib/dist/services/storage/asyncTestCaseStorage.d.ts +18 -4
  189. package/lib/dist/services/storage/asyncTestCaseStorage.d.ts.map +1 -1
  190. package/lib/dist/services/storage/asyncTestCaseStorage.js +21 -4
  191. package/lib/dist/services/storage/asyncTestCaseStorage.js.map +1 -1
  192. package/lib/dist/services/storage/opensearchClient.d.ts +47 -12
  193. package/lib/dist/services/storage/opensearchClient.d.ts.map +1 -1
  194. package/lib/dist/services/storage/opensearchClient.js +12 -5
  195. package/lib/dist/services/storage/opensearchClient.js.map +1 -1
  196. package/lib/dist/services/traces/browserRecovery.d.ts +12 -1
  197. package/lib/dist/services/traces/browserRecovery.d.ts.map +1 -1
  198. package/lib/dist/services/traces/browserRecovery.js +35 -5
  199. package/lib/dist/services/traces/browserRecovery.js.map +1 -1
  200. package/lib/dist/services/traces/index.d.ts +8 -1
  201. package/lib/dist/services/traces/index.d.ts.map +1 -1
  202. package/lib/dist/services/traces/index.js +33 -12
  203. package/lib/dist/services/traces/index.js.map +1 -1
  204. package/lib/dist/services/traces/judgeAgentsHints.d.ts +98 -3
  205. package/lib/dist/services/traces/judgeAgentsHints.d.ts.map +1 -1
  206. package/lib/dist/services/traces/judgeAgentsHints.js +144 -3
  207. package/lib/dist/services/traces/judgeAgentsHints.js.map +1 -1
  208. package/lib/dist/services/traces/messageExtraction.d.ts.map +1 -1
  209. package/lib/dist/services/traces/messageExtraction.js +95 -33
  210. package/lib/dist/services/traces/messageExtraction.js.map +1 -1
  211. package/lib/dist/services/traces/spansToTrajectory.d.ts.map +1 -1
  212. package/lib/dist/services/traces/spansToTrajectory.js +51 -13
  213. package/lib/dist/services/traces/spansToTrajectory.js.map +1 -1
  214. package/lib/dist/services/traces/tracePoller.d.ts +27 -3
  215. package/lib/dist/services/traces/tracePoller.d.ts.map +1 -1
  216. package/lib/dist/services/traces/tracePoller.js +196 -33
  217. package/lib/dist/services/traces/tracePoller.js.map +1 -1
  218. package/lib/dist/services/traces/trajectoryMerge.d.ts +79 -0
  219. package/lib/dist/services/traces/trajectoryMerge.d.ts.map +1 -0
  220. package/lib/dist/services/traces/trajectoryMerge.js +109 -0
  221. package/lib/dist/services/traces/trajectoryMerge.js.map +1 -0
  222. package/lib/dist/types/index.d.ts +250 -3
  223. package/lib/dist/types/index.d.ts.map +1 -1
  224. package/lib/dist/types/index.js +24 -0
  225. package/lib/dist/types/index.js.map +1 -1
  226. package/package.json +8 -5
  227. package/server/dist/app.js +5490 -1557
  228. package/server/dist/index.js +5490 -1557
  229. package/dist/assets/index-CCQRDlO0.js +0 -243
  230. package/dist/assets/index-CNHQVbcj.css +0 -1
@@ -0,0 +1,120 @@
1
+ /**
2
+ * Pure, isomorphic predicates for the run-lifecycle action matrix (delete /
3
+ * cancel / re-run / retry-judgement) shared by every run surface (runs list,
4
+ * benchmark runs list, run detail/report page, inspector header) AND by the
5
+ * server routes that enforce the same rules server-side. No storage/IO here
6
+ * — callers pass in the run document they already have.
7
+ *
8
+ * Action matrix (see AGENTS.md / PR description for the full writeup):
9
+ * - Delete: any run, any status. Always available (existing endpoint).
10
+ * - Cancel: only while `status === 'running'`.
11
+ * - Re-run: only top-level EvaluationRun docs (docType === 'evaluation-run').
12
+ * Legacy benchmark-embedded BenchmarkRun rows don't support the
13
+ * provenance-tracked rerun endpoint (pre-existing constraint — see
14
+ * RunConfigDialog / EvalRunsPage).
15
+ * - Retry judgement: only EvaluationRun docs, only when the run is
16
+ * terminal (not running) AND it has at least one test case whose agent
17
+ * execution completed but the judge produced NO verdict (a judge-failed
18
+ * / "errored" case — trace timeout, judge 400, "evaluator could not
19
+ * run" — as opposed to an agent-failed one — retrying the judge on a
20
+ * case the agent itself never finished has nothing to re-grade). Same
21
+ * predicate the retry-judgement pipeline itself selects on
22
+ * (services/evaluation/retryJudgement.ts `isJudgeFailedCase`, keyed on
23
+ * the report's `metricsStatus: 'error'`, which the runner mirrors onto
24
+ * the run's results map as a `completed` result with no
25
+ * `passFailStatus`) and that `lib/runStats` buckets as `errored`.
26
+ */
27
+ import type { BenchmarkRun, EvaluationRun } from '../types/index.js';
28
+ /** Minimal shape both BenchmarkRun and EvaluationRun satisfy for these checks. */
29
+ export type RunLike = Pick<BenchmarkRun | EvaluationRun, 'status' | 'results'> & {
30
+ docType?: string;
31
+ };
32
+ /**
33
+ * True when `run` is a top-level EvaluationRun document (created via
34
+ * `POST /api/storage/evaluation-runs`), as opposed to a legacy
35
+ * benchmark-embedded BenchmarkRun (`benchmark.runs[]`). The two share a lot
36
+ * of shape but only EvaluationRun docs carry `docType: 'evaluation-run'` and
37
+ * support the rerun/retry-judgement endpoints.
38
+ *
39
+ * Null-tolerant wrapper over the typed predicate in `types/index.ts` (the
40
+ * single source of truth for the docType discriminator) — kept so callers
41
+ * holding a possibly-null run don't need their own guard.
42
+ */
43
+ export declare function isEvaluationRun(run: RunLike | null | undefined): run is EvaluationRun;
44
+ /** True while the run has an in-progress executor that a Cancel action could stop. */
45
+ export declare function isRunRunning(run: RunLike | null | undefined): boolean;
46
+ /** True once a run has reached any terminal state (not running/pending). */
47
+ export declare function isRunTerminal(run: RunLike | null | undefined): boolean;
48
+ /**
49
+ * Count test cases where the AGENT finished (`status === 'completed'`) but
50
+ * the JUDGE produced no verdict (`passFailStatus` neither 'passed' nor
51
+ * 'failed') — the "errored" bucket of `lib/runStats` `bucketRunResults`
52
+ * (issue #242) and exactly the set `POST .../retry-judgement` (default
53
+ * `scope=errored`) will re-judge. Deliberately excludes:
54
+ * - `status !== 'completed'` (agent-failed/cancelled/pending cases — no
55
+ * trajectory to re-judge, or nothing ran).
56
+ * - a real 'failed' verdict — the judge DID run and graded the case; that
57
+ * is a legitimate result, not a judge failure (re-grading it is
58
+ * `scope=all`, opt-in from the inspector's dedicated button).
59
+ *
60
+ * `passFailStatus` isn't declared on `EvaluationRun['results']`'s static
61
+ * type (a pre-existing gap — evaluationRunner.ts writes it via an `as any`
62
+ * spread) so this reads it defensively.
63
+ */
64
+ export declare function countJudgeFailed(run: RunLike | null | undefined): number;
65
+ export interface RunActionVisibility {
66
+ /** Delete is always available for any run in any status. */
67
+ canDelete: boolean;
68
+ /** Cancel is available only while the run is actively running. */
69
+ canCancel: boolean;
70
+ /** Re-run is available only for top-level EvaluationRun docs. */
71
+ canRerun: boolean;
72
+ /** Reason to show (e.g. as a disabled-item tooltip) when canRerun is false. */
73
+ rerunDisabledReason?: string;
74
+ /** Retry judgement: EvaluationRun, terminal, with >0 judge-failed cases. */
75
+ canRetryJudgement: boolean;
76
+ /** Reason to show when canRetryJudgement is false. */
77
+ retryJudgementDisabledReason?: string;
78
+ /** Number of judge-failed test cases (0 when not applicable/unknown). */
79
+ judgeFailedCount: number;
80
+ }
81
+ /**
82
+ * Minimum time a run must have been persisted before a Cancel request with
83
+ * no in-memory executor token is allowed to take the "zombie" fallback path
84
+ * (mark cancelled directly — see getRunActionVisibility callers in the
85
+ * cancel routes). Guards the narrow window right after a run is created:
86
+ * the doc is persisted (and therefore visible to a concurrent Cancel
87
+ * request) strictly before its executor registers its cancellation token,
88
+ * so a Cancel that lands in that gap would otherwise mark a run "cancelled"
89
+ * moments before its own executor starts making progress on it. A brand-new
90
+ * run is also the case the fallback is LEAST useful for — "zombie" (no
91
+ * executor anywhere) is far more plausible once a run has been running for
92
+ * a while than in its first couple of seconds.
93
+ */
94
+ export declare const ZOMBIE_CANCEL_MIN_AGE_MS = 5000;
95
+ /**
96
+ * True once a run is old enough that a tokenless Cancel request can safely
97
+ * assume its executor (if any) would already have registered a
98
+ * cancellation token — i.e. it's safe to treat "no token" as "no executor"
99
+ * rather than "executor hasn't started yet".
100
+ *
101
+ * NOTE — known limitation, not fixed by this check: cancellation tokens are
102
+ * tracked in an in-memory `Map` scoped to ONE server process. In a
103
+ * multi-process/clustered deployment, a Cancel request routed to a
104
+ * DIFFERENT process than the one executing the run will always find no
105
+ * token there, regardless of run age, and this zombie fallback will mark
106
+ * the run cancelled in storage even though it's alive and progressing on
107
+ * another process. This mirrors a pre-existing, documented constraint of
108
+ * this codebase's run-execution model (see AGENTS.md's "orphan-run
109
+ * recovery" notes: "active is tracked per-process"); fixing it for real
110
+ * needs the same heartbeat-based ownership (`run.heartbeatAt`) that doc
111
+ * already calls out as the eventual replacement. Out of scope here.
112
+ */
113
+ export declare function isOldEnoughForZombieCancel(createdAt: string | undefined, now?: number): boolean;
114
+ /**
115
+ * Compute the full action-visibility matrix for one run. Pure function —
116
+ * safe to call from both React components and server-side route validation
117
+ * so the two never drift.
118
+ */
119
+ export declare function getRunActionVisibility(run: RunLike | null | undefined): RunActionVisibility;
120
+ //# sourceMappingURL=runActions.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"runActions.d.ts","sourceRoot":"","sources":["../../runActions.ts"],"names":[],"mappings":"AAKA;;;;;;;;;;;;;;;;;;;;;;;;;GAyBG;AAEH,OAAO,KAAK,EAAE,YAAY,EAAE,aAAa,EAAE,MAAM,SAAS,CAAC;AAG3D,kFAAkF;AAClF,MAAM,MAAM,OAAO,GAAG,IAAI,CAAC,YAAY,GAAG,aAAa,EAAE,QAAQ,GAAG,SAAS,CAAC,GAAG;IAC/E,OAAO,CAAC,EAAE,MAAM,CAAC;CAClB,CAAC;AAEF;;;;;;;;;;GAUG;AACH,wBAAgB,eAAe,CAAC,GAAG,EAAE,OAAO,GAAG,IAAI,GAAG,SAAS,GAAG,GAAG,IAAI,aAAa,CAErF;AAED,sFAAsF;AACtF,wBAAgB,YAAY,CAAC,GAAG,EAAE,OAAO,GAAG,IAAI,GAAG,SAAS,GAAG,OAAO,CAErE;AAED,4EAA4E;AAC5E,wBAAgB,aAAa,CAAC,GAAG,EAAE,OAAO,GAAG,IAAI,GAAG,SAAS,GAAG,OAAO,CAEtE;AAED;;;;;;;;;;;;;;;GAeG;AACH,wBAAgB,gBAAgB,CAAC,GAAG,EAAE,OAAO,GAAG,IAAI,GAAG,SAAS,GAAG,MAAM,CAUxE;AAED,MAAM,WAAW,mBAAmB;IAClC,4DAA4D;IAC5D,SAAS,EAAE,OAAO,CAAC;IACnB,kEAAkE;IAClE,SAAS,EAAE,OAAO,CAAC;IACnB,iEAAiE;IACjE,QAAQ,EAAE,OAAO,CAAC;IAClB,+EAA+E;IAC/E,mBAAmB,CAAC,EAAE,MAAM,CAAC;IAC7B,4EAA4E;IAC5E,iBAAiB,EAAE,OAAO,CAAC;IAC3B,sDAAsD;IACtD,4BAA4B,CAAC,EAAE,MAAM,CAAC;IACtC,yEAAyE;IACzE,gBAAgB,EAAE,MAAM,CAAC;CAC1B;AAOD;;;;;;;;;;;;GAYG;AACH,eAAO,MAAM,wBAAwB,OAAO,CAAC;AAE7C;;;;;;;;;;;;;;;;;GAiBG;AACH,wBAAgB,0BAA0B,CAAC,SAAS,EAAE,MAAM,GAAG,SAAS,EAAE,GAAG,GAAE,MAAmB,GAAG,OAAO,CAI3G;AAED;;;;GAIG;AACH,wBAAgB,sBAAsB,CAAC,GAAG,EAAE,OAAO,GAAG,IAAI,GAAG,SAAS,GAAG,mBAAmB,CAuB3F"}
@@ -0,0 +1,130 @@
1
+ /*
2
+ * Copyright OpenSearch Contributors
3
+ * SPDX-License-Identifier: Apache-2.0
4
+ */
5
+ import { isEvaluationRun as isEvaluationRunDoc } from '../types/index.js';
6
+ /**
7
+ * True when `run` is a top-level EvaluationRun document (created via
8
+ * `POST /api/storage/evaluation-runs`), as opposed to a legacy
9
+ * benchmark-embedded BenchmarkRun (`benchmark.runs[]`). The two share a lot
10
+ * of shape but only EvaluationRun docs carry `docType: 'evaluation-run'` and
11
+ * support the rerun/retry-judgement endpoints.
12
+ *
13
+ * Null-tolerant wrapper over the typed predicate in `types/index.ts` (the
14
+ * single source of truth for the docType discriminator) — kept so callers
15
+ * holding a possibly-null run don't need their own guard.
16
+ */
17
+ export function isEvaluationRun(run) {
18
+ return !!run && isEvaluationRunDoc(run);
19
+ }
20
+ /** True while the run has an in-progress executor that a Cancel action could stop. */
21
+ export function isRunRunning(run) {
22
+ return run?.status === 'running';
23
+ }
24
+ /** True once a run has reached any terminal state (not running/pending). */
25
+ export function isRunTerminal(run) {
26
+ return !!run && (run.status === 'completed' || run.status === 'failed' || run.status === 'cancelled');
27
+ }
28
+ /**
29
+ * Count test cases where the AGENT finished (`status === 'completed'`) but
30
+ * the JUDGE produced no verdict (`passFailStatus` neither 'passed' nor
31
+ * 'failed') — the "errored" bucket of `lib/runStats` `bucketRunResults`
32
+ * (issue #242) and exactly the set `POST .../retry-judgement` (default
33
+ * `scope=errored`) will re-judge. Deliberately excludes:
34
+ * - `status !== 'completed'` (agent-failed/cancelled/pending cases — no
35
+ * trajectory to re-judge, or nothing ran).
36
+ * - a real 'failed' verdict — the judge DID run and graded the case; that
37
+ * is a legitimate result, not a judge failure (re-grading it is
38
+ * `scope=all`, opt-in from the inspector's dedicated button).
39
+ *
40
+ * `passFailStatus` isn't declared on `EvaluationRun['results']`'s static
41
+ * type (a pre-existing gap — evaluationRunner.ts writes it via an `as any`
42
+ * spread) so this reads it defensively.
43
+ */
44
+ export function countJudgeFailed(run) {
45
+ if (!run?.results)
46
+ return 0;
47
+ let count = 0;
48
+ for (const r of Object.values(run.results)) {
49
+ const result = r;
50
+ if (result.status !== 'completed')
51
+ continue;
52
+ if (result.passFailStatus === 'passed' || result.passFailStatus === 'failed')
53
+ continue;
54
+ count++;
55
+ }
56
+ return count;
57
+ }
58
+ const RERUN_NOT_SUPPORTED_REASON = "Re-run isn't available for legacy benchmark-embedded runs";
59
+ const RETRY_JUDGEMENT_NOT_SUPPORTED_REASON = "Retry judgement isn't available for legacy benchmark-embedded runs";
60
+ const RETRY_JUDGEMENT_STILL_RUNNING_REASON = 'Retry judgement is only available once the run finishes';
61
+ const RETRY_JUDGEMENT_NONE_FAILED_REASON = 'No judge-failed test cases to retry';
62
+ /**
63
+ * Minimum time a run must have been persisted before a Cancel request with
64
+ * no in-memory executor token is allowed to take the "zombie" fallback path
65
+ * (mark cancelled directly — see getRunActionVisibility callers in the
66
+ * cancel routes). Guards the narrow window right after a run is created:
67
+ * the doc is persisted (and therefore visible to a concurrent Cancel
68
+ * request) strictly before its executor registers its cancellation token,
69
+ * so a Cancel that lands in that gap would otherwise mark a run "cancelled"
70
+ * moments before its own executor starts making progress on it. A brand-new
71
+ * run is also the case the fallback is LEAST useful for — "zombie" (no
72
+ * executor anywhere) is far more plausible once a run has been running for
73
+ * a while than in its first couple of seconds.
74
+ */
75
+ export const ZOMBIE_CANCEL_MIN_AGE_MS = 5000;
76
+ /**
77
+ * True once a run is old enough that a tokenless Cancel request can safely
78
+ * assume its executor (if any) would already have registered a
79
+ * cancellation token — i.e. it's safe to treat "no token" as "no executor"
80
+ * rather than "executor hasn't started yet".
81
+ *
82
+ * NOTE — known limitation, not fixed by this check: cancellation tokens are
83
+ * tracked in an in-memory `Map` scoped to ONE server process. In a
84
+ * multi-process/clustered deployment, a Cancel request routed to a
85
+ * DIFFERENT process than the one executing the run will always find no
86
+ * token there, regardless of run age, and this zombie fallback will mark
87
+ * the run cancelled in storage even though it's alive and progressing on
88
+ * another process. This mirrors a pre-existing, documented constraint of
89
+ * this codebase's run-execution model (see AGENTS.md's "orphan-run
90
+ * recovery" notes: "active is tracked per-process"); fixing it for real
91
+ * needs the same heartbeat-based ownership (`run.heartbeatAt`) that doc
92
+ * already calls out as the eventual replacement. Out of scope here.
93
+ */
94
+ export function isOldEnoughForZombieCancel(createdAt, now = Date.now()) {
95
+ const created = createdAt ? Date.parse(createdAt) : NaN;
96
+ if (Number.isNaN(created))
97
+ return true; // no timestamp to compare against — don't block on it
98
+ return now - created >= ZOMBIE_CANCEL_MIN_AGE_MS;
99
+ }
100
+ /**
101
+ * Compute the full action-visibility matrix for one run. Pure function —
102
+ * safe to call from both React components and server-side route validation
103
+ * so the two never drift.
104
+ */
105
+ export function getRunActionVisibility(run) {
106
+ const evalRun = isEvaluationRun(run);
107
+ const running = isRunRunning(run);
108
+ const terminal = isRunTerminal(run);
109
+ const judgeFailedCount = evalRun ? countJudgeFailed(run) : 0;
110
+ const canRetryJudgement = evalRun && terminal && judgeFailedCount > 0;
111
+ let retryJudgementDisabledReason;
112
+ if (!canRetryJudgement) {
113
+ if (!evalRun)
114
+ retryJudgementDisabledReason = RETRY_JUDGEMENT_NOT_SUPPORTED_REASON;
115
+ else if (!terminal)
116
+ retryJudgementDisabledReason = RETRY_JUDGEMENT_STILL_RUNNING_REASON;
117
+ else
118
+ retryJudgementDisabledReason = RETRY_JUDGEMENT_NONE_FAILED_REASON;
119
+ }
120
+ return {
121
+ canDelete: true,
122
+ canCancel: running,
123
+ canRerun: evalRun,
124
+ rerunDisabledReason: evalRun ? undefined : RERUN_NOT_SUPPORTED_REASON,
125
+ canRetryJudgement,
126
+ retryJudgementDisabledReason,
127
+ judgeFailedCount,
128
+ };
129
+ }
130
+ //# sourceMappingURL=runActions.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"runActions.js","sourceRoot":"","sources":["../../runActions.ts"],"names":[],"mappings":"AAAA;;;GAGG;AA8BH,OAAO,EAAE,eAAe,IAAI,kBAAkB,EAAE,MAAM,SAAS,CAAC;AAOhE;;;;;;;;;;GAUG;AACH,MAAM,UAAU,eAAe,CAAC,GAA+B;IAC7D,OAAO,CAAC,CAAC,GAAG,IAAI,kBAAkB,CAAC,GAAmC,CAAC,CAAC;AAC1E,CAAC;AAED,sFAAsF;AACtF,MAAM,UAAU,YAAY,CAAC,GAA+B;IAC1D,OAAO,GAAG,EAAE,MAAM,KAAK,SAAS,CAAC;AACnC,CAAC;AAED,4EAA4E;AAC5E,MAAM,UAAU,aAAa,CAAC,GAA+B;IAC3D,OAAO,CAAC,CAAC,GAAG,IAAI,CAAC,GAAG,CAAC,MAAM,KAAK,WAAW,IAAI,GAAG,CAAC,MAAM,KAAK,QAAQ,IAAI,GAAG,CAAC,MAAM,KAAK,WAAW,CAAC,CAAC;AACxG,CAAC;AAED;;;;;;;;;;;;;;;GAeG;AACH,MAAM,UAAU,gBAAgB,CAAC,GAA+B;IAC9D,IAAI,CAAC,GAAG,EAAE,OAAO;QAAE,OAAO,CAAC,CAAC;IAC5B,IAAI,KAAK,GAAG,CAAC,CAAC;IACd,KAAK,MAAM,CAAC,IAAI,MAAM,CAAC,MAAM,CAAC,GAAG,CAAC,OAAO,CAAC,EAAE,CAAC;QAC3C,MAAM,MAAM,GAAG,CAAwD,CAAC;QACxE,IAAI,MAAM,CAAC,MAAM,KAAK,WAAW;YAAE,SAAS;QAC5C,IAAI,MAAM,CAAC,cAAc,KAAK,QAAQ,IAAI,MAAM,CAAC,cAAc,KAAK,QAAQ;YAAE,SAAS;QACvF,KAAK,EAAE,CAAC;IACV,CAAC;IACD,OAAO,KAAK,CAAC;AACf,CAAC;AAmBD,MAAM,0BAA0B,GAAG,2DAA2D,CAAC;AAC/F,MAAM,oCAAoC,GAAG,oEAAoE,CAAC;AAClH,MAAM,oCAAoC,GAAG,yDAAyD,CAAC;AACvG,MAAM,kCAAkC,GAAG,qCAAqC,CAAC;AAEjF;;;;;;;;;;;;GAYG;AACH,MAAM,CAAC,MAAM,wBAAwB,GAAG,IAAI,CAAC;AAE7C;;;;;;;;;;;;;;;;;GAiBG;AACH,MAAM,UAAU,0BAA0B,CAAC,SAA6B,EAAE,MAAc,IAAI,CAAC,GAAG,EAAE;IAChG,MAAM,OAAO,GAAG,SAAS,CAAC,CAAC,CAAC,IAAI,CAAC,KAAK,CAAC,SAAS,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC;IACxD,IAAI,MAAM,CAAC,KAAK,CAAC,OAAO,CAAC;QAAE,OAAO,IAAI,CAAC,CAAC,sDAAsD;IAC9F,OAAO,GAAG,GAAG,OAAO,IAAI,wBAAwB,CAAC;AACnD,CAAC;AAED;;;;GAIG;AACH,MAAM,UAAU,sBAAsB,CAAC,GAA+B;IACpE,MAAM,OAAO,GAAG,eAAe,CAAC,GAAG,CAAC,CAAC;IACrC,MAAM,OAAO,GAAG,YAAY,CAAC,GAAG,CAAC,CAAC;IAClC,MAAM,QAAQ,GAAG,aAAa,CAAC,GAAG,CAAC,CAAC;IACpC,MAAM,gBAAgB,GAAG,OAAO,CAAC,CAAC,CAAC,gBAAgB,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;IAE7D,MAAM,iBAAiB,GAAG,OAAO,IAAI,QAAQ,IAAI,gBAAgB,GAAG,CAAC,CAAC;IACtE,IAAI,4BAAgD,CAAC;IACrD,IAAI,CAAC,iBAAiB,EAAE,CAAC;QACvB,IAAI,CAAC,OAAO;YAAE,4BAA4B,GAAG,oCAAoC,CAAC;aAC7E,IAAI,CAAC,QAAQ;YAAE,4BAA4B,GAAG,oCAAoC,CAAC;;YACnF,4BAA4B,GAAG,kCAAkC,CAAC;IACzE,CAAC;IAED,OAAO;QACL,SAAS,EAAE,IAAI;QACf,SAAS,EAAE,OAAO;QAClB,QAAQ,EAAE,OAAO;QACjB,mBAAmB,EAAE,OAAO,CAAC,CAAC,CAAC,SAAS,CAAC,CAAC,CAAC,0BAA0B;QACrE,iBAAiB;QACjB,4BAA4B;QAC5B,gBAAgB;KACjB,CAAC;AACJ,CAAC"}
@@ -0,0 +1,86 @@
1
+ /**
2
+ * Deterministic (no-LLM) aggregation helpers for the run-report "insights"
3
+ * pane (RunInsightsPane.tsx), shown on the bare
4
+ * `/benchmarks/:benchmarkId/runs/:runId` route when no test case is
5
+ * selected — see owner feedback on goyamegh/run-report-redesign: "if no
6
+ * test case is selected, the right side can show an aggregated view ...
7
+ * why did the failing tests fail — something that is complete info."
8
+ *
9
+ * Everything here is a pure function over already-fetched data (report
10
+ * summaries + test-case categories). No network calls, no LLM calls — v1
11
+ * is explicitly deterministic-only per the product ask.
12
+ */
13
+ export interface CategoryStatusRow {
14
+ category: string;
15
+ /** Any ResultStatus value; only 'passed' / 'failed' / 'errored' are counted distinctly, everything else falls into the bar's `total` only (pending/running cases). */
16
+ status: string;
17
+ }
18
+ export interface CategoryBar {
19
+ category: string;
20
+ passed: number;
21
+ failed: number;
22
+ errored: number;
23
+ total: number;
24
+ }
25
+ /**
26
+ * Group rows by test-case category and tally pass/fail/errored/total.
27
+ * Deterministic order: largest category first, ties broken alphabetically.
28
+ */
29
+ export declare function computeCategoryBars(rows: CategoryStatusRow[]): CategoryBar[];
30
+ /**
31
+ * Normalize a judge-reasoning string down to its first sentence,
32
+ * lowercased, punctuation-stripped, whitespace-collapsed. Used both as the
33
+ * clustering input and as the theme's stable `key`.
34
+ */
35
+ export declare function normalizeReasoningKey(reasoning: string): string;
36
+ export interface FailureThemeInput {
37
+ testCaseId: string;
38
+ reasoning: string;
39
+ }
40
+ export interface FailureTheme {
41
+ /** Stable cluster key — the normalized first sentence of the theme's representative case. */
42
+ key: string;
43
+ count: number;
44
+ /** Trimmed, human-readable first sentence sampled from the theme's most common exact phrasing. */
45
+ sampleSnippet: string;
46
+ testCaseIds: string[];
47
+ }
48
+ /**
49
+ * Cluster failing test cases into "why they failed" themes using a
50
+ * deterministic, LLM-free heuristic: normalize each case's judge-reasoning
51
+ * first sentence, then union-find cases that share at least one contiguous
52
+ * N-word shingle. This is robust to minor paraphrasing (a judge saying
53
+ * "unable to retrieve" vs "failed to retrieve" the same underlying tool
54
+ * connectivity failure) while still keeping genuinely distinct failure
55
+ * modes (e.g. "missing required facts" vs "MCP server unavailable")
56
+ * separate, because they share no contiguous phrase.
57
+ *
58
+ * Verified against a real production run (418-verify, 64 failing cases):
59
+ * 57 of 64 connectivity-flavored reasonings collapse into ONE dominant
60
+ * theme; the remaining 7 ("Required facts evaluation: ...", a genuinely
61
+ * different failure shape) form a second, correctly separate theme.
62
+ *
63
+ * Output order: largest theme first, ties broken by the theme's
64
+ * lowest-sorting testCaseId (deterministic, no dependency on input order).
65
+ */
66
+ export declare function clusterFailureThemes(items: FailureThemeInput[], shingleSize?: number): FailureTheme[];
67
+ /**
68
+ * "Based on N of M failing cases" note shown under the theme list when the
69
+ * reasoning fetch was capped (RunInsightsPane caps at the first 100 failing
70
+ * cases). Returns null when nothing was capped (fetchedCount >= totalCount).
71
+ */
72
+ export declare function formatCappedNote(fetchedCount: number, totalFailingCount: number): string | null;
73
+ export interface RankedCase {
74
+ testCaseId: string;
75
+ value: number;
76
+ }
77
+ /**
78
+ * Top-N cases by a numeric value (duration, cost, ...), descending.
79
+ * Cases with a null/undefined/NaN value are excluded. Deterministic tie
80
+ * break: lower testCaseId first.
81
+ */
82
+ export declare function pickTopN(cases: {
83
+ testCaseId: string;
84
+ value: number | null | undefined;
85
+ }[], n: number): RankedCase[];
86
+ //# sourceMappingURL=runInsights.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"runInsights.d.ts","sourceRoot":"","sources":["../../runInsights.ts"],"names":[],"mappings":"AAKA;;;;;;;;;;;GAWG;AAIH,MAAM,WAAW,iBAAiB;IAChC,QAAQ,EAAE,MAAM,CAAC;IACjB,sKAAsK;IACtK,MAAM,EAAE,MAAM,CAAC;CAChB;AAED,MAAM,WAAW,WAAW;IAC1B,QAAQ,EAAE,MAAM,CAAC;IACjB,MAAM,EAAE,MAAM,CAAC;IACf,MAAM,EAAE,MAAM,CAAC;IACf,OAAO,EAAE,MAAM,CAAC;IAChB,KAAK,EAAE,MAAM,CAAC;CACf;AAID;;;GAGG;AACH,wBAAgB,mBAAmB,CAAC,IAAI,EAAE,iBAAiB,EAAE,GAAG,WAAW,EAAE,CAiB5E;AAID;;;;GAIG;AACH,wBAAgB,qBAAqB,CAAC,SAAS,EAAE,MAAM,GAAG,MAAM,CAM/D;AAqBD,MAAM,WAAW,iBAAiB;IAChC,UAAU,EAAE,MAAM,CAAC;IACnB,SAAS,EAAE,MAAM,CAAC;CACnB;AAED,MAAM,WAAW,YAAY;IAC3B,6FAA6F;IAC7F,GAAG,EAAE,MAAM,CAAC;IACZ,KAAK,EAAE,MAAM,CAAC;IACd,kGAAkG;IAClG,aAAa,EAAE,MAAM,CAAC;IACtB,WAAW,EAAE,MAAM,EAAE,CAAC;CACvB;AAOD;;;;;;;;;;;;;;;;;GAiBG;AACH,wBAAgB,oBAAoB,CAClC,KAAK,EAAE,iBAAiB,EAAE,EAC1B,WAAW,GAAE,MAA6B,GACzC,YAAY,EAAE,CAwEhB;AAED;;;;GAIG;AACH,wBAAgB,gBAAgB,CAAC,YAAY,EAAE,MAAM,EAAE,iBAAiB,EAAE,MAAM,GAAG,MAAM,GAAG,IAAI,CAG/F;AAID,MAAM,WAAW,UAAU;IACzB,UAAU,EAAE,MAAM,CAAC;IACnB,KAAK,EAAE,MAAM,CAAC;CACf;AAED;;;;GAIG;AACH,wBAAgB,QAAQ,CAAC,KAAK,EAAE;IAAE,UAAU,EAAE,MAAM,CAAC;IAAC,KAAK,EAAE,MAAM,GAAG,IAAI,GAAG,SAAS,CAAA;CAAE,EAAE,EAAE,CAAC,EAAE,MAAM,GAAG,UAAU,EAAE,CAKnH"}
@@ -0,0 +1,185 @@
1
+ /*
2
+ * Copyright OpenSearch Contributors
3
+ * SPDX-License-Identifier: Apache-2.0
4
+ */
5
+ const UNCATEGORIZED = 'Uncategorized';
6
+ /**
7
+ * Group rows by test-case category and tally pass/fail/errored/total.
8
+ * Deterministic order: largest category first, ties broken alphabetically.
9
+ */
10
+ export function computeCategoryBars(rows) {
11
+ const map = new Map();
12
+ for (const row of rows) {
13
+ const category = (row.category || '').trim() || UNCATEGORIZED;
14
+ let bar = map.get(category);
15
+ if (!bar) {
16
+ bar = { category, passed: 0, failed: 0, errored: 0, total: 0 };
17
+ map.set(category, bar);
18
+ }
19
+ bar.total++;
20
+ if (row.status === 'passed')
21
+ bar.passed++;
22
+ else if (row.status === 'errored')
23
+ bar.errored++;
24
+ else if (row.status === 'failed')
25
+ bar.failed++;
26
+ }
27
+ return Array.from(map.values()).sort((a, b) => b.total - a.total || a.category.localeCompare(b.category));
28
+ }
29
+ // ── Failure-theme clustering ───────────────────────────────────────────────
30
+ /**
31
+ * Normalize a judge-reasoning string down to its first sentence,
32
+ * lowercased, punctuation-stripped, whitespace-collapsed. Used both as the
33
+ * clustering input and as the theme's stable `key`.
34
+ */
35
+ export function normalizeReasoningKey(reasoning) {
36
+ const trimmed = (reasoning || '').trim();
37
+ if (!trimmed)
38
+ return '';
39
+ const sentenceMatch = trimmed.match(/^[^.!?\n]+[.!?]?/);
40
+ const firstSentence = (sentenceMatch ? sentenceMatch[0] : trimmed).toLowerCase();
41
+ return firstSentence.replace(/[^a-z0-9\s]/g, ' ').replace(/\s+/g, ' ').trim();
42
+ }
43
+ /** Returns the literal (non-lowercased) first sentence, trimmed — used as the human-facing sample snippet. */
44
+ function literalFirstSentence(reasoning) {
45
+ const trimmed = (reasoning || '').trim();
46
+ if (!trimmed)
47
+ return '';
48
+ const sentenceMatch = trimmed.match(/^[^.!?\n]+[.!?]?/);
49
+ return (sentenceMatch ? sentenceMatch[0] : trimmed).trim();
50
+ }
51
+ function shingles(normalized, size) {
52
+ const words = normalized.split(' ').filter(Boolean);
53
+ if (words.length === 0)
54
+ return [];
55
+ if (words.length <= size)
56
+ return [words.join(' ')];
57
+ const out = [];
58
+ for (let i = 0; i <= words.length - size; i++) {
59
+ out.push(words.slice(i, i + size).join(' '));
60
+ }
61
+ return out;
62
+ }
63
+ /** Contiguous-word window size used for near-duplicate matching. Paraphrases of the same failure ("agent was unable to retrieve..." vs "agent failed to retrieve...") reliably share at least one 6-word run even when their leading words differ. */
64
+ const DEFAULT_SHINGLE_SIZE = 6;
65
+ /** Reasoning first-sentences shorter than this many words don't carry enough signal to safely dedupe against unrelated failures — left as their own singleton theme rather than risk clustering unrelated short reasons together (e.g. "The agent failed to retrieve the required information."). */
66
+ const MIN_WORDS_FOR_SHINGLING = 3;
67
+ /**
68
+ * Cluster failing test cases into "why they failed" themes using a
69
+ * deterministic, LLM-free heuristic: normalize each case's judge-reasoning
70
+ * first sentence, then union-find cases that share at least one contiguous
71
+ * N-word shingle. This is robust to minor paraphrasing (a judge saying
72
+ * "unable to retrieve" vs "failed to retrieve" the same underlying tool
73
+ * connectivity failure) while still keeping genuinely distinct failure
74
+ * modes (e.g. "missing required facts" vs "MCP server unavailable")
75
+ * separate, because they share no contiguous phrase.
76
+ *
77
+ * Verified against a real production run (418-verify, 64 failing cases):
78
+ * 57 of 64 connectivity-flavored reasonings collapse into ONE dominant
79
+ * theme; the remaining 7 ("Required facts evaluation: ...", a genuinely
80
+ * different failure shape) form a second, correctly separate theme.
81
+ *
82
+ * Output order: largest theme first, ties broken by the theme's
83
+ * lowest-sorting testCaseId (deterministic, no dependency on input order).
84
+ */
85
+ export function clusterFailureThemes(items, shingleSize = DEFAULT_SHINGLE_SIZE) {
86
+ const n = items.length;
87
+ if (n === 0)
88
+ return [];
89
+ const parent = Array.from({ length: n }, (_, i) => i);
90
+ function find(i) {
91
+ while (parent[i] !== i) {
92
+ parent[i] = parent[parent[i]];
93
+ i = parent[i];
94
+ }
95
+ return i;
96
+ }
97
+ function union(a, b) {
98
+ const ra = find(a);
99
+ const rb = find(b);
100
+ if (ra !== rb)
101
+ parent[Math.max(ra, rb)] = Math.min(ra, rb);
102
+ }
103
+ const normalized = items.map(it => normalizeReasoningKey(it.reasoning));
104
+ const shingleBuckets = new Map(); // shingle -> first index seen
105
+ normalized.forEach((norm, idx) => {
106
+ const words = norm.split(' ').filter(Boolean);
107
+ if (words.length < MIN_WORDS_FOR_SHINGLING)
108
+ return;
109
+ for (const sh of shingles(norm, shingleSize)) {
110
+ const firstIdx = shingleBuckets.get(sh);
111
+ if (firstIdx === undefined)
112
+ shingleBuckets.set(sh, idx);
113
+ else
114
+ union(firstIdx, idx);
115
+ }
116
+ });
117
+ const groups = new Map();
118
+ for (let i = 0; i < n; i++) {
119
+ const root = find(i);
120
+ const arr = groups.get(root);
121
+ if (arr)
122
+ arr.push(i);
123
+ else
124
+ groups.set(root, [i]);
125
+ }
126
+ const themes = [];
127
+ for (const idxs of groups.values()) {
128
+ // Representative snippet: the most frequent literal first-sentence
129
+ // phrasing within the cluster (ties broken by first occurrence) - so a
130
+ // theme with 4 identical sentences + a handful of near-duplicate
131
+ // paraphrases surfaces the exact sentence, not an arbitrary member.
132
+ const counts = new Map();
133
+ for (const idx of idxs) {
134
+ const key = normalized[idx];
135
+ const snippet = literalFirstSentence(items[idx].reasoning);
136
+ const existing = counts.get(key);
137
+ if (existing)
138
+ existing.count++;
139
+ else
140
+ counts.set(key, { count: 1, firstIdx: idx, snippet });
141
+ }
142
+ let best = null;
143
+ for (const v of counts.values()) {
144
+ if (!best || v.count > best.count || (v.count === best.count && v.firstIdx < best.firstIdx))
145
+ best = v;
146
+ }
147
+ const testCaseIds = idxs.map(idx => items[idx].testCaseId);
148
+ themes.push({
149
+ key: normalized[idxs[0]] || `theme-${idxs[0]}`,
150
+ count: idxs.length,
151
+ sampleSnippet: best?.snippet || '',
152
+ testCaseIds,
153
+ });
154
+ }
155
+ themes.sort((a, b) => {
156
+ if (b.count !== a.count)
157
+ return b.count - a.count;
158
+ const aMin = [...a.testCaseIds].sort()[0] || '';
159
+ const bMin = [...b.testCaseIds].sort()[0] || '';
160
+ return aMin < bMin ? -1 : aMin > bMin ? 1 : 0;
161
+ });
162
+ return themes;
163
+ }
164
+ /**
165
+ * "Based on N of M failing cases" note shown under the theme list when the
166
+ * reasoning fetch was capped (RunInsightsPane caps at the first 100 failing
167
+ * cases). Returns null when nothing was capped (fetchedCount >= totalCount).
168
+ */
169
+ export function formatCappedNote(fetchedCount, totalFailingCount) {
170
+ if (totalFailingCount <= fetchedCount)
171
+ return null;
172
+ return `Based on ${fetchedCount} of ${totalFailingCount} failing cases`;
173
+ }
174
+ /**
175
+ * Top-N cases by a numeric value (duration, cost, ...), descending.
176
+ * Cases with a null/undefined/NaN value are excluded. Deterministic tie
177
+ * break: lower testCaseId first.
178
+ */
179
+ export function pickTopN(cases, n) {
180
+ return cases
181
+ .filter((c) => typeof c.value === 'number' && Number.isFinite(c.value))
182
+ .sort((a, b) => (b.value !== a.value ? b.value - a.value : a.testCaseId.localeCompare(b.testCaseId)))
183
+ .slice(0, n);
184
+ }
185
+ //# sourceMappingURL=runInsights.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"runInsights.js","sourceRoot":"","sources":["../../runInsights.ts"],"names":[],"mappings":"AAAA;;;GAGG;AA+BH,MAAM,aAAa,GAAG,eAAe,CAAC;AAEtC;;;GAGG;AACH,MAAM,UAAU,mBAAmB,CAAC,IAAyB;IAC3D,MAAM,GAAG,GAAG,IAAI,GAAG,EAAuB,CAAC;IAC3C,KAAK,MAAM,GAAG,IAAI,IAAI,EAAE,CAAC;QACvB,MAAM,QAAQ,GAAG,CAAC,GAAG,CAAC,QAAQ,IAAI,EAAE,CAAC,CAAC,IAAI,EAAE,IAAI,aAAa,CAAC;QAC9D,IAAI,GAAG,GAAG,GAAG,CAAC,GAAG,CAAC,QAAQ,CAAC,CAAC;QAC5B,IAAI,CAAC,GAAG,EAAE,CAAC;YACT,GAAG,GAAG,EAAE,QAAQ,EAAE,MAAM,EAAE,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,OAAO,EAAE,CAAC,EAAE,KAAK,EAAE,CAAC,EAAE,CAAC;YAC/D,GAAG,CAAC,GAAG,CAAC,QAAQ,EAAE,GAAG,CAAC,CAAC;QACzB,CAAC;QACD,GAAG,CAAC,KAAK,EAAE,CAAC;QACZ,IAAI,GAAG,CAAC,MAAM,KAAK,QAAQ;YAAE,GAAG,CAAC,MAAM,EAAE,CAAC;aACrC,IAAI,GAAG,CAAC,MAAM,KAAK,SAAS;YAAE,GAAG,CAAC,OAAO,EAAE,CAAC;aAC5C,IAAI,GAAG,CAAC,MAAM,KAAK,QAAQ;YAAE,GAAG,CAAC,MAAM,EAAE,CAAC;IACjD,CAAC;IACD,OAAO,KAAK,CAAC,IAAI,CAAC,GAAG,CAAC,MAAM,EAAE,CAAC,CAAC,IAAI,CAClC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,KAAK,GAAG,CAAC,CAAC,KAAK,IAAI,CAAC,CAAC,QAAQ,CAAC,aAAa,CAAC,CAAC,CAAC,QAAQ,CAAC,CACpE,CAAC;AACJ,CAAC;AAED,8EAA8E;AAE9E;;;;GAIG;AACH,MAAM,UAAU,qBAAqB,CAAC,SAAiB;IACrD,MAAM,OAAO,GAAG,CAAC,SAAS,IAAI,EAAE,CAAC,CAAC,IAAI,EAAE,CAAC;IACzC,IAAI,CAAC,OAAO;QAAE,OAAO,EAAE,CAAC;IACxB,MAAM,aAAa,GAAG,OAAO,CAAC,KAAK,CAAC,kBAAkB,CAAC,CAAC;IACxD,MAAM,aAAa,GAAG,CAAC,aAAa,CAAC,CAAC,CAAC,aAAa,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,WAAW,EAAE,CAAC;IACjF,OAAO,aAAa,CAAC,OAAO,CAAC,cAAc,EAAE,GAAG,CAAC,CAAC,OAAO,CAAC,MAAM,EAAE,GAAG,CAAC,CAAC,IAAI,EAAE,CAAC;AAChF,CAAC;AAED,8GAA8G;AAC9G,SAAS,oBAAoB,CAAC,SAAiB;IAC7C,MAAM,OAAO,GAAG,CAAC,SAAS,IAAI,EAAE,CAAC,CAAC,IAAI,EAAE,CAAC;IACzC,IAAI,CAAC,OAAO;QAAE,OAAO,EAAE,CAAC;IACxB,MAAM,aAAa,GAAG,OAAO,CAAC,KAAK,CAAC,kBAAkB,CAAC,CAAC;IACxD,OAAO,CAAC,aAAa,CAAC,CAAC,CAAC,aAAa,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,IAAI,EAAE,CAAC;AAC7D,CAAC;AAED,SAAS,QAAQ,CAAC,UAAkB,EAAE,IAAY;IAChD,MAAM,KAAK,GAAG,UAAU,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,MAAM,CAAC,OAAO,CAAC,CAAC;IACpD,IAAI,KAAK,CAAC,MAAM,KAAK,CAAC;QAAE,OAAO,EAAE,CAAC;IAClC,IAAI,KAAK,CAAC,MAAM,IAAI,IAAI;QAAE,OAAO,CAAC,KAAK,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,CAAC;IACnD,MAAM,GAAG,GAAa,EAAE,CAAC;IACzB,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,IAAI,KAAK,CAAC,MAAM,GAAG,IAAI,EAAE,CAAC,EAAE,EAAE,CAAC;QAC9C,GAAG,CAAC,IAAI,CAAC,KAAK,CAAC,KAAK,CAAC,CAAC,EAAE,CAAC,GAAG,IAAI,CAAC,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,CAAC;IAC/C,CAAC;IACD,OAAO,GAAG,CAAC;AACb,CAAC;AAgBD,sPAAsP;AACtP,MAAM,oBAAoB,GAAG,CAAC,CAAC;AAC/B,qSAAqS;AACrS,MAAM,uBAAuB,GAAG,CAAC,CAAC;AAElC;;;;;;;;;;;;;;;;;GAiBG;AACH,MAAM,UAAU,oBAAoB,CAClC,KAA0B,EAC1B,cAAsB,oBAAoB;IAE1C,MAAM,CAAC,GAAG,KAAK,CAAC,MAAM,CAAC;IACvB,IAAI,CAAC,KAAK,CAAC;QAAE,OAAO,EAAE,CAAC;IAEvB,MAAM,MAAM,GAAG,KAAK,CAAC,IAAI,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,CAAC;IACtD,SAAS,IAAI,CAAC,CAAS;QACrB,OAAO,MAAM,CAAC,CAAC,CAAC,KAAK,CAAC,EAAE,CAAC;YACvB,MAAM,CAAC,CAAC,CAAC,GAAG,MAAM,CAAC,MAAM,CAAC,CAAC,CAAC,CAAC,CAAC;YAC9B,CAAC,GAAG,MAAM,CAAC,CAAC,CAAC,CAAC;QAChB,CAAC;QACD,OAAO,CAAC,CAAC;IACX,CAAC;IACD,SAAS,KAAK,CAAC,CAAS,EAAE,CAAS;QACjC,MAAM,EAAE,GAAG,IAAI,CAAC,CAAC,CAAC,CAAC;QACnB,MAAM,EAAE,GAAG,IAAI,CAAC,CAAC,CAAC,CAAC;QACnB,IAAI,EAAE,KAAK,EAAE;YAAE,MAAM,CAAC,IAAI,CAAC,GAAG,CAAC,EAAE,EAAE,EAAE,CAAC,CAAC,GAAG,IAAI,CAAC,GAAG,CAAC,EAAE,EAAE,EAAE,CAAC,CAAC;IAC7D,CAAC;IAED,MAAM,UAAU,GAAG,KAAK,CAAC,GAAG,CAAC,EAAE,CAAC,EAAE,CAAC,qBAAqB,CAAC,EAAE,CAAC,SAAS,CAAC,CAAC,CAAC;IACxE,MAAM,cAAc,GAAG,IAAI,GAAG,EAAkB,CAAC,CAAC,8BAA8B;IAChF,UAAU,CAAC,OAAO,CAAC,CAAC,IAAI,EAAE,GAAG,EAAE,EAAE;QAC/B,MAAM,KAAK,GAAG,IAAI,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,MAAM,CAAC,OAAO,CAAC,CAAC;QAC9C,IAAI,KAAK,CAAC,MAAM,GAAG,uBAAuB;YAAE,OAAO;QACnD,KAAK,MAAM,EAAE,IAAI,QAAQ,CAAC,IAAI,EAAE,WAAW,CAAC,EAAE,CAAC;YAC7C,MAAM,QAAQ,GAAG,cAAc,CAAC,GAAG,CAAC,EAAE,CAAC,CAAC;YACxC,IAAI,QAAQ,KAAK,SAAS;gBAAE,cAAc,CAAC,GAAG,CAAC,EAAE,EAAE,GAAG,CAAC,CAAC;;gBACnD,KAAK,CAAC,QAAQ,EAAE,GAAG,CAAC,CAAC;QAC5B,CAAC;IACH,CAAC,CAAC,CAAC;IAEH,MAAM,MAAM,GAAG,IAAI,GAAG,EAAoB,CAAC;IAC3C,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC;QAC3B,MAAM,IAAI,GAAG,IAAI,CAAC,CAAC,CAAC,CAAC;QACrB,MAAM,GAAG,GAAG,MAAM,CAAC,GAAG,CAAC,IAAI,CAAC,CAAC;QAC7B,IAAI,GAAG;YAAE,GAAG,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC;;YAChB,MAAM,CAAC,GAAG,CAAC,IAAI,EAAE,CAAC,CAAC,CAAC,CAAC,CAAC;IAC7B,CAAC;IAED,MAAM,MAAM,GAAmB,EAAE,CAAC;IAClC,KAAK,MAAM,IAAI,IAAI,MAAM,CAAC,MAAM,EAAE,EAAE,CAAC;QACnC,mEAAmE;QACnE,uEAAuE;QACvE,iEAAiE;QACjE,oEAAoE;QACpE,MAAM,MAAM,GAAG,IAAI,GAAG,EAAgE,CAAC;QACvF,KAAK,MAAM,GAAG,IAAI,IAAI,EAAE,CAAC;YACvB,MAAM,GAAG,GAAG,UAAU,CAAC,GAAG,CAAC,CAAC;YAC5B,MAAM,OAAO,GAAG,oBAAoB,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,SAAS,CAAC,CAAC;YAC3D,MAAM,QAAQ,GAAG,MAAM,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC;YACjC,IAAI,QAAQ;gBAAE,QAAQ,CAAC,KAAK,EAAE,CAAC;;gBAC1B,MAAM,CAAC,GAAG,CAAC,GAAG,EAAE,EAAE,KAAK,EAAE,CAAC,EAAE,QAAQ,EAAE,GAAG,EAAE,OAAO,EAAE,CAAC,CAAC;QAC7D,CAAC;QACD,IAAI,IAAI,GAAgE,IAAI,CAAC;QAC7E,KAAK,MAAM,CAAC,IAAI,MAAM,CAAC,MAAM,EAAE,EAAE,CAAC;YAChC,IAAI,CAAC,IAAI,IAAI,CAAC,CAAC,KAAK,GAAG,IAAI,CAAC,KAAK,IAAI,CAAC,CAAC,CAAC,KAAK,KAAK,IAAI,CAAC,KAAK,IAAI,CAAC,CAAC,QAAQ,GAAG,IAAI,CAAC,QAAQ,CAAC;gBAAE,IAAI,GAAG,CAAC,CAAC;QACxG,CAAC;QACD,MAAM,WAAW,GAAG,IAAI,CAAC,GAAG,CAAC,GAAG,CAAC,EAAE,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,UAAU,CAAC,CAAC;QAC3D,MAAM,CAAC,IAAI,CAAC;YACV,GAAG,EAAE,UAAU,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC,IAAI,SAAS,IAAI,CAAC,CAAC,CAAC,EAAE;YAC9C,KAAK,EAAE,IAAI,CAAC,MAAM;YAClB,aAAa,EAAE,IAAI,EAAE,OAAO,IAAI,EAAE;YAClC,WAAW;SACZ,CAAC,CAAC;IACL,CAAC;IAED,MAAM,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE;QACnB,IAAI,CAAC,CAAC,KAAK,KAAK,CAAC,CAAC,KAAK;YAAE,OAAO,CAAC,CAAC,KAAK,GAAG,CAAC,CAAC,KAAK,CAAC;QAClD,MAAM,IAAI,GAAG,CAAC,GAAG,CAAC,CAAC,WAAW,CAAC,CAAC,IAAI,EAAE,CAAC,CAAC,CAAC,IAAI,EAAE,CAAC;QAChD,MAAM,IAAI,GAAG,CAAC,GAAG,CAAC,CAAC,WAAW,CAAC,CAAC,IAAI,EAAE,CAAC,CAAC,CAAC,IAAI,EAAE,CAAC;QAChD,OAAO,IAAI,GAAG,IAAI,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,IAAI,GAAG,IAAI,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;IAChD,CAAC,CAAC,CAAC;IACH,OAAO,MAAM,CAAC;AAChB,CAAC;AAED;;;;GAIG;AACH,MAAM,UAAU,gBAAgB,CAAC,YAAoB,EAAE,iBAAyB;IAC9E,IAAI,iBAAiB,IAAI,YAAY;QAAE,OAAO,IAAI,CAAC;IACnD,OAAO,YAAY,YAAY,OAAO,iBAAiB,gBAAgB,CAAC;AAC1E,CAAC;AASD;;;;GAIG;AACH,MAAM,UAAU,QAAQ,CAAC,KAAiE,EAAE,CAAS;IACnG,OAAO,KAAK;SACT,MAAM,CAAC,CAAC,CAAC,EAAmB,EAAE,CAAC,OAAO,CAAC,CAAC,KAAK,KAAK,QAAQ,IAAI,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC;SACvF,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,CAAC,KAAK,KAAK,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC,KAAK,GAAG,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC,UAAU,CAAC,aAAa,CAAC,CAAC,CAAC,UAAU,CAAC,CAAC,CAAC;SACpG,KAAK,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC;AACjB,CAAC"}
@@ -0,0 +1,29 @@
1
+ /**
2
+ * Validation for renaming an evaluation run (PATCH /api/storage/evaluation-runs/:id
3
+ * `{ name }`). This is the server route's authoritative check.
4
+ *
5
+ * Rename is intentionally narrow: it only ever touches `name`. No version bump,
6
+ * no stats recompute — this file has no opinion on anything but the string.
7
+ */
8
+ /** Renamed evaluation runs are capped at this length (arbitrary but generous —
9
+ * long enough for any real run name, short enough to keep list/table UIs sane). */
10
+ export declare const MAX_RUN_NAME_LENGTH = 200;
11
+ export type RunNameValidation = {
12
+ ok: true;
13
+ value: string;
14
+ } | {
15
+ ok: false;
16
+ error: string;
17
+ };
18
+ /**
19
+ * Validate + normalize a candidate evaluation-run name.
20
+ *
21
+ * - Must be a string.
22
+ * - Trimmed value must be non-empty.
23
+ * - Trimmed value must be at most {@link MAX_RUN_NAME_LENGTH} characters.
24
+ *
25
+ * Returns the trimmed value on success so callers persist a normalized name
26
+ * (no leading/trailing whitespace) rather than the raw input.
27
+ */
28
+ export declare function validateRunNameUpdate(name: unknown): RunNameValidation;
29
+ //# sourceMappingURL=runName.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"runName.d.ts","sourceRoot":"","sources":["../../runName.ts"],"names":[],"mappings":"AAKA;;;;;;GAMG;AAEH;oFACoF;AACpF,eAAO,MAAM,mBAAmB,MAAM,CAAC;AAEvC,MAAM,MAAM,iBAAiB,GACzB;IAAE,EAAE,EAAE,IAAI,CAAC;IAAC,KAAK,EAAE,MAAM,CAAA;CAAE,GAC3B;IAAE,EAAE,EAAE,KAAK,CAAC;IAAC,KAAK,EAAE,MAAM,CAAA;CAAE,CAAC;AAEjC;;;;;;;;;GASG;AACH,wBAAgB,qBAAqB,CAAC,IAAI,EAAE,OAAO,GAAG,iBAAiB,CAYtE"}
@@ -0,0 +1,38 @@
1
+ /*
2
+ * Copyright OpenSearch Contributors
3
+ * SPDX-License-Identifier: Apache-2.0
4
+ */
5
+ /**
6
+ * Validation for renaming an evaluation run (PATCH /api/storage/evaluation-runs/:id
7
+ * `{ name }`). This is the server route's authoritative check.
8
+ *
9
+ * Rename is intentionally narrow: it only ever touches `name`. No version bump,
10
+ * no stats recompute — this file has no opinion on anything but the string.
11
+ */
12
+ /** Renamed evaluation runs are capped at this length (arbitrary but generous —
13
+ * long enough for any real run name, short enough to keep list/table UIs sane). */
14
+ export const MAX_RUN_NAME_LENGTH = 200;
15
+ /**
16
+ * Validate + normalize a candidate evaluation-run name.
17
+ *
18
+ * - Must be a string.
19
+ * - Trimmed value must be non-empty.
20
+ * - Trimmed value must be at most {@link MAX_RUN_NAME_LENGTH} characters.
21
+ *
22
+ * Returns the trimmed value on success so callers persist a normalized name
23
+ * (no leading/trailing whitespace) rather than the raw input.
24
+ */
25
+ export function validateRunNameUpdate(name) {
26
+ if (typeof name !== 'string') {
27
+ return { ok: false, error: 'name must be a string' };
28
+ }
29
+ const trimmed = name.trim();
30
+ if (!trimmed) {
31
+ return { ok: false, error: 'name must not be empty' };
32
+ }
33
+ if (trimmed.length > MAX_RUN_NAME_LENGTH) {
34
+ return { ok: false, error: `name must be ${MAX_RUN_NAME_LENGTH} characters or fewer` };
35
+ }
36
+ return { ok: true, value: trimmed };
37
+ }
38
+ //# sourceMappingURL=runName.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"runName.js","sourceRoot":"","sources":["../../runName.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAEH;;;;;;GAMG;AAEH;oFACoF;AACpF,MAAM,CAAC,MAAM,mBAAmB,GAAG,GAAG,CAAC;AAMvC;;;;;;;;;GASG;AACH,MAAM,UAAU,qBAAqB,CAAC,IAAa;IACjD,IAAI,OAAO,IAAI,KAAK,QAAQ,EAAE,CAAC;QAC7B,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,uBAAuB,EAAE,CAAC;IACvD,CAAC;IACD,MAAM,OAAO,GAAG,IAAI,CAAC,IAAI,EAAE,CAAC;IAC5B,IAAI,CAAC,OAAO,EAAE,CAAC;QACb,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,wBAAwB,EAAE,CAAC;IACxD,CAAC;IACD,IAAI,OAAO,CAAC,MAAM,GAAG,mBAAmB,EAAE,CAAC;QACzC,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,gBAAgB,mBAAmB,sBAAsB,EAAE,CAAC;IACzF,CAAC;IACD,OAAO,EAAE,EAAE,EAAE,IAAI,EAAE,KAAK,EAAE,OAAO,EAAE,CAAC;AACtC,CAAC"}
@@ -0,0 +1,16 @@
1
+ /**
2
+ * Canonical run-report path for a run surfaced on the compare page.
3
+ *
4
+ * Benchmark runs (embedded in a benchmark's `runs[]`) resolve at
5
+ * `/evaluations/benchmarks/:benchmarkId/runs/:runId`; everything else —
6
+ * ad-hoc / SDK eval-runs, or runs whose `benchmarkId` is merely a label — goes
7
+ * to the bare `/evaluations/runs/:runId` route (the benchmark route would
8
+ * 404/redirect for those, and the bare route 404s for benchmark run ids).
9
+ *
10
+ * Shared by the scoreboard's run-name link, its "Open run" icon, and the
11
+ * per-case table's run headers so the three can never disagree. Callers pass
12
+ * the benchmarkId from `ComparisonPage`'s `runBenchmarkIdById` map (which is
13
+ * `undefined` unless the run is a real benchmark member).
14
+ */
15
+ export declare const runReportPath: (runId: string, benchmarkId?: string) => string;
16
+ //# sourceMappingURL=runReportPath.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"runReportPath.d.ts","sourceRoot":"","sources":["../../runReportPath.ts"],"names":[],"mappings":"AAKA;;;;;;;;;;;;;GAaG;AACH,eAAO,MAAM,aAAa,GAAI,OAAO,MAAM,EAAE,cAAc,MAAM,KAAG,MAGd,CAAC"}