@opensearch-project/agent-health 0.6.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. package/cli/dist/index.js +722 -330
  2. package/dist/assets/index-D-Np_l_T.js +246 -0
  3. package/dist/assets/index-vZt9QZKf.css +1 -0
  4. package/dist/index.html +2 -2
  5. package/docs/CLI.md +138 -1
  6. package/docs/CONNECTORS.md +1 -1
  7. package/docs/SDK.md +126 -2
  8. package/docs/skills/add-connector/SKILL.md +5 -1
  9. package/lib/dist/lib/agentTrends.d.ts +210 -0
  10. package/lib/dist/lib/agentTrends.d.ts.map +1 -0
  11. package/lib/dist/lib/agentTrends.js +360 -0
  12. package/lib/dist/lib/agentTrends.js.map +1 -0
  13. package/lib/dist/lib/benchmarkCaseReview.d.ts +114 -0
  14. package/lib/dist/lib/benchmarkCaseReview.d.ts.map +1 -0
  15. package/lib/dist/lib/benchmarkCaseReview.js +177 -0
  16. package/lib/dist/lib/benchmarkCaseReview.js.map +1 -0
  17. package/lib/dist/lib/benchmarkRunsTable.d.ts +109 -0
  18. package/lib/dist/lib/benchmarkRunsTable.d.ts.map +1 -0
  19. package/lib/dist/lib/benchmarkRunsTable.js +212 -0
  20. package/lib/dist/lib/benchmarkRunsTable.js.map +1 -0
  21. package/lib/dist/lib/comparisonInsights.d.ts +49 -2
  22. package/lib/dist/lib/comparisonInsights.d.ts.map +1 -1
  23. package/lib/dist/lib/comparisonInsights.js +65 -7
  24. package/lib/dist/lib/comparisonInsights.js.map +1 -1
  25. package/lib/dist/lib/config/loader.d.ts.map +1 -1
  26. package/lib/dist/lib/config/loader.js +11 -1
  27. package/lib/dist/lib/config/loader.js.map +1 -1
  28. package/lib/dist/lib/dashboardMetrics.d.ts +11 -2
  29. package/lib/dist/lib/dashboardMetrics.d.ts.map +1 -1
  30. package/lib/dist/lib/dashboardMetrics.js +38 -3
  31. package/lib/dist/lib/dashboardMetrics.js.map +1 -1
  32. package/lib/dist/lib/evaluationRerun.d.ts +39 -0
  33. package/lib/dist/lib/evaluationRerun.d.ts.map +1 -1
  34. package/lib/dist/lib/evaluationRerun.js +49 -0
  35. package/lib/dist/lib/evaluationRerun.js.map +1 -1
  36. package/lib/dist/lib/judgeFailureSummary.d.ts +66 -0
  37. package/lib/dist/lib/judgeFailureSummary.d.ts.map +1 -0
  38. package/lib/dist/lib/judgeFailureSummary.js +68 -0
  39. package/lib/dist/lib/judgeFailureSummary.js.map +1 -0
  40. package/lib/dist/lib/judgeStrategies.d.ts +108 -0
  41. package/lib/dist/lib/judgeStrategies.d.ts.map +1 -0
  42. package/lib/dist/lib/judgeStrategies.js +135 -0
  43. package/lib/dist/lib/judgeStrategies.js.map +1 -0
  44. package/lib/dist/lib/matchers/expect.d.ts +21 -1
  45. package/lib/dist/lib/matchers/expect.d.ts.map +1 -1
  46. package/lib/dist/lib/matchers/expect.js +51 -0
  47. package/lib/dist/lib/matchers/expect.js.map +1 -1
  48. package/lib/dist/lib/matchers/judgeAccessor.d.ts +4 -0
  49. package/lib/dist/lib/matchers/judgeAccessor.d.ts.map +1 -1
  50. package/lib/dist/lib/matchers/judgeAccessor.js +13 -2
  51. package/lib/dist/lib/matchers/judgeAccessor.js.map +1 -1
  52. package/lib/dist/lib/matchers/judgeReasoningParse.d.ts +49 -0
  53. package/lib/dist/lib/matchers/judgeReasoningParse.d.ts.map +1 -0
  54. package/lib/dist/lib/matchers/judgeReasoningParse.js +128 -0
  55. package/lib/dist/lib/matchers/judgeReasoningParse.js.map +1 -0
  56. package/lib/dist/lib/matchers/types.d.ts +25 -0
  57. package/lib/dist/lib/matchers/types.d.ts.map +1 -1
  58. package/lib/dist/lib/resolveCanonicalRun.d.ts +23 -0
  59. package/lib/dist/lib/resolveCanonicalRun.d.ts.map +1 -0
  60. package/lib/dist/lib/resolveCanonicalRun.js +26 -0
  61. package/lib/dist/lib/resolveCanonicalRun.js.map +1 -0
  62. package/lib/dist/lib/runActions.d.ts +120 -0
  63. package/lib/dist/lib/runActions.d.ts.map +1 -0
  64. package/lib/dist/lib/runActions.js +130 -0
  65. package/lib/dist/lib/runActions.js.map +1 -0
  66. package/lib/dist/lib/runInsights.d.ts +86 -0
  67. package/lib/dist/lib/runInsights.d.ts.map +1 -0
  68. package/lib/dist/lib/runInsights.js +185 -0
  69. package/lib/dist/lib/runInsights.js.map +1 -0
  70. package/lib/dist/lib/runName.d.ts +29 -0
  71. package/lib/dist/lib/runName.d.ts.map +1 -0
  72. package/lib/dist/lib/runName.js +38 -0
  73. package/lib/dist/lib/runName.js.map +1 -0
  74. package/lib/dist/lib/runReportPath.d.ts +16 -0
  75. package/lib/dist/lib/runReportPath.d.ts.map +1 -0
  76. package/lib/dist/lib/runReportPath.js +22 -0
  77. package/lib/dist/lib/runReportPath.js.map +1 -0
  78. package/lib/dist/lib/runSort.d.ts +27 -0
  79. package/lib/dist/lib/runSort.d.ts.map +1 -0
  80. package/lib/dist/lib/runSort.js +31 -0
  81. package/lib/dist/lib/runSort.js.map +1 -0
  82. package/lib/dist/lib/runStats.d.ts +86 -6
  83. package/lib/dist/lib/runStats.d.ts.map +1 -1
  84. package/lib/dist/lib/runStats.js +170 -19
  85. package/lib/dist/lib/runStats.js.map +1 -1
  86. package/lib/dist/lib/testCases/define.d.ts.map +1 -1
  87. package/lib/dist/lib/testCases/define.js +93 -46
  88. package/lib/dist/lib/testCases/define.js.map +1 -1
  89. package/lib/dist/lib/testCases/judge.d.ts.map +1 -1
  90. package/lib/dist/lib/testCases/judge.js +22 -4
  91. package/lib/dist/lib/testCases/judge.js.map +1 -1
  92. package/lib/dist/lib/testCases/loader.d.ts +24 -0
  93. package/lib/dist/lib/testCases/loader.d.ts.map +1 -1
  94. package/lib/dist/lib/testCases/loader.js +253 -36
  95. package/lib/dist/lib/testCases/loader.js.map +1 -1
  96. package/lib/dist/lib/trajectoryStepDisplay.d.ts +25 -0
  97. package/lib/dist/lib/trajectoryStepDisplay.d.ts.map +1 -0
  98. package/lib/dist/lib/trajectoryStepDisplay.js +42 -0
  99. package/lib/dist/lib/trajectoryStepDisplay.js.map +1 -0
  100. package/lib/dist/lib/utils.d.ts +19 -0
  101. package/lib/dist/lib/utils.d.ts.map +1 -1
  102. package/lib/dist/lib/utils.js +27 -0
  103. package/lib/dist/lib/utils.js.map +1 -1
  104. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.d.ts +68 -32
  105. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.d.ts.map +1 -1
  106. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.js +148 -125
  107. package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.js.map +1 -1
  108. package/lib/dist/services/connectors/kiro/KiroConnector.d.ts +21 -13
  109. package/lib/dist/services/connectors/kiro/KiroConnector.d.ts.map +1 -1
  110. package/lib/dist/services/connectors/kiro/KiroConnector.js +22 -25
  111. package/lib/dist/services/connectors/kiro/KiroConnector.js.map +1 -1
  112. package/lib/dist/services/connectors/pi/PiConnector.d.ts +49 -10
  113. package/lib/dist/services/connectors/pi/PiConnector.d.ts.map +1 -1
  114. package/lib/dist/services/connectors/pi/PiConnector.js +102 -86
  115. package/lib/dist/services/connectors/pi/PiConnector.js.map +1 -1
  116. package/lib/dist/services/connectors/subprocess/SubprocessConnector.d.ts +59 -14
  117. package/lib/dist/services/connectors/subprocess/SubprocessConnector.d.ts.map +1 -1
  118. package/lib/dist/services/connectors/subprocess/SubprocessConnector.js +87 -61
  119. package/lib/dist/services/connectors/subprocess/SubprocessConnector.js.map +1 -1
  120. package/lib/dist/services/evaluation/bedrockJudge.d.ts +11 -0
  121. package/lib/dist/services/evaluation/bedrockJudge.d.ts.map +1 -1
  122. package/lib/dist/services/evaluation/bedrockJudge.js +2 -0
  123. package/lib/dist/services/evaluation/bedrockJudge.js.map +1 -1
  124. package/lib/dist/services/evaluation/index.d.ts +18 -1
  125. package/lib/dist/services/evaluation/index.d.ts.map +1 -1
  126. package/lib/dist/services/evaluation/index.js +156 -19
  127. package/lib/dist/services/evaluation/index.js.map +1 -1
  128. package/lib/dist/services/metrics.d.ts +55 -0
  129. package/lib/dist/services/metrics.d.ts.map +1 -0
  130. package/lib/dist/services/metrics.js +89 -0
  131. package/lib/dist/services/metrics.js.map +1 -0
  132. package/lib/dist/services/storage/asyncBenchmarkStorage.d.ts.map +1 -1
  133. package/lib/dist/services/storage/asyncBenchmarkStorage.js +16 -0
  134. package/lib/dist/services/storage/asyncBenchmarkStorage.js.map +1 -1
  135. package/lib/dist/services/storage/asyncRunStorage.d.ts +11 -0
  136. package/lib/dist/services/storage/asyncRunStorage.d.ts.map +1 -1
  137. package/lib/dist/services/storage/asyncRunStorage.js +49 -2
  138. package/lib/dist/services/storage/asyncRunStorage.js.map +1 -1
  139. package/lib/dist/services/storage/asyncTestCaseStorage.d.ts +1 -1
  140. package/lib/dist/services/storage/asyncTestCaseStorage.d.ts.map +1 -1
  141. package/lib/dist/services/storage/asyncTestCaseStorage.js +2 -1
  142. package/lib/dist/services/storage/asyncTestCaseStorage.js.map +1 -1
  143. package/lib/dist/services/storage/opensearchClient.d.ts +12 -0
  144. package/lib/dist/services/storage/opensearchClient.d.ts.map +1 -1
  145. package/lib/dist/services/storage/opensearchClient.js.map +1 -1
  146. package/lib/dist/services/traces/browserRecovery.d.ts.map +1 -1
  147. package/lib/dist/services/traces/browserRecovery.js +3 -0
  148. package/lib/dist/services/traces/browserRecovery.js.map +1 -1
  149. package/lib/dist/services/traces/index.d.ts +8 -1
  150. package/lib/dist/services/traces/index.d.ts.map +1 -1
  151. package/lib/dist/services/traces/index.js +33 -12
  152. package/lib/dist/services/traces/index.js.map +1 -1
  153. package/lib/dist/services/traces/judgeAgentsHints.d.ts +98 -3
  154. package/lib/dist/services/traces/judgeAgentsHints.d.ts.map +1 -1
  155. package/lib/dist/services/traces/judgeAgentsHints.js +144 -3
  156. package/lib/dist/services/traces/judgeAgentsHints.js.map +1 -1
  157. package/lib/dist/services/traces/spansToTrajectory.js +4 -4
  158. package/lib/dist/services/traces/spansToTrajectory.js.map +1 -1
  159. package/lib/dist/services/traces/tracePoller.d.ts.map +1 -1
  160. package/lib/dist/services/traces/tracePoller.js +17 -20
  161. package/lib/dist/services/traces/tracePoller.js.map +1 -1
  162. package/lib/dist/services/traces/trajectoryMerge.d.ts +79 -0
  163. package/lib/dist/services/traces/trajectoryMerge.d.ts.map +1 -0
  164. package/lib/dist/services/traces/trajectoryMerge.js +109 -0
  165. package/lib/dist/services/traces/trajectoryMerge.js.map +1 -0
  166. package/lib/dist/types/index.d.ts +195 -2
  167. package/lib/dist/types/index.d.ts.map +1 -1
  168. package/lib/dist/types/index.js +24 -0
  169. package/lib/dist/types/index.js.map +1 -1
  170. package/package.json +6 -5
  171. package/server/dist/app.js +3398 -926
  172. package/server/dist/index.js +3401 -929
  173. package/dist/assets/index-BfxtxmKc.css +0 -1
  174. package/dist/assets/index-CrjAfDHu.js +0 -243
@@ -0,0 +1,128 @@
1
+ /*
2
+ * Copyright OpenSearch Contributors
3
+ * SPDX-License-Identifier: Apache-2.0
4
+ */
5
+ /** Map a matched verdict keyword onto the canonical kind. */
6
+ function verdictKind(keyword) {
7
+ const k = keyword.toUpperCase();
8
+ if (k.includes('CONTRADICT') && !k.includes('MISSING'))
9
+ return 'contradicted';
10
+ if (k.includes('MISSING') || k.includes('NOT STATED'))
11
+ return 'missing';
12
+ if (k.includes('PARTIAL'))
13
+ return 'partial';
14
+ return 'stated';
15
+ }
16
+ // Verdict keywords the judge writes inline after a fact. Order matters:
17
+ // longest / most specific first so "MISSING/CONTRADICTED" isn't split.
18
+ const VERDICT_KEYWORD = /(MISSING\s*\/\s*CONTRADICTED|FULLY\s+STATED|PARTIALLY\s+STATED|NOT\s+STATED|CONTRADICTED|MISSING|PARTIAL(?:LY)?)/i;
19
+ // Parse at most this much reasoning — real verdicts are a few KB; a hard cap
20
+ // bounds regex work on pathological inputs (defense against backtracking
21
+ // blowups on adversarial multi-hundred-KB strings).
22
+ const MAX_PARSE_CHARS = 20_000;
23
+ // A fact item: list marker (`1.`, `1)`, `**Fact 1:`, `Required fact 1 (`),
24
+ // then the fact text (quoted or plain), a separator (—, -, :, en-dash, `):`),
25
+ // then the verdict keyword. Fact text is capped to keep matches sane.
26
+ // NOTE: deliberately no lookbehind — constructed-at-import regexes with
27
+ // lookbehind hard-crash report rendering on older WebKit. The leading
28
+ // whitespace/newline is consumed instead (harmless: markers never overlap).
29
+ const FACT_ITEM = new RegExp(String.raw `(?:^|[\s\n])` + // start of string or after whitespace (inline numbered lists)
30
+ String.raw `(?:\*\*)?(?:Required\s+fact\s+\d+|Fact\s+\d+|\d+)\s*[.):\u2013\u2014-]?\s*` + // marker
31
+ String.raw `(?:\*\*)?\s*` +
32
+ String.raw `['‘"“(]?(.{4,240}?)['’"”)]?` + // fact text (lazy)
33
+ String.raw `(?:\*\*)?\s*[\u2013\u2014:(\u2015-]+\s*(?:\*\*)?\s*` + // separator
34
+ VERDICT_KEYWORD.source + // verdict
35
+ String.raw `(?:\s+stated)?` + // "PARTIALLY stated"
36
+ String.raw `[.!]?\s*` +
37
+ // Trailing note: lazy, stopped by the NEXT numbered/bold fact item on the
38
+ // same line (inline lists put every fact in one paragraph) or line end.
39
+ String.raw `([^\n]*?)(?=\s\d+[.)]\s*['‘"“(*]|\s\*\*(?:Required\s+fact|Fact)|\n|$)`, 'gi');
40
+ /**
41
+ * Extract per-required-fact verdicts from judge reasoning prose.
42
+ * Returns `[]` when nothing that looks like a fact list is present —
43
+ * callers must fall back to showing the raw reasoning.
44
+ */
45
+ export function parseFactVerdicts(reasoning) {
46
+ if (!reasoning || reasoning.length < 20)
47
+ return [];
48
+ const text = reasoning.slice(0, MAX_PARSE_CHARS);
49
+ const out = [];
50
+ const seen = new Set();
51
+ for (const m of text.matchAll(FACT_ITEM)) {
52
+ const factRaw = (m[1] ?? '').trim();
53
+ const keyword = m[2] ?? '';
54
+ // Guard against summary-phrase false positives like
55
+ // "4 facts fully stated (1.0 each)": require real fact text, not a
56
+ // recap that itself talks about facts/statements in aggregate.
57
+ if (!factRaw || factRaw.length < 8)
58
+ continue;
59
+ if (/^facts?\b/i.test(factRaw) || /\bfacts fully stated\b/i.test(factRaw))
60
+ continue;
61
+ // Strip markdown/bold leftovers and trailing separators.
62
+ const fact = factRaw.replace(/\*\*/g, '').replace(/[\u2013\u2014:-]+$/, '').trim();
63
+ const key = fact.toLowerCase();
64
+ if (seen.has(key))
65
+ continue;
66
+ seen.add(key);
67
+ let note = (m[3] ?? '').replace(/\*\*/g, '').trim().slice(0, 220);
68
+ // Notes that are just the start of the next list item are noise.
69
+ if (/^\d+[.)]/.test(note))
70
+ note = '';
71
+ out.push({ fact, verdict: verdictKind(keyword), ...(note ? { note } : {}) });
72
+ if (out.length >= 12)
73
+ break; // sanity cap
74
+ }
75
+ return out;
76
+ }
77
+ // Hex-ish document/article ids (8+ chars, at least one a–f so bare integers
78
+ // like "10000000" never qualify) — what RAG corpora and judge reasoning use
79
+ // when naming expected vs cited sources.
80
+ const HEX_ID_SCAN = /\b(?=[0-9]*[a-f])[0-9a-f]{8,64}\b/i;
81
+ /** First hex id within `window` chars after `index` in `text`, if any. */
82
+ function firstIdAfter(text, index, window = 140) {
83
+ const m = text.slice(index, index + window).match(HEX_ID_SCAN);
84
+ return m?.[0];
85
+ }
86
+ /**
87
+ * Detect an "expected source X but cited/retrieved Y" statement.
88
+ * Conservative: both ids must be present and differ.
89
+ */
90
+ export function parseSourceMismatch(reasoning) {
91
+ if (!reasoning)
92
+ return null;
93
+ const text = reasoning.slice(0, MAX_PARSE_CHARS);
94
+ // Expected id: first hex id shortly after an "expected source …" mention.
95
+ // (A character-class "gap" can't work here — prose like "document is
96
+ // article" contains hex letters — so scan a window for the first id.)
97
+ const expectedKw = text.match(/expected\s+source\s+(?:document|article)?/i);
98
+ if (!expectedKw || expectedKw.index === undefined)
99
+ return null;
100
+ const expected = firstIdAfter(text, expectedKw.index + expectedKw[0].length);
101
+ if (!expected)
102
+ return null;
103
+ // Cited id: first differing hex id shortly after a cite/retrieve verb.
104
+ const citedKw = /(?:\bcited\b|\bretrieved\b|\busing\s+article\b|\bcites\b)/gi;
105
+ for (const m of text.matchAll(citedKw)) {
106
+ if (m.index === undefined)
107
+ continue;
108
+ const id = firstIdAfter(text, m.index + m[0].length);
109
+ if (id && id.toLowerCase() !== expected.toLowerCase()) {
110
+ return { expected, cited: id };
111
+ }
112
+ }
113
+ // Compact form: "(b6c9353c vs 49d9e88f)" — either order relative to expected.
114
+ const vs = text.match(new RegExp(String.raw `([0-9a-f]{8,64})\s*(?:vs\.?|versus)\s*([0-9a-f]{8,64})`, 'i'));
115
+ if (vs) {
116
+ const [a, b] = [vs[1], vs[2]];
117
+ const other = a.toLowerCase() === expected.toLowerCase() ? b : a;
118
+ if (other && other.toLowerCase() !== expected.toLowerCase()) {
119
+ return { expected, cited: other };
120
+ }
121
+ }
122
+ return null;
123
+ }
124
+ /** Shorten a long doc id for display: `49d9e88fadbf…` (first 8 chars). */
125
+ export function shortId(id) {
126
+ return id.length > 12 ? `${id.slice(0, 8)}…` : id;
127
+ }
128
+ //# sourceMappingURL=judgeReasoningParse.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"judgeReasoningParse.js","sourceRoot":"","sources":["../../../matchers/judgeReasoningParse.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAyCH,6DAA6D;AAC7D,SAAS,WAAW,CAAC,OAAe;IAClC,MAAM,CAAC,GAAG,OAAO,CAAC,WAAW,EAAE,CAAC;IAChC,IAAI,CAAC,CAAC,QAAQ,CAAC,YAAY,CAAC,IAAI,CAAC,CAAC,CAAC,QAAQ,CAAC,SAAS,CAAC;QAAE,OAAO,cAAc,CAAC;IAC9E,IAAI,CAAC,CAAC,QAAQ,CAAC,SAAS,CAAC,IAAI,CAAC,CAAC,QAAQ,CAAC,YAAY,CAAC;QAAE,OAAO,SAAS,CAAC;IACxE,IAAI,CAAC,CAAC,QAAQ,CAAC,SAAS,CAAC;QAAE,OAAO,SAAS,CAAC;IAC5C,OAAO,QAAQ,CAAC;AAClB,CAAC;AAED,wEAAwE;AACxE,uEAAuE;AACvE,MAAM,eAAe,GACnB,mHAAmH,CAAC;AAEtH,6EAA6E;AAC7E,yEAAyE;AACzE,oDAAoD;AACpD,MAAM,eAAe,GAAG,MAAM,CAAC;AAE/B,2EAA2E;AAC3E,8EAA8E;AAC9E,sEAAsE;AACtE,wEAAwE;AACxE,sEAAsE;AACtE,4EAA4E;AAC5E,MAAM,SAAS,GAAG,IAAI,MAAM,CAC1B,MAAM,CAAC,GAAG,CAAA,cAAc,GAAG,8DAA8D;IACvF,MAAM,CAAC,GAAG,CAAA,4EAA4E,GAAG,SAAS;IAClG,MAAM,CAAC,GAAG,CAAA,cAAc;IACxB,MAAM,CAAC,GAAG,CAAA,6BAA6B,GAAG,mBAAmB;IAC7D,MAAM,CAAC,GAAG,CAAA,qDAAqD,GAAG,YAAY;IAC9E,eAAe,CAAC,MAAM,GAAG,UAAU;IACnC,MAAM,CAAC,GAAG,CAAA,gBAAgB,GAAG,qBAAqB;IAClD,MAAM,CAAC,GAAG,CAAA,UAAU;IACpB,0EAA0E;IAC1E,wEAAwE;IACxE,MAAM,CAAC,GAAG,CAAA,uEAAuE,EACnF,IAAI,CACL,CAAC;AAEF;;;;GAIG;AACH,MAAM,UAAU,iBAAiB,CAAC,SAA6B;IAC7D,IAAI,CAAC,SAAS,IAAI,SAAS,CAAC,MAAM,GAAG,EAAE;QAAE,OAAO,EAAE,CAAC;IACnD,MAAM,IAAI,GAAG,SAAS,CAAC,KAAK,CAAC,CAAC,EAAE,eAAe,CAAC,CAAC;IACjD,MAAM,GAAG,GAAwB,EAAE,CAAC;IACpC,MAAM,IAAI,GAAG,IAAI,GAAG,EAAU,CAAC;IAC/B,KAAK,MAAM,CAAC,IAAI,IAAI,CAAC,QAAQ,CAAC,SAAS,CAAC,EAAE,CAAC;QACzC,MAAM,OAAO,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,IAAI,EAAE,CAAC,CAAC,IAAI,EAAE,CAAC;QACpC,MAAM,OAAO,GAAG,CAAC,CAAC,CAAC,CAAC,IAAI,EAAE,CAAC;QAC3B,oDAAoD;QACpD,mEAAmE;QACnE,+DAA+D;QAC/D,IAAI,CAAC,OAAO,IAAI,OAAO,CAAC,MAAM,GAAG,CAAC;YAAE,SAAS;QAC7C,IAAI,YAAY,CAAC,IAAI,CAAC,OAAO,CAAC,IAAI,yBAAyB,CAAC,IAAI,CAAC,OAAO,CAAC;YAAE,SAAS;QACpF,yDAAyD;QACzD,MAAM,IAAI,GAAG,OAAO,CAAC,OAAO,CAAC,OAAO,EAAE,EAAE,CAAC,CAAC,OAAO,CAAC,oBAAoB,EAAE,EAAE,CAAC,CAAC,IAAI,EAAE,CAAC;QACnF,MAAM,GAAG,GAAG,IAAI,CAAC,WAAW,EAAE,CAAC;QAC/B,IAAI,IAAI,CAAC,GAAG,CAAC,GAAG,CAAC;YAAE,SAAS;QAC5B,IAAI,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC;QACd,IAAI,IAAI,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,IAAI,EAAE,CAAC,CAAC,OAAO,CAAC,OAAO,EAAE,EAAE,CAAC,CAAC,IAAI,EAAE,CAAC,KAAK,CAAC,CAAC,EAAE,GAAG,CAAC,CAAC;QAClE,iEAAiE;QACjE,IAAI,UAAU,CAAC,IAAI,CAAC,IAAI,CAAC;YAAE,IAAI,GAAG,EAAE,CAAC;QACrC,GAAG,CAAC,IAAI,CAAC,EAAE,IAAI,EAAE,OAAO,EAAE,WAAW,CAAC,OAAO,CAAC,EAAE,GAAG,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,IAAI,EAAE,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,CAAC,CAAC;QAC7E,IAAI,GAAG,CAAC,MAAM,IAAI,EAAE;YAAE,MAAM,CAAC,aAAa;IAC5C,CAAC;IACD,OAAO,GAAG,CAAC;AACb,CAAC;AAED,4EAA4E;AAC5E,4EAA4E;AAC5E,yCAAyC;AACzC,MAAM,WAAW,GAAG,oCAAoC,CAAC;AAEzD,0EAA0E;AAC1E,SAAS,YAAY,CAAC,IAAY,EAAE,KAAa,EAAE,MAAM,GAAG,GAAG;IAC7D,MAAM,CAAC,GAAG,IAAI,CAAC,KAAK,CAAC,KAAK,EAAE,KAAK,GAAG,MAAM,CAAC,CAAC,KAAK,CAAC,WAAW,CAAC,CAAC;IAC/D,OAAO,CAAC,EAAE,CAAC,CAAC,CAAC,CAAC;AAChB,CAAC;AAED;;;GAGG;AACH,MAAM,UAAU,mBAAmB,CAAC,SAA6B;IAC/D,IAAI,CAAC,SAAS;QAAE,OAAO,IAAI,CAAC;IAC5B,MAAM,IAAI,GAAG,SAAS,CAAC,KAAK,CAAC,CAAC,EAAE,eAAe,CAAC,CAAC;IAEjD,0EAA0E;IAC1E,qEAAqE;IACrE,sEAAsE;IACtE,MAAM,UAAU,GAAG,IAAI,CAAC,KAAK,CAAC,4CAA4C,CAAC,CAAC;IAC5E,IAAI,CAAC,UAAU,IAAI,UAAU,CAAC,KAAK,KAAK,SAAS;QAAE,OAAO,IAAI,CAAC;IAC/D,MAAM,QAAQ,GAAG,YAAY,CAAC,IAAI,EAAE,UAAU,CAAC,KAAK,GAAG,UAAU,CAAC,CAAC,CAAC,CAAC,MAAM,CAAC,CAAC;IAC7E,IAAI,CAAC,QAAQ;QAAE,OAAO,IAAI,CAAC;IAE3B,uEAAuE;IACvE,MAAM,OAAO,GAAG,6DAA6D,CAAC;IAC9E,KAAK,MAAM,CAAC,IAAI,IAAI,CAAC,QAAQ,CAAC,OAAO,CAAC,EAAE,CAAC;QACvC,IAAI,CAAC,CAAC,KAAK,KAAK,SAAS;YAAE,SAAS;QACpC,MAAM,EAAE,GAAG,YAAY,CAAC,IAAI,EAAE,CAAC,CAAC,KAAK,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,MAAM,CAAC,CAAC;QACrD,IAAI,EAAE,IAAI,EAAE,CAAC,WAAW,EAAE,KAAK,QAAQ,CAAC,WAAW,EAAE,EAAE,CAAC;YACtD,OAAO,EAAE,QAAQ,EAAE,KAAK,EAAE,EAAE,EAAE,CAAC;QACjC,CAAC;IACH,CAAC;IAED,8EAA8E;IAC9E,MAAM,EAAE,GAAG,IAAI,CAAC,KAAK,CACnB,IAAI,MAAM,CAAC,MAAM,CAAC,GAAG,CAAA,wDAAwD,EAAE,GAAG,CAAC,CACpF,CAAC;IACF,IAAI,EAAE,EAAE,CAAC;QACP,MAAM,CAAC,CAAC,EAAE,CAAC,CAAC,GAAG,CAAC,EAAE,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,CAAC,CAAC;QAC9B,MAAM,KAAK,GAAG,CAAC,CAAC,WAAW,EAAE,KAAK,QAAQ,CAAC,WAAW,EAAE,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;QACjE,IAAI,KAAK,IAAI,KAAK,CAAC,WAAW,EAAE,KAAK,QAAQ,CAAC,WAAW,EAAE,EAAE,CAAC;YAC5D,OAAO,EAAE,QAAQ,EAAE,KAAK,EAAE,KAAK,EAAE,CAAC;QACpC,CAAC;IACH,CAAC;IACD,OAAO,IAAI,CAAC;AACd,CAAC;AAED,0EAA0E;AAC1E,MAAM,UAAU,OAAO,CAAC,EAAU;IAChC,OAAO,EAAE,CAAC,MAAM,GAAG,EAAE,CAAC,CAAC,CAAC,GAAG,EAAE,CAAC,KAAK,CAAC,CAAC,EAAE,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,CAAC;AACpD,CAAC"}
@@ -71,5 +71,30 @@ export interface MatcherResult {
71
71
  trajectory_alignment_score?: number;
72
72
  [k: string]: number | undefined;
73
73
  };
74
+ /**
75
+ * True for a synthetic entry the runner appends when the test body threw
76
+ * before reaching further matcher calls in source order — distinct from
77
+ * `pass: false` (a matcher that DID run and failed). This row was never
78
+ * *attempted*; it marks that later expect()/judge()/evaluate() calls (if
79
+ * any) never ran because chai's bail-on-first-failure semantics stopped
80
+ * the body at the first throw. Always excluded from gate/pass-rate
81
+ * aggregation. See `appendNotReachedMarker()` in services/evaluation/
82
+ * index.ts and `expect.soft()` in lib/matchers/expect.ts for the mode
83
+ * that avoids needing this marker by not bailing at all.
84
+ */
85
+ notReached?: boolean;
86
+ /**
87
+ * Structured, non-metric judge output beyond the typed wire fields — the
88
+ * SDK-side mirror of `JudgeResponse.extraFields`. Any JSON key a judge
89
+ * prompt emits beyond the known schema lands here (captured by
90
+ * `server/services/judgeResponseParser.ts`), so prompt iteration surfaces
91
+ * new structure without code changes. Conventional keys the UI knows how
92
+ * to render when present:
93
+ *
94
+ * - `facts`: Array<{ fact: string; verdict: 'stated'|'partial'|'missing'|'contradicted'; rationale?: string; credit?: number }>
95
+ * - `failure_causes`: Array<{ cause: string; detail?: string; dimension?: string }>
96
+ * - `evidence`: { expected_sources?: string[]; cited_sources?: Array<string | { id: string; title?: string }> }
97
+ */
98
+ judgeExtraFields?: Record<string, unknown>;
74
99
  }
75
100
  //# sourceMappingURL=types.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"types.d.ts","sourceRoot":"","sources":["../../../matchers/types.ts"],"names":[],"mappings":"AAKA;;;;;;;;;;;GAWG;AAEH,MAAM,MAAM,aAAa,GACrB,gBAAgB,GAChB,WAAW,GACX,QAAQ,GACR,WAAW,CAAC;AAEhB,MAAM,WAAW,aAAa;IAC5B,kEAAkE;IAClE,WAAW,EAAE,MAAM,CAAC;IACpB,mCAAmC;IACnC,IAAI,EAAE,OAAO,CAAC;IACd,qCAAqC;IACrC,MAAM,EAAE,aAAa,CAAC;IACtB;;;;;;OAMG;IACH,IAAI,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IAC1B;;;;OAIG;IACH,OAAO,CAAC,EAAE,OAAO,CAAC;IAClB,0DAA0D;IAC1D,UAAU,CAAC,EAAE,MAAM,CAAC;IAGpB,6DAA6D;IAC7D,MAAM,CAAC,EAAE,OAAO,CAAC;IACjB,kEAAkE;IAClE,QAAQ,CAAC,EAAE,OAAO,CAAC;IACnB,kEAAkE;IAClE,YAAY,CAAC,EAAE,MAAM,CAAC;IAGtB,+DAA+D;IAC/D,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,iDAAiD;IACjD,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,gDAAgD;IAChD,KAAK,CAAC,EAAE,MAAM,CAAC;IASf;;;;;OAKG;IACH,qBAAqB,CAAC,EAAE,KAAK,CAAC;QAC5B,QAAQ,EAAE,MAAM,CAAC;QACjB,KAAK,EAAE,MAAM,CAAC;QACd,cAAc,EAAE,MAAM,CAAC;QACvB,QAAQ,EAAE,MAAM,GAAG,QAAQ,GAAG,KAAK,CAAC;KACrC,CAAC,CAAC;IAEH;;;;;OAKG;IACH,YAAY,CAAC,EAAE;QACb,QAAQ,CAAC,EAAE,MAAM,CAAC;QAClB,YAAY,CAAC,EAAE,MAAM,CAAC;QACtB,aAAa,CAAC,EAAE,MAAM,CAAC;QACvB,0BAA0B,CAAC,EAAE,MAAM,CAAC;QACpC,CAAC,CAAC,EAAE,MAAM,GAAG,MAAM,GAAG,SAAS,CAAC;KACjC,CAAC;CACH"}
1
+ {"version":3,"file":"types.d.ts","sourceRoot":"","sources":["../../../matchers/types.ts"],"names":[],"mappings":"AAKA;;;;;;;;;;;GAWG;AAEH,MAAM,MAAM,aAAa,GACrB,gBAAgB,GAChB,WAAW,GACX,QAAQ,GACR,WAAW,CAAC;AAEhB,MAAM,WAAW,aAAa;IAC5B,kEAAkE;IAClE,WAAW,EAAE,MAAM,CAAC;IACpB,mCAAmC;IACnC,IAAI,EAAE,OAAO,CAAC;IACd,qCAAqC;IACrC,MAAM,EAAE,aAAa,CAAC;IACtB;;;;;;OAMG;IACH,IAAI,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IAC1B;;;;OAIG;IACH,OAAO,CAAC,EAAE,OAAO,CAAC;IAClB,0DAA0D;IAC1D,UAAU,CAAC,EAAE,MAAM,CAAC;IAGpB,6DAA6D;IAC7D,MAAM,CAAC,EAAE,OAAO,CAAC;IACjB,kEAAkE;IAClE,QAAQ,CAAC,EAAE,OAAO,CAAC;IACnB,kEAAkE;IAClE,YAAY,CAAC,EAAE,MAAM,CAAC;IAGtB,+DAA+D;IAC/D,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,iDAAiD;IACjD,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,gDAAgD;IAChD,KAAK,CAAC,EAAE,MAAM,CAAC;IASf;;;;;OAKG;IACH,qBAAqB,CAAC,EAAE,KAAK,CAAC;QAC5B,QAAQ,EAAE,MAAM,CAAC;QACjB,KAAK,EAAE,MAAM,CAAC;QACd,cAAc,EAAE,MAAM,CAAC;QACvB,QAAQ,EAAE,MAAM,GAAG,QAAQ,GAAG,KAAK,CAAC;KACrC,CAAC,CAAC;IAEH;;;;;OAKG;IACH,YAAY,CAAC,EAAE;QACb,QAAQ,CAAC,EAAE,MAAM,CAAC;QAClB,YAAY,CAAC,EAAE,MAAM,CAAC;QACtB,aAAa,CAAC,EAAE,MAAM,CAAC;QACvB,0BAA0B,CAAC,EAAE,MAAM,CAAC;QACpC,CAAC,CAAC,EAAE,MAAM,GAAG,MAAM,GAAG,SAAS,CAAC;KACjC,CAAC;IAEF;;;;;;;;;;OAUG;IACH,UAAU,CAAC,EAAE,OAAO,CAAC;IAErB;;;;;;;;;;;OAWG;IACH,gBAAgB,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;CAC5C"}
@@ -0,0 +1,23 @@
1
+ /**
2
+ * Resolve the CANONICAL run object for an id that may exist in two shapes:
3
+ * a legacy `BenchmarkRun` projection embedded in `benchmark.runs[]` (never
4
+ * carries `docType`, never kept in sync after the initial write) and a
5
+ * first-class `EvaluationRun` doc (`docType: 'evaluation-run'`). Runs
6
+ * created WITH a benchmarkId are dual-written as both (see
7
+ * `server/routes/storage/evaluationRuns.ts`) -- the first-class doc is
8
+ * always the freshest/most capable representation when it exists.
9
+ *
10
+ * Extracted out of RunInspectorPage.tsx (the first caller) so the
11
+ * resolution logic is reusable and independently testable rather than a
12
+ * page-local pattern other components/route handlers would have to
13
+ * reinvent (see #462's Retry-judgement work, which needs the exact same
14
+ * resolution to make its own EvaluationRun-only capability checks
15
+ * meaningful on the benchmark-scoped route).
16
+ *
17
+ * `fetchEvaluationRun` is injected (rather than importing
18
+ * `services/client` directly) so this stays a plain, synchronously
19
+ * testable function with no module-mocking required.
20
+ */
21
+ import type { BenchmarkRun, EvaluationRun } from '../types/index.js';
22
+ export declare function resolveCanonicalEvaluationRun(runId: string, embeddedProjection: BenchmarkRun, fetchEvaluationRun: (id: string) => Promise<EvaluationRun>): Promise<BenchmarkRun | EvaluationRun>;
23
+ //# sourceMappingURL=resolveCanonicalRun.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"resolveCanonicalRun.d.ts","sourceRoot":"","sources":["../../resolveCanonicalRun.ts"],"names":[],"mappings":"AAKA;;;;;;;;;;;;;;;;;;;GAmBG;AAEH,OAAO,KAAK,EAAE,YAAY,EAAE,aAAa,EAAE,MAAM,kBAAkB,CAAC;AAEpE,wBAAsB,6BAA6B,CACjD,KAAK,EAAE,MAAM,EACb,kBAAkB,EAAE,YAAY,EAChC,kBAAkB,EAAE,CAAC,EAAE,EAAE,MAAM,KAAK,OAAO,CAAC,aAAa,CAAC,GACzD,OAAO,CAAC,YAAY,GAAG,aAAa,CAAC,CAsBvC"}
@@ -0,0 +1,26 @@
1
+ /*
2
+ * Copyright OpenSearch Contributors
3
+ * SPDX-License-Identifier: Apache-2.0
4
+ */
5
+ export async function resolveCanonicalEvaluationRun(runId, embeddedProjection, fetchEvaluationRun) {
6
+ try {
7
+ // Defensive `?? embeddedProjection`: some test doubles / API layers
8
+ // resolve to a falsy value on "not found" instead of throwing.
9
+ return (await fetchEvaluationRun(runId)) ?? embeddedProjection;
10
+ }
11
+ catch (err) {
12
+ // A 404 means this run only ever exists as a legacy BenchmarkRun
13
+ // (pre-#399, no first-class doc) -- expected, silent fallback. Any
14
+ // OTHER failure (500, network error, auth) must NOT be silently
15
+ // treated the same way: falling back is still the right availability
16
+ // choice for a read-only inspector page (this page already degrades
17
+ // gracefully elsewhere -- see loadData()'s report-summary fallback),
18
+ // but masking a real failure identically to "doesn't exist" would
19
+ // hide it from anyone debugging why results/stats look stale.
20
+ if (err?.status !== 404) {
21
+ console.warn(`[resolveCanonicalEvaluationRun] Failed to fetch first-class EvaluationRun doc for ${runId} (falling back to the embedded projection):`, err?.message ?? err);
22
+ }
23
+ return embeddedProjection;
24
+ }
25
+ }
26
+ //# sourceMappingURL=resolveCanonicalRun.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"resolveCanonicalRun.js","sourceRoot":"","sources":["../../resolveCanonicalRun.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAyBH,MAAM,CAAC,KAAK,UAAU,6BAA6B,CACjD,KAAa,EACb,kBAAgC,EAChC,kBAA0D;IAE1D,IAAI,CAAC;QACH,oEAAoE;QACpE,+DAA+D;QAC/D,OAAO,CAAC,MAAM,kBAAkB,CAAC,KAAK,CAAC,CAAC,IAAI,kBAAkB,CAAC;IACjE,CAAC;IAAC,OAAO,GAAQ,EAAE,CAAC;QAClB,iEAAiE;QACjE,mEAAmE;QACnE,gEAAgE;QAChE,qEAAqE;QACrE,oEAAoE;QACpE,qEAAqE;QACrE,kEAAkE;QAClE,8DAA8D;QAC9D,IAAI,GAAG,EAAE,MAAM,KAAK,GAAG,EAAE,CAAC;YACxB,OAAO,CAAC,IAAI,CACV,qFAAqF,KAAK,6CAA6C,EACvI,GAAG,EAAE,OAAO,IAAI,GAAG,CACpB,CAAC;QACJ,CAAC;QACD,OAAO,kBAAkB,CAAC;IAC5B,CAAC;AACH,CAAC"}
@@ -0,0 +1,120 @@
1
+ /**
2
+ * Pure, isomorphic predicates for the run-lifecycle action matrix (delete /
3
+ * cancel / re-run / retry-judgement) shared by every run surface (runs list,
4
+ * benchmark runs list, run detail/report page, inspector header) AND by the
5
+ * server routes that enforce the same rules server-side. No storage/IO here
6
+ * — callers pass in the run document they already have.
7
+ *
8
+ * Action matrix (see AGENTS.md / PR description for the full writeup):
9
+ * - Delete: any run, any status. Always available (existing endpoint).
10
+ * - Cancel: only while `status === 'running'`.
11
+ * - Re-run: only top-level EvaluationRun docs (docType === 'evaluation-run').
12
+ * Legacy benchmark-embedded BenchmarkRun rows don't support the
13
+ * provenance-tracked rerun endpoint (pre-existing constraint — see
14
+ * RunConfigDialog / EvalRunsPage).
15
+ * - Retry judgement: only EvaluationRun docs, only when the run is
16
+ * terminal (not running) AND it has at least one test case whose agent
17
+ * execution completed but the judge produced NO verdict (a judge-failed
18
+ * / "errored" case — trace timeout, judge 400, "evaluator could not
19
+ * run" — as opposed to an agent-failed one — retrying the judge on a
20
+ * case the agent itself never finished has nothing to re-grade). Same
21
+ * predicate the retry-judgement pipeline itself selects on
22
+ * (services/evaluation/retryJudgement.ts `isJudgeFailedCase`, keyed on
23
+ * the report's `metricsStatus: 'error'`, which the runner mirrors onto
24
+ * the run's results map as a `completed` result with no
25
+ * `passFailStatus`) and that `lib/runStats` buckets as `errored`.
26
+ */
27
+ import type { BenchmarkRun, EvaluationRun } from '../types/index.js';
28
+ /** Minimal shape both BenchmarkRun and EvaluationRun satisfy for these checks. */
29
+ export type RunLike = Pick<BenchmarkRun | EvaluationRun, 'status' | 'results'> & {
30
+ docType?: string;
31
+ };
32
+ /**
33
+ * True when `run` is a top-level EvaluationRun document (created via
34
+ * `POST /api/storage/evaluation-runs`), as opposed to a legacy
35
+ * benchmark-embedded BenchmarkRun (`benchmark.runs[]`). The two share a lot
36
+ * of shape but only EvaluationRun docs carry `docType: 'evaluation-run'` and
37
+ * support the rerun/retry-judgement endpoints.
38
+ *
39
+ * Null-tolerant wrapper over the typed predicate in `types/index.ts` (the
40
+ * single source of truth for the docType discriminator) — kept so callers
41
+ * holding a possibly-null run don't need their own guard.
42
+ */
43
+ export declare function isEvaluationRun(run: RunLike | null | undefined): run is EvaluationRun;
44
+ /** True while the run has an in-progress executor that a Cancel action could stop. */
45
+ export declare function isRunRunning(run: RunLike | null | undefined): boolean;
46
+ /** True once a run has reached any terminal state (not running/pending). */
47
+ export declare function isRunTerminal(run: RunLike | null | undefined): boolean;
48
+ /**
49
+ * Count test cases where the AGENT finished (`status === 'completed'`) but
50
+ * the JUDGE produced no verdict (`passFailStatus` neither 'passed' nor
51
+ * 'failed') — the "errored" bucket of `lib/runStats` `bucketRunResults`
52
+ * (issue #242) and exactly the set `POST .../retry-judgement` (default
53
+ * `scope=errored`) will re-judge. Deliberately excludes:
54
+ * - `status !== 'completed'` (agent-failed/cancelled/pending cases — no
55
+ * trajectory to re-judge, or nothing ran).
56
+ * - a real 'failed' verdict — the judge DID run and graded the case; that
57
+ * is a legitimate result, not a judge failure (re-grading it is
58
+ * `scope=all`, opt-in from the inspector's dedicated button).
59
+ *
60
+ * `passFailStatus` isn't declared on `EvaluationRun['results']`'s static
61
+ * type (a pre-existing gap — evaluationRunner.ts writes it via an `as any`
62
+ * spread) so this reads it defensively.
63
+ */
64
+ export declare function countJudgeFailed(run: RunLike | null | undefined): number;
65
+ export interface RunActionVisibility {
66
+ /** Delete is always available for any run in any status. */
67
+ canDelete: boolean;
68
+ /** Cancel is available only while the run is actively running. */
69
+ canCancel: boolean;
70
+ /** Re-run is available only for top-level EvaluationRun docs. */
71
+ canRerun: boolean;
72
+ /** Reason to show (e.g. as a disabled-item tooltip) when canRerun is false. */
73
+ rerunDisabledReason?: string;
74
+ /** Retry judgement: EvaluationRun, terminal, with >0 judge-failed cases. */
75
+ canRetryJudgement: boolean;
76
+ /** Reason to show when canRetryJudgement is false. */
77
+ retryJudgementDisabledReason?: string;
78
+ /** Number of judge-failed test cases (0 when not applicable/unknown). */
79
+ judgeFailedCount: number;
80
+ }
81
+ /**
82
+ * Minimum time a run must have been persisted before a Cancel request with
83
+ * no in-memory executor token is allowed to take the "zombie" fallback path
84
+ * (mark cancelled directly — see getRunActionVisibility callers in the
85
+ * cancel routes). Guards the narrow window right after a run is created:
86
+ * the doc is persisted (and therefore visible to a concurrent Cancel
87
+ * request) strictly before its executor registers its cancellation token,
88
+ * so a Cancel that lands in that gap would otherwise mark a run "cancelled"
89
+ * moments before its own executor starts making progress on it. A brand-new
90
+ * run is also the case the fallback is LEAST useful for — "zombie" (no
91
+ * executor anywhere) is far more plausible once a run has been running for
92
+ * a while than in its first couple of seconds.
93
+ */
94
+ export declare const ZOMBIE_CANCEL_MIN_AGE_MS = 5000;
95
+ /**
96
+ * True once a run is old enough that a tokenless Cancel request can safely
97
+ * assume its executor (if any) would already have registered a
98
+ * cancellation token — i.e. it's safe to treat "no token" as "no executor"
99
+ * rather than "executor hasn't started yet".
100
+ *
101
+ * NOTE — known limitation, not fixed by this check: cancellation tokens are
102
+ * tracked in an in-memory `Map` scoped to ONE server process. In a
103
+ * multi-process/clustered deployment, a Cancel request routed to a
104
+ * DIFFERENT process than the one executing the run will always find no
105
+ * token there, regardless of run age, and this zombie fallback will mark
106
+ * the run cancelled in storage even though it's alive and progressing on
107
+ * another process. This mirrors a pre-existing, documented constraint of
108
+ * this codebase's run-execution model (see AGENTS.md's "orphan-run
109
+ * recovery" notes: "active is tracked per-process"); fixing it for real
110
+ * needs the same heartbeat-based ownership (`run.heartbeatAt`) that doc
111
+ * already calls out as the eventual replacement. Out of scope here.
112
+ */
113
+ export declare function isOldEnoughForZombieCancel(createdAt: string | undefined, now?: number): boolean;
114
+ /**
115
+ * Compute the full action-visibility matrix for one run. Pure function —
116
+ * safe to call from both React components and server-side route validation
117
+ * so the two never drift.
118
+ */
119
+ export declare function getRunActionVisibility(run: RunLike | null | undefined): RunActionVisibility;
120
+ //# sourceMappingURL=runActions.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"runActions.d.ts","sourceRoot":"","sources":["../../runActions.ts"],"names":[],"mappings":"AAKA;;;;;;;;;;;;;;;;;;;;;;;;;GAyBG;AAEH,OAAO,KAAK,EAAE,YAAY,EAAE,aAAa,EAAE,MAAM,SAAS,CAAC;AAG3D,kFAAkF;AAClF,MAAM,MAAM,OAAO,GAAG,IAAI,CAAC,YAAY,GAAG,aAAa,EAAE,QAAQ,GAAG,SAAS,CAAC,GAAG;IAC/E,OAAO,CAAC,EAAE,MAAM,CAAC;CAClB,CAAC;AAEF;;;;;;;;;;GAUG;AACH,wBAAgB,eAAe,CAAC,GAAG,EAAE,OAAO,GAAG,IAAI,GAAG,SAAS,GAAG,GAAG,IAAI,aAAa,CAErF;AAED,sFAAsF;AACtF,wBAAgB,YAAY,CAAC,GAAG,EAAE,OAAO,GAAG,IAAI,GAAG,SAAS,GAAG,OAAO,CAErE;AAED,4EAA4E;AAC5E,wBAAgB,aAAa,CAAC,GAAG,EAAE,OAAO,GAAG,IAAI,GAAG,SAAS,GAAG,OAAO,CAEtE;AAED;;;;;;;;;;;;;;;GAeG;AACH,wBAAgB,gBAAgB,CAAC,GAAG,EAAE,OAAO,GAAG,IAAI,GAAG,SAAS,GAAG,MAAM,CAUxE;AAED,MAAM,WAAW,mBAAmB;IAClC,4DAA4D;IAC5D,SAAS,EAAE,OAAO,CAAC;IACnB,kEAAkE;IAClE,SAAS,EAAE,OAAO,CAAC;IACnB,iEAAiE;IACjE,QAAQ,EAAE,OAAO,CAAC;IAClB,+EAA+E;IAC/E,mBAAmB,CAAC,EAAE,MAAM,CAAC;IAC7B,4EAA4E;IAC5E,iBAAiB,EAAE,OAAO,CAAC;IAC3B,sDAAsD;IACtD,4BAA4B,CAAC,EAAE,MAAM,CAAC;IACtC,yEAAyE;IACzE,gBAAgB,EAAE,MAAM,CAAC;CAC1B;AAOD;;;;;;;;;;;;GAYG;AACH,eAAO,MAAM,wBAAwB,OAAO,CAAC;AAE7C;;;;;;;;;;;;;;;;;GAiBG;AACH,wBAAgB,0BAA0B,CAAC,SAAS,EAAE,MAAM,GAAG,SAAS,EAAE,GAAG,GAAE,MAAmB,GAAG,OAAO,CAI3G;AAED;;;;GAIG;AACH,wBAAgB,sBAAsB,CAAC,GAAG,EAAE,OAAO,GAAG,IAAI,GAAG,SAAS,GAAG,mBAAmB,CAuB3F"}
@@ -0,0 +1,130 @@
1
+ /*
2
+ * Copyright OpenSearch Contributors
3
+ * SPDX-License-Identifier: Apache-2.0
4
+ */
5
+ import { isEvaluationRun as isEvaluationRunDoc } from '../types/index.js';
6
+ /**
7
+ * True when `run` is a top-level EvaluationRun document (created via
8
+ * `POST /api/storage/evaluation-runs`), as opposed to a legacy
9
+ * benchmark-embedded BenchmarkRun (`benchmark.runs[]`). The two share a lot
10
+ * of shape but only EvaluationRun docs carry `docType: 'evaluation-run'` and
11
+ * support the rerun/retry-judgement endpoints.
12
+ *
13
+ * Null-tolerant wrapper over the typed predicate in `types/index.ts` (the
14
+ * single source of truth for the docType discriminator) — kept so callers
15
+ * holding a possibly-null run don't need their own guard.
16
+ */
17
+ export function isEvaluationRun(run) {
18
+ return !!run && isEvaluationRunDoc(run);
19
+ }
20
+ /** True while the run has an in-progress executor that a Cancel action could stop. */
21
+ export function isRunRunning(run) {
22
+ return run?.status === 'running';
23
+ }
24
+ /** True once a run has reached any terminal state (not running/pending). */
25
+ export function isRunTerminal(run) {
26
+ return !!run && (run.status === 'completed' || run.status === 'failed' || run.status === 'cancelled');
27
+ }
28
+ /**
29
+ * Count test cases where the AGENT finished (`status === 'completed'`) but
30
+ * the JUDGE produced no verdict (`passFailStatus` neither 'passed' nor
31
+ * 'failed') — the "errored" bucket of `lib/runStats` `bucketRunResults`
32
+ * (issue #242) and exactly the set `POST .../retry-judgement` (default
33
+ * `scope=errored`) will re-judge. Deliberately excludes:
34
+ * - `status !== 'completed'` (agent-failed/cancelled/pending cases — no
35
+ * trajectory to re-judge, or nothing ran).
36
+ * - a real 'failed' verdict — the judge DID run and graded the case; that
37
+ * is a legitimate result, not a judge failure (re-grading it is
38
+ * `scope=all`, opt-in from the inspector's dedicated button).
39
+ *
40
+ * `passFailStatus` isn't declared on `EvaluationRun['results']`'s static
41
+ * type (a pre-existing gap — evaluationRunner.ts writes it via an `as any`
42
+ * spread) so this reads it defensively.
43
+ */
44
+ export function countJudgeFailed(run) {
45
+ if (!run?.results)
46
+ return 0;
47
+ let count = 0;
48
+ for (const r of Object.values(run.results)) {
49
+ const result = r;
50
+ if (result.status !== 'completed')
51
+ continue;
52
+ if (result.passFailStatus === 'passed' || result.passFailStatus === 'failed')
53
+ continue;
54
+ count++;
55
+ }
56
+ return count;
57
+ }
58
+ const RERUN_NOT_SUPPORTED_REASON = "Re-run isn't available for legacy benchmark-embedded runs";
59
+ const RETRY_JUDGEMENT_NOT_SUPPORTED_REASON = "Retry judgement isn't available for legacy benchmark-embedded runs";
60
+ const RETRY_JUDGEMENT_STILL_RUNNING_REASON = 'Retry judgement is only available once the run finishes';
61
+ const RETRY_JUDGEMENT_NONE_FAILED_REASON = 'No judge-failed test cases to retry';
62
+ /**
63
+ * Minimum time a run must have been persisted before a Cancel request with
64
+ * no in-memory executor token is allowed to take the "zombie" fallback path
65
+ * (mark cancelled directly — see getRunActionVisibility callers in the
66
+ * cancel routes). Guards the narrow window right after a run is created:
67
+ * the doc is persisted (and therefore visible to a concurrent Cancel
68
+ * request) strictly before its executor registers its cancellation token,
69
+ * so a Cancel that lands in that gap would otherwise mark a run "cancelled"
70
+ * moments before its own executor starts making progress on it. A brand-new
71
+ * run is also the case the fallback is LEAST useful for — "zombie" (no
72
+ * executor anywhere) is far more plausible once a run has been running for
73
+ * a while than in its first couple of seconds.
74
+ */
75
+ export const ZOMBIE_CANCEL_MIN_AGE_MS = 5000;
76
+ /**
77
+ * True once a run is old enough that a tokenless Cancel request can safely
78
+ * assume its executor (if any) would already have registered a
79
+ * cancellation token — i.e. it's safe to treat "no token" as "no executor"
80
+ * rather than "executor hasn't started yet".
81
+ *
82
+ * NOTE — known limitation, not fixed by this check: cancellation tokens are
83
+ * tracked in an in-memory `Map` scoped to ONE server process. In a
84
+ * multi-process/clustered deployment, a Cancel request routed to a
85
+ * DIFFERENT process than the one executing the run will always find no
86
+ * token there, regardless of run age, and this zombie fallback will mark
87
+ * the run cancelled in storage even though it's alive and progressing on
88
+ * another process. This mirrors a pre-existing, documented constraint of
89
+ * this codebase's run-execution model (see AGENTS.md's "orphan-run
90
+ * recovery" notes: "active is tracked per-process"); fixing it for real
91
+ * needs the same heartbeat-based ownership (`run.heartbeatAt`) that doc
92
+ * already calls out as the eventual replacement. Out of scope here.
93
+ */
94
+ export function isOldEnoughForZombieCancel(createdAt, now = Date.now()) {
95
+ const created = createdAt ? Date.parse(createdAt) : NaN;
96
+ if (Number.isNaN(created))
97
+ return true; // no timestamp to compare against — don't block on it
98
+ return now - created >= ZOMBIE_CANCEL_MIN_AGE_MS;
99
+ }
100
+ /**
101
+ * Compute the full action-visibility matrix for one run. Pure function —
102
+ * safe to call from both React components and server-side route validation
103
+ * so the two never drift.
104
+ */
105
+ export function getRunActionVisibility(run) {
106
+ const evalRun = isEvaluationRun(run);
107
+ const running = isRunRunning(run);
108
+ const terminal = isRunTerminal(run);
109
+ const judgeFailedCount = evalRun ? countJudgeFailed(run) : 0;
110
+ const canRetryJudgement = evalRun && terminal && judgeFailedCount > 0;
111
+ let retryJudgementDisabledReason;
112
+ if (!canRetryJudgement) {
113
+ if (!evalRun)
114
+ retryJudgementDisabledReason = RETRY_JUDGEMENT_NOT_SUPPORTED_REASON;
115
+ else if (!terminal)
116
+ retryJudgementDisabledReason = RETRY_JUDGEMENT_STILL_RUNNING_REASON;
117
+ else
118
+ retryJudgementDisabledReason = RETRY_JUDGEMENT_NONE_FAILED_REASON;
119
+ }
120
+ return {
121
+ canDelete: true,
122
+ canCancel: running,
123
+ canRerun: evalRun,
124
+ rerunDisabledReason: evalRun ? undefined : RERUN_NOT_SUPPORTED_REASON,
125
+ canRetryJudgement,
126
+ retryJudgementDisabledReason,
127
+ judgeFailedCount,
128
+ };
129
+ }
130
+ //# sourceMappingURL=runActions.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"runActions.js","sourceRoot":"","sources":["../../runActions.ts"],"names":[],"mappings":"AAAA;;;GAGG;AA8BH,OAAO,EAAE,eAAe,IAAI,kBAAkB,EAAE,MAAM,SAAS,CAAC;AAOhE;;;;;;;;;;GAUG;AACH,MAAM,UAAU,eAAe,CAAC,GAA+B;IAC7D,OAAO,CAAC,CAAC,GAAG,IAAI,kBAAkB,CAAC,GAAmC,CAAC,CAAC;AAC1E,CAAC;AAED,sFAAsF;AACtF,MAAM,UAAU,YAAY,CAAC,GAA+B;IAC1D,OAAO,GAAG,EAAE,MAAM,KAAK,SAAS,CAAC;AACnC,CAAC;AAED,4EAA4E;AAC5E,MAAM,UAAU,aAAa,CAAC,GAA+B;IAC3D,OAAO,CAAC,CAAC,GAAG,IAAI,CAAC,GAAG,CAAC,MAAM,KAAK,WAAW,IAAI,GAAG,CAAC,MAAM,KAAK,QAAQ,IAAI,GAAG,CAAC,MAAM,KAAK,WAAW,CAAC,CAAC;AACxG,CAAC;AAED;;;;;;;;;;;;;;;GAeG;AACH,MAAM,UAAU,gBAAgB,CAAC,GAA+B;IAC9D,IAAI,CAAC,GAAG,EAAE,OAAO;QAAE,OAAO,CAAC,CAAC;IAC5B,IAAI,KAAK,GAAG,CAAC,CAAC;IACd,KAAK,MAAM,CAAC,IAAI,MAAM,CAAC,MAAM,CAAC,GAAG,CAAC,OAAO,CAAC,EAAE,CAAC;QAC3C,MAAM,MAAM,GAAG,CAAwD,CAAC;QACxE,IAAI,MAAM,CAAC,MAAM,KAAK,WAAW;YAAE,SAAS;QAC5C,IAAI,MAAM,CAAC,cAAc,KAAK,QAAQ,IAAI,MAAM,CAAC,cAAc,KAAK,QAAQ;YAAE,SAAS;QACvF,KAAK,EAAE,CAAC;IACV,CAAC;IACD,OAAO,KAAK,CAAC;AACf,CAAC;AAmBD,MAAM,0BAA0B,GAAG,2DAA2D,CAAC;AAC/F,MAAM,oCAAoC,GAAG,oEAAoE,CAAC;AAClH,MAAM,oCAAoC,GAAG,yDAAyD,CAAC;AACvG,MAAM,kCAAkC,GAAG,qCAAqC,CAAC;AAEjF;;;;;;;;;;;;GAYG;AACH,MAAM,CAAC,MAAM,wBAAwB,GAAG,IAAI,CAAC;AAE7C;;;;;;;;;;;;;;;;;GAiBG;AACH,MAAM,UAAU,0BAA0B,CAAC,SAA6B,EAAE,MAAc,IAAI,CAAC,GAAG,EAAE;IAChG,MAAM,OAAO,GAAG,SAAS,CAAC,CAAC,CAAC,IAAI,CAAC,KAAK,CAAC,SAAS,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC;IACxD,IAAI,MAAM,CAAC,KAAK,CAAC,OAAO,CAAC;QAAE,OAAO,IAAI,CAAC,CAAC,sDAAsD;IAC9F,OAAO,GAAG,GAAG,OAAO,IAAI,wBAAwB,CAAC;AACnD,CAAC;AAED;;;;GAIG;AACH,MAAM,UAAU,sBAAsB,CAAC,GAA+B;IACpE,MAAM,OAAO,GAAG,eAAe,CAAC,GAAG,CAAC,CAAC;IACrC,MAAM,OAAO,GAAG,YAAY,CAAC,GAAG,CAAC,CAAC;IAClC,MAAM,QAAQ,GAAG,aAAa,CAAC,GAAG,CAAC,CAAC;IACpC,MAAM,gBAAgB,GAAG,OAAO,CAAC,CAAC,CAAC,gBAAgB,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;IAE7D,MAAM,iBAAiB,GAAG,OAAO,IAAI,QAAQ,IAAI,gBAAgB,GAAG,CAAC,CAAC;IACtE,IAAI,4BAAgD,CAAC;IACrD,IAAI,CAAC,iBAAiB,EAAE,CAAC;QACvB,IAAI,CAAC,OAAO;YAAE,4BAA4B,GAAG,oCAAoC,CAAC;aAC7E,IAAI,CAAC,QAAQ;YAAE,4BAA4B,GAAG,oCAAoC,CAAC;;YACnF,4BAA4B,GAAG,kCAAkC,CAAC;IACzE,CAAC;IAED,OAAO;QACL,SAAS,EAAE,IAAI;QACf,SAAS,EAAE,OAAO;QAClB,QAAQ,EAAE,OAAO;QACjB,mBAAmB,EAAE,OAAO,CAAC,CAAC,CAAC,SAAS,CAAC,CAAC,CAAC,0BAA0B;QACrE,iBAAiB;QACjB,4BAA4B;QAC5B,gBAAgB;KACjB,CAAC;AACJ,CAAC"}
@@ -0,0 +1,86 @@
1
+ /**
2
+ * Deterministic (no-LLM) aggregation helpers for the run-report "insights"
3
+ * pane (RunInsightsPane.tsx), shown on the bare
4
+ * `/benchmarks/:benchmarkId/runs/:runId` route when no test case is
5
+ * selected — see owner feedback on goyamegh/run-report-redesign: "if no
6
+ * test case is selected, the right side can show an aggregated view ...
7
+ * why did the failing tests fail — something that is complete info."
8
+ *
9
+ * Everything here is a pure function over already-fetched data (report
10
+ * summaries + test-case categories). No network calls, no LLM calls — v1
11
+ * is explicitly deterministic-only per the product ask.
12
+ */
13
+ export interface CategoryStatusRow {
14
+ category: string;
15
+ /** Any ResultStatus value; only 'passed' / 'failed' / 'errored' are counted distinctly, everything else falls into the bar's `total` only (pending/running cases). */
16
+ status: string;
17
+ }
18
+ export interface CategoryBar {
19
+ category: string;
20
+ passed: number;
21
+ failed: number;
22
+ errored: number;
23
+ total: number;
24
+ }
25
+ /**
26
+ * Group rows by test-case category and tally pass/fail/errored/total.
27
+ * Deterministic order: largest category first, ties broken alphabetically.
28
+ */
29
+ export declare function computeCategoryBars(rows: CategoryStatusRow[]): CategoryBar[];
30
+ /**
31
+ * Normalize a judge-reasoning string down to its first sentence,
32
+ * lowercased, punctuation-stripped, whitespace-collapsed. Used both as the
33
+ * clustering input and as the theme's stable `key`.
34
+ */
35
+ export declare function normalizeReasoningKey(reasoning: string): string;
36
+ export interface FailureThemeInput {
37
+ testCaseId: string;
38
+ reasoning: string;
39
+ }
40
+ export interface FailureTheme {
41
+ /** Stable cluster key — the normalized first sentence of the theme's representative case. */
42
+ key: string;
43
+ count: number;
44
+ /** Trimmed, human-readable first sentence sampled from the theme's most common exact phrasing. */
45
+ sampleSnippet: string;
46
+ testCaseIds: string[];
47
+ }
48
+ /**
49
+ * Cluster failing test cases into "why they failed" themes using a
50
+ * deterministic, LLM-free heuristic: normalize each case's judge-reasoning
51
+ * first sentence, then union-find cases that share at least one contiguous
52
+ * N-word shingle. This is robust to minor paraphrasing (a judge saying
53
+ * "unable to retrieve" vs "failed to retrieve" the same underlying tool
54
+ * connectivity failure) while still keeping genuinely distinct failure
55
+ * modes (e.g. "missing required facts" vs "MCP server unavailable")
56
+ * separate, because they share no contiguous phrase.
57
+ *
58
+ * Verified against a real production run (418-verify, 64 failing cases):
59
+ * 57 of 64 connectivity-flavored reasonings collapse into ONE dominant
60
+ * theme; the remaining 7 ("Required facts evaluation: ...", a genuinely
61
+ * different failure shape) form a second, correctly separate theme.
62
+ *
63
+ * Output order: largest theme first, ties broken by the theme's
64
+ * lowest-sorting testCaseId (deterministic, no dependency on input order).
65
+ */
66
+ export declare function clusterFailureThemes(items: FailureThemeInput[], shingleSize?: number): FailureTheme[];
67
+ /**
68
+ * "Based on N of M failing cases" note shown under the theme list when the
69
+ * reasoning fetch was capped (RunInsightsPane caps at the first 100 failing
70
+ * cases). Returns null when nothing was capped (fetchedCount >= totalCount).
71
+ */
72
+ export declare function formatCappedNote(fetchedCount: number, totalFailingCount: number): string | null;
73
+ export interface RankedCase {
74
+ testCaseId: string;
75
+ value: number;
76
+ }
77
+ /**
78
+ * Top-N cases by a numeric value (duration, cost, ...), descending.
79
+ * Cases with a null/undefined/NaN value are excluded. Deterministic tie
80
+ * break: lower testCaseId first.
81
+ */
82
+ export declare function pickTopN(cases: {
83
+ testCaseId: string;
84
+ value: number | null | undefined;
85
+ }[], n: number): RankedCase[];
86
+ //# sourceMappingURL=runInsights.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"runInsights.d.ts","sourceRoot":"","sources":["../../runInsights.ts"],"names":[],"mappings":"AAKA;;;;;;;;;;;GAWG;AAIH,MAAM,WAAW,iBAAiB;IAChC,QAAQ,EAAE,MAAM,CAAC;IACjB,sKAAsK;IACtK,MAAM,EAAE,MAAM,CAAC;CAChB;AAED,MAAM,WAAW,WAAW;IAC1B,QAAQ,EAAE,MAAM,CAAC;IACjB,MAAM,EAAE,MAAM,CAAC;IACf,MAAM,EAAE,MAAM,CAAC;IACf,OAAO,EAAE,MAAM,CAAC;IAChB,KAAK,EAAE,MAAM,CAAC;CACf;AAID;;;GAGG;AACH,wBAAgB,mBAAmB,CAAC,IAAI,EAAE,iBAAiB,EAAE,GAAG,WAAW,EAAE,CAiB5E;AAID;;;;GAIG;AACH,wBAAgB,qBAAqB,CAAC,SAAS,EAAE,MAAM,GAAG,MAAM,CAM/D;AAqBD,MAAM,WAAW,iBAAiB;IAChC,UAAU,EAAE,MAAM,CAAC;IACnB,SAAS,EAAE,MAAM,CAAC;CACnB;AAED,MAAM,WAAW,YAAY;IAC3B,6FAA6F;IAC7F,GAAG,EAAE,MAAM,CAAC;IACZ,KAAK,EAAE,MAAM,CAAC;IACd,kGAAkG;IAClG,aAAa,EAAE,MAAM,CAAC;IACtB,WAAW,EAAE,MAAM,EAAE,CAAC;CACvB;AAOD;;;;;;;;;;;;;;;;;GAiBG;AACH,wBAAgB,oBAAoB,CAClC,KAAK,EAAE,iBAAiB,EAAE,EAC1B,WAAW,GAAE,MAA6B,GACzC,YAAY,EAAE,CAwEhB;AAED;;;;GAIG;AACH,wBAAgB,gBAAgB,CAAC,YAAY,EAAE,MAAM,EAAE,iBAAiB,EAAE,MAAM,GAAG,MAAM,GAAG,IAAI,CAG/F;AAID,MAAM,WAAW,UAAU;IACzB,UAAU,EAAE,MAAM,CAAC;IACnB,KAAK,EAAE,MAAM,CAAC;CACf;AAED;;;;GAIG;AACH,wBAAgB,QAAQ,CAAC,KAAK,EAAE;IAAE,UAAU,EAAE,MAAM,CAAC;IAAC,KAAK,EAAE,MAAM,GAAG,IAAI,GAAG,SAAS,CAAA;CAAE,EAAE,EAAE,CAAC,EAAE,MAAM,GAAG,UAAU,EAAE,CAKnH"}