@opensearch-project/agent-health 0.6.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/cli/dist/index.js +722 -330
- package/dist/assets/index-D-Np_l_T.js +246 -0
- package/dist/assets/index-vZt9QZKf.css +1 -0
- package/dist/index.html +2 -2
- package/docs/CLI.md +138 -1
- package/docs/CONNECTORS.md +1 -1
- package/docs/SDK.md +126 -2
- package/docs/skills/add-connector/SKILL.md +5 -1
- package/lib/dist/lib/agentTrends.d.ts +210 -0
- package/lib/dist/lib/agentTrends.d.ts.map +1 -0
- package/lib/dist/lib/agentTrends.js +360 -0
- package/lib/dist/lib/agentTrends.js.map +1 -0
- package/lib/dist/lib/benchmarkCaseReview.d.ts +114 -0
- package/lib/dist/lib/benchmarkCaseReview.d.ts.map +1 -0
- package/lib/dist/lib/benchmarkCaseReview.js +177 -0
- package/lib/dist/lib/benchmarkCaseReview.js.map +1 -0
- package/lib/dist/lib/benchmarkRunsTable.d.ts +109 -0
- package/lib/dist/lib/benchmarkRunsTable.d.ts.map +1 -0
- package/lib/dist/lib/benchmarkRunsTable.js +212 -0
- package/lib/dist/lib/benchmarkRunsTable.js.map +1 -0
- package/lib/dist/lib/comparisonInsights.d.ts +49 -2
- package/lib/dist/lib/comparisonInsights.d.ts.map +1 -1
- package/lib/dist/lib/comparisonInsights.js +65 -7
- package/lib/dist/lib/comparisonInsights.js.map +1 -1
- package/lib/dist/lib/config/loader.d.ts.map +1 -1
- package/lib/dist/lib/config/loader.js +11 -1
- package/lib/dist/lib/config/loader.js.map +1 -1
- package/lib/dist/lib/dashboardMetrics.d.ts +11 -2
- package/lib/dist/lib/dashboardMetrics.d.ts.map +1 -1
- package/lib/dist/lib/dashboardMetrics.js +38 -3
- package/lib/dist/lib/dashboardMetrics.js.map +1 -1
- package/lib/dist/lib/evaluationRerun.d.ts +39 -0
- package/lib/dist/lib/evaluationRerun.d.ts.map +1 -1
- package/lib/dist/lib/evaluationRerun.js +49 -0
- package/lib/dist/lib/evaluationRerun.js.map +1 -1
- package/lib/dist/lib/judgeFailureSummary.d.ts +66 -0
- package/lib/dist/lib/judgeFailureSummary.d.ts.map +1 -0
- package/lib/dist/lib/judgeFailureSummary.js +68 -0
- package/lib/dist/lib/judgeFailureSummary.js.map +1 -0
- package/lib/dist/lib/judgeStrategies.d.ts +108 -0
- package/lib/dist/lib/judgeStrategies.d.ts.map +1 -0
- package/lib/dist/lib/judgeStrategies.js +135 -0
- package/lib/dist/lib/judgeStrategies.js.map +1 -0
- package/lib/dist/lib/matchers/expect.d.ts +21 -1
- package/lib/dist/lib/matchers/expect.d.ts.map +1 -1
- package/lib/dist/lib/matchers/expect.js +51 -0
- package/lib/dist/lib/matchers/expect.js.map +1 -1
- package/lib/dist/lib/matchers/judgeAccessor.d.ts +4 -0
- package/lib/dist/lib/matchers/judgeAccessor.d.ts.map +1 -1
- package/lib/dist/lib/matchers/judgeAccessor.js +13 -2
- package/lib/dist/lib/matchers/judgeAccessor.js.map +1 -1
- package/lib/dist/lib/matchers/judgeReasoningParse.d.ts +49 -0
- package/lib/dist/lib/matchers/judgeReasoningParse.d.ts.map +1 -0
- package/lib/dist/lib/matchers/judgeReasoningParse.js +128 -0
- package/lib/dist/lib/matchers/judgeReasoningParse.js.map +1 -0
- package/lib/dist/lib/matchers/types.d.ts +25 -0
- package/lib/dist/lib/matchers/types.d.ts.map +1 -1
- package/lib/dist/lib/resolveCanonicalRun.d.ts +23 -0
- package/lib/dist/lib/resolveCanonicalRun.d.ts.map +1 -0
- package/lib/dist/lib/resolveCanonicalRun.js +26 -0
- package/lib/dist/lib/resolveCanonicalRun.js.map +1 -0
- package/lib/dist/lib/runActions.d.ts +120 -0
- package/lib/dist/lib/runActions.d.ts.map +1 -0
- package/lib/dist/lib/runActions.js +130 -0
- package/lib/dist/lib/runActions.js.map +1 -0
- package/lib/dist/lib/runInsights.d.ts +86 -0
- package/lib/dist/lib/runInsights.d.ts.map +1 -0
- package/lib/dist/lib/runInsights.js +185 -0
- package/lib/dist/lib/runInsights.js.map +1 -0
- package/lib/dist/lib/runName.d.ts +29 -0
- package/lib/dist/lib/runName.d.ts.map +1 -0
- package/lib/dist/lib/runName.js +38 -0
- package/lib/dist/lib/runName.js.map +1 -0
- package/lib/dist/lib/runReportPath.d.ts +16 -0
- package/lib/dist/lib/runReportPath.d.ts.map +1 -0
- package/lib/dist/lib/runReportPath.js +22 -0
- package/lib/dist/lib/runReportPath.js.map +1 -0
- package/lib/dist/lib/runSort.d.ts +27 -0
- package/lib/dist/lib/runSort.d.ts.map +1 -0
- package/lib/dist/lib/runSort.js +31 -0
- package/lib/dist/lib/runSort.js.map +1 -0
- package/lib/dist/lib/runStats.d.ts +86 -6
- package/lib/dist/lib/runStats.d.ts.map +1 -1
- package/lib/dist/lib/runStats.js +170 -19
- package/lib/dist/lib/runStats.js.map +1 -1
- package/lib/dist/lib/testCases/define.d.ts.map +1 -1
- package/lib/dist/lib/testCases/define.js +93 -46
- package/lib/dist/lib/testCases/define.js.map +1 -1
- package/lib/dist/lib/testCases/judge.d.ts.map +1 -1
- package/lib/dist/lib/testCases/judge.js +22 -4
- package/lib/dist/lib/testCases/judge.js.map +1 -1
- package/lib/dist/lib/testCases/loader.d.ts +24 -0
- package/lib/dist/lib/testCases/loader.d.ts.map +1 -1
- package/lib/dist/lib/testCases/loader.js +253 -36
- package/lib/dist/lib/testCases/loader.js.map +1 -1
- package/lib/dist/lib/trajectoryStepDisplay.d.ts +25 -0
- package/lib/dist/lib/trajectoryStepDisplay.d.ts.map +1 -0
- package/lib/dist/lib/trajectoryStepDisplay.js +42 -0
- package/lib/dist/lib/trajectoryStepDisplay.js.map +1 -0
- package/lib/dist/lib/utils.d.ts +19 -0
- package/lib/dist/lib/utils.d.ts.map +1 -1
- package/lib/dist/lib/utils.js +27 -0
- package/lib/dist/lib/utils.js.map +1 -1
- package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.d.ts +68 -32
- package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.d.ts.map +1 -1
- package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.js +148 -125
- package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.js.map +1 -1
- package/lib/dist/services/connectors/kiro/KiroConnector.d.ts +21 -13
- package/lib/dist/services/connectors/kiro/KiroConnector.d.ts.map +1 -1
- package/lib/dist/services/connectors/kiro/KiroConnector.js +22 -25
- package/lib/dist/services/connectors/kiro/KiroConnector.js.map +1 -1
- package/lib/dist/services/connectors/pi/PiConnector.d.ts +49 -10
- package/lib/dist/services/connectors/pi/PiConnector.d.ts.map +1 -1
- package/lib/dist/services/connectors/pi/PiConnector.js +102 -86
- package/lib/dist/services/connectors/pi/PiConnector.js.map +1 -1
- package/lib/dist/services/connectors/subprocess/SubprocessConnector.d.ts +59 -14
- package/lib/dist/services/connectors/subprocess/SubprocessConnector.d.ts.map +1 -1
- package/lib/dist/services/connectors/subprocess/SubprocessConnector.js +87 -61
- package/lib/dist/services/connectors/subprocess/SubprocessConnector.js.map +1 -1
- package/lib/dist/services/evaluation/bedrockJudge.d.ts +11 -0
- package/lib/dist/services/evaluation/bedrockJudge.d.ts.map +1 -1
- package/lib/dist/services/evaluation/bedrockJudge.js +2 -0
- package/lib/dist/services/evaluation/bedrockJudge.js.map +1 -1
- package/lib/dist/services/evaluation/index.d.ts +18 -1
- package/lib/dist/services/evaluation/index.d.ts.map +1 -1
- package/lib/dist/services/evaluation/index.js +156 -19
- package/lib/dist/services/evaluation/index.js.map +1 -1
- package/lib/dist/services/metrics.d.ts +55 -0
- package/lib/dist/services/metrics.d.ts.map +1 -0
- package/lib/dist/services/metrics.js +89 -0
- package/lib/dist/services/metrics.js.map +1 -0
- package/lib/dist/services/storage/asyncBenchmarkStorage.d.ts.map +1 -1
- package/lib/dist/services/storage/asyncBenchmarkStorage.js +16 -0
- package/lib/dist/services/storage/asyncBenchmarkStorage.js.map +1 -1
- package/lib/dist/services/storage/asyncRunStorage.d.ts +11 -0
- package/lib/dist/services/storage/asyncRunStorage.d.ts.map +1 -1
- package/lib/dist/services/storage/asyncRunStorage.js +49 -2
- package/lib/dist/services/storage/asyncRunStorage.js.map +1 -1
- package/lib/dist/services/storage/asyncTestCaseStorage.d.ts +1 -1
- package/lib/dist/services/storage/asyncTestCaseStorage.d.ts.map +1 -1
- package/lib/dist/services/storage/asyncTestCaseStorage.js +2 -1
- package/lib/dist/services/storage/asyncTestCaseStorage.js.map +1 -1
- package/lib/dist/services/storage/opensearchClient.d.ts +12 -0
- package/lib/dist/services/storage/opensearchClient.d.ts.map +1 -1
- package/lib/dist/services/storage/opensearchClient.js.map +1 -1
- package/lib/dist/services/traces/browserRecovery.d.ts.map +1 -1
- package/lib/dist/services/traces/browserRecovery.js +3 -0
- package/lib/dist/services/traces/browserRecovery.js.map +1 -1
- package/lib/dist/services/traces/index.d.ts +8 -1
- package/lib/dist/services/traces/index.d.ts.map +1 -1
- package/lib/dist/services/traces/index.js +33 -12
- package/lib/dist/services/traces/index.js.map +1 -1
- package/lib/dist/services/traces/judgeAgentsHints.d.ts +98 -3
- package/lib/dist/services/traces/judgeAgentsHints.d.ts.map +1 -1
- package/lib/dist/services/traces/judgeAgentsHints.js +144 -3
- package/lib/dist/services/traces/judgeAgentsHints.js.map +1 -1
- package/lib/dist/services/traces/spansToTrajectory.js +4 -4
- package/lib/dist/services/traces/spansToTrajectory.js.map +1 -1
- package/lib/dist/services/traces/tracePoller.d.ts.map +1 -1
- package/lib/dist/services/traces/tracePoller.js +17 -20
- package/lib/dist/services/traces/tracePoller.js.map +1 -1
- package/lib/dist/services/traces/trajectoryMerge.d.ts +79 -0
- package/lib/dist/services/traces/trajectoryMerge.d.ts.map +1 -0
- package/lib/dist/services/traces/trajectoryMerge.js +109 -0
- package/lib/dist/services/traces/trajectoryMerge.js.map +1 -0
- package/lib/dist/types/index.d.ts +195 -2
- package/lib/dist/types/index.d.ts.map +1 -1
- package/lib/dist/types/index.js +24 -0
- package/lib/dist/types/index.js.map +1 -1
- package/package.json +6 -5
- package/server/dist/app.js +3398 -926
- package/server/dist/index.js +3401 -929
- package/dist/assets/index-BfxtxmKc.css +0 -1
- package/dist/assets/index-CrjAfDHu.js +0 -243
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Copyright OpenSearch Contributors
|
|
3
|
+
* SPDX-License-Identifier: Apache-2.0
|
|
4
|
+
*/
|
|
5
|
+
/** Map a matched verdict keyword onto the canonical kind. */
|
|
6
|
+
function verdictKind(keyword) {
|
|
7
|
+
const k = keyword.toUpperCase();
|
|
8
|
+
if (k.includes('CONTRADICT') && !k.includes('MISSING'))
|
|
9
|
+
return 'contradicted';
|
|
10
|
+
if (k.includes('MISSING') || k.includes('NOT STATED'))
|
|
11
|
+
return 'missing';
|
|
12
|
+
if (k.includes('PARTIAL'))
|
|
13
|
+
return 'partial';
|
|
14
|
+
return 'stated';
|
|
15
|
+
}
|
|
16
|
+
// Verdict keywords the judge writes inline after a fact. Order matters:
|
|
17
|
+
// longest / most specific first so "MISSING/CONTRADICTED" isn't split.
|
|
18
|
+
const VERDICT_KEYWORD = /(MISSING\s*\/\s*CONTRADICTED|FULLY\s+STATED|PARTIALLY\s+STATED|NOT\s+STATED|CONTRADICTED|MISSING|PARTIAL(?:LY)?)/i;
|
|
19
|
+
// Parse at most this much reasoning — real verdicts are a few KB; a hard cap
|
|
20
|
+
// bounds regex work on pathological inputs (defense against backtracking
|
|
21
|
+
// blowups on adversarial multi-hundred-KB strings).
|
|
22
|
+
const MAX_PARSE_CHARS = 20_000;
|
|
23
|
+
// A fact item: list marker (`1.`, `1)`, `**Fact 1:`, `Required fact 1 (`),
|
|
24
|
+
// then the fact text (quoted or plain), a separator (—, -, :, en-dash, `):`),
|
|
25
|
+
// then the verdict keyword. Fact text is capped to keep matches sane.
|
|
26
|
+
// NOTE: deliberately no lookbehind — constructed-at-import regexes with
|
|
27
|
+
// lookbehind hard-crash report rendering on older WebKit. The leading
|
|
28
|
+
// whitespace/newline is consumed instead (harmless: markers never overlap).
|
|
29
|
+
const FACT_ITEM = new RegExp(String.raw `(?:^|[\s\n])` + // start of string or after whitespace (inline numbered lists)
|
|
30
|
+
String.raw `(?:\*\*)?(?:Required\s+fact\s+\d+|Fact\s+\d+|\d+)\s*[.):\u2013\u2014-]?\s*` + // marker
|
|
31
|
+
String.raw `(?:\*\*)?\s*` +
|
|
32
|
+
String.raw `['‘"“(]?(.{4,240}?)['’"”)]?` + // fact text (lazy)
|
|
33
|
+
String.raw `(?:\*\*)?\s*[\u2013\u2014:(\u2015-]+\s*(?:\*\*)?\s*` + // separator
|
|
34
|
+
VERDICT_KEYWORD.source + // verdict
|
|
35
|
+
String.raw `(?:\s+stated)?` + // "PARTIALLY stated"
|
|
36
|
+
String.raw `[.!]?\s*` +
|
|
37
|
+
// Trailing note: lazy, stopped by the NEXT numbered/bold fact item on the
|
|
38
|
+
// same line (inline lists put every fact in one paragraph) or line end.
|
|
39
|
+
String.raw `([^\n]*?)(?=\s\d+[.)]\s*['‘"“(*]|\s\*\*(?:Required\s+fact|Fact)|\n|$)`, 'gi');
|
|
40
|
+
/**
|
|
41
|
+
* Extract per-required-fact verdicts from judge reasoning prose.
|
|
42
|
+
* Returns `[]` when nothing that looks like a fact list is present —
|
|
43
|
+
* callers must fall back to showing the raw reasoning.
|
|
44
|
+
*/
|
|
45
|
+
export function parseFactVerdicts(reasoning) {
|
|
46
|
+
if (!reasoning || reasoning.length < 20)
|
|
47
|
+
return [];
|
|
48
|
+
const text = reasoning.slice(0, MAX_PARSE_CHARS);
|
|
49
|
+
const out = [];
|
|
50
|
+
const seen = new Set();
|
|
51
|
+
for (const m of text.matchAll(FACT_ITEM)) {
|
|
52
|
+
const factRaw = (m[1] ?? '').trim();
|
|
53
|
+
const keyword = m[2] ?? '';
|
|
54
|
+
// Guard against summary-phrase false positives like
|
|
55
|
+
// "4 facts fully stated (1.0 each)": require real fact text, not a
|
|
56
|
+
// recap that itself talks about facts/statements in aggregate.
|
|
57
|
+
if (!factRaw || factRaw.length < 8)
|
|
58
|
+
continue;
|
|
59
|
+
if (/^facts?\b/i.test(factRaw) || /\bfacts fully stated\b/i.test(factRaw))
|
|
60
|
+
continue;
|
|
61
|
+
// Strip markdown/bold leftovers and trailing separators.
|
|
62
|
+
const fact = factRaw.replace(/\*\*/g, '').replace(/[\u2013\u2014:-]+$/, '').trim();
|
|
63
|
+
const key = fact.toLowerCase();
|
|
64
|
+
if (seen.has(key))
|
|
65
|
+
continue;
|
|
66
|
+
seen.add(key);
|
|
67
|
+
let note = (m[3] ?? '').replace(/\*\*/g, '').trim().slice(0, 220);
|
|
68
|
+
// Notes that are just the start of the next list item are noise.
|
|
69
|
+
if (/^\d+[.)]/.test(note))
|
|
70
|
+
note = '';
|
|
71
|
+
out.push({ fact, verdict: verdictKind(keyword), ...(note ? { note } : {}) });
|
|
72
|
+
if (out.length >= 12)
|
|
73
|
+
break; // sanity cap
|
|
74
|
+
}
|
|
75
|
+
return out;
|
|
76
|
+
}
|
|
77
|
+
// Hex-ish document/article ids (8+ chars, at least one a–f so bare integers
|
|
78
|
+
// like "10000000" never qualify) — what RAG corpora and judge reasoning use
|
|
79
|
+
// when naming expected vs cited sources.
|
|
80
|
+
const HEX_ID_SCAN = /\b(?=[0-9]*[a-f])[0-9a-f]{8,64}\b/i;
|
|
81
|
+
/** First hex id within `window` chars after `index` in `text`, if any. */
|
|
82
|
+
function firstIdAfter(text, index, window = 140) {
|
|
83
|
+
const m = text.slice(index, index + window).match(HEX_ID_SCAN);
|
|
84
|
+
return m?.[0];
|
|
85
|
+
}
|
|
86
|
+
/**
|
|
87
|
+
* Detect an "expected source X but cited/retrieved Y" statement.
|
|
88
|
+
* Conservative: both ids must be present and differ.
|
|
89
|
+
*/
|
|
90
|
+
export function parseSourceMismatch(reasoning) {
|
|
91
|
+
if (!reasoning)
|
|
92
|
+
return null;
|
|
93
|
+
const text = reasoning.slice(0, MAX_PARSE_CHARS);
|
|
94
|
+
// Expected id: first hex id shortly after an "expected source …" mention.
|
|
95
|
+
// (A character-class "gap" can't work here — prose like "document is
|
|
96
|
+
// article" contains hex letters — so scan a window for the first id.)
|
|
97
|
+
const expectedKw = text.match(/expected\s+source\s+(?:document|article)?/i);
|
|
98
|
+
if (!expectedKw || expectedKw.index === undefined)
|
|
99
|
+
return null;
|
|
100
|
+
const expected = firstIdAfter(text, expectedKw.index + expectedKw[0].length);
|
|
101
|
+
if (!expected)
|
|
102
|
+
return null;
|
|
103
|
+
// Cited id: first differing hex id shortly after a cite/retrieve verb.
|
|
104
|
+
const citedKw = /(?:\bcited\b|\bretrieved\b|\busing\s+article\b|\bcites\b)/gi;
|
|
105
|
+
for (const m of text.matchAll(citedKw)) {
|
|
106
|
+
if (m.index === undefined)
|
|
107
|
+
continue;
|
|
108
|
+
const id = firstIdAfter(text, m.index + m[0].length);
|
|
109
|
+
if (id && id.toLowerCase() !== expected.toLowerCase()) {
|
|
110
|
+
return { expected, cited: id };
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
// Compact form: "(b6c9353c vs 49d9e88f)" — either order relative to expected.
|
|
114
|
+
const vs = text.match(new RegExp(String.raw `([0-9a-f]{8,64})\s*(?:vs\.?|versus)\s*([0-9a-f]{8,64})`, 'i'));
|
|
115
|
+
if (vs) {
|
|
116
|
+
const [a, b] = [vs[1], vs[2]];
|
|
117
|
+
const other = a.toLowerCase() === expected.toLowerCase() ? b : a;
|
|
118
|
+
if (other && other.toLowerCase() !== expected.toLowerCase()) {
|
|
119
|
+
return { expected, cited: other };
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
return null;
|
|
123
|
+
}
|
|
124
|
+
/** Shorten a long doc id for display: `49d9e88fadbf…` (first 8 chars). */
|
|
125
|
+
export function shortId(id) {
|
|
126
|
+
return id.length > 12 ? `${id.slice(0, 8)}…` : id;
|
|
127
|
+
}
|
|
128
|
+
//# sourceMappingURL=judgeReasoningParse.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"judgeReasoningParse.js","sourceRoot":"","sources":["../../../matchers/judgeReasoningParse.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAyCH,6DAA6D;AAC7D,SAAS,WAAW,CAAC,OAAe;IAClC,MAAM,CAAC,GAAG,OAAO,CAAC,WAAW,EAAE,CAAC;IAChC,IAAI,CAAC,CAAC,QAAQ,CAAC,YAAY,CAAC,IAAI,CAAC,CAAC,CAAC,QAAQ,CAAC,SAAS,CAAC;QAAE,OAAO,cAAc,CAAC;IAC9E,IAAI,CAAC,CAAC,QAAQ,CAAC,SAAS,CAAC,IAAI,CAAC,CAAC,QAAQ,CAAC,YAAY,CAAC;QAAE,OAAO,SAAS,CAAC;IACxE,IAAI,CAAC,CAAC,QAAQ,CAAC,SAAS,CAAC;QAAE,OAAO,SAAS,CAAC;IAC5C,OAAO,QAAQ,CAAC;AAClB,CAAC;AAED,wEAAwE;AACxE,uEAAuE;AACvE,MAAM,eAAe,GACnB,mHAAmH,CAAC;AAEtH,6EAA6E;AAC7E,yEAAyE;AACzE,oDAAoD;AACpD,MAAM,eAAe,GAAG,MAAM,CAAC;AAE/B,2EAA2E;AAC3E,8EAA8E;AAC9E,sEAAsE;AACtE,wEAAwE;AACxE,sEAAsE;AACtE,4EAA4E;AAC5E,MAAM,SAAS,GAAG,IAAI,MAAM,CAC1B,MAAM,CAAC,GAAG,CAAA,cAAc,GAAG,8DAA8D;IACvF,MAAM,CAAC,GAAG,CAAA,4EAA4E,GAAG,SAAS;IAClG,MAAM,CAAC,GAAG,CAAA,cAAc;IACxB,MAAM,CAAC,GAAG,CAAA,6BAA6B,GAAG,mBAAmB;IAC7D,MAAM,CAAC,GAAG,CAAA,qDAAqD,GAAG,YAAY;IAC9E,eAAe,CAAC,MAAM,GAAG,UAAU;IACnC,MAAM,CAAC,GAAG,CAAA,gBAAgB,GAAG,qBAAqB;IAClD,MAAM,CAAC,GAAG,CAAA,UAAU;IACpB,0EAA0E;IAC1E,wEAAwE;IACxE,MAAM,CAAC,GAAG,CAAA,uEAAuE,EACnF,IAAI,CACL,CAAC;AAEF;;;;GAIG;AACH,MAAM,UAAU,iBAAiB,CAAC,SAA6B;IAC7D,IAAI,CAAC,SAAS,IAAI,SAAS,CAAC,MAAM,GAAG,EAAE;QAAE,OAAO,EAAE,CAAC;IACnD,MAAM,IAAI,GAAG,SAAS,CAAC,KAAK,CAAC,CAAC,EAAE,eAAe,CAAC,CAAC;IACjD,MAAM,GAAG,GAAwB,EAAE,CAAC;IACpC,MAAM,IAAI,GAAG,IAAI,GAAG,EAAU,CAAC;IAC/B,KAAK,MAAM,CAAC,IAAI,IAAI,CAAC,QAAQ,CAAC,SAAS,CAAC,EAAE,CAAC;QACzC,MAAM,OAAO,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,IAAI,EAAE,CAAC,CAAC,IAAI,EAAE,CAAC;QACpC,MAAM,OAAO,GAAG,CAAC,CAAC,CAAC,CAAC,IAAI,EAAE,CAAC;QAC3B,oDAAoD;QACpD,mEAAmE;QACnE,+DAA+D;QAC/D,IAAI,CAAC,OAAO,IAAI,OAAO,CAAC,MAAM,GAAG,CAAC;YAAE,SAAS;QAC7C,IAAI,YAAY,CAAC,IAAI,CAAC,OAAO,CAAC,IAAI,yBAAyB,CAAC,IAAI,CAAC,OAAO,CAAC;YAAE,SAAS;QACpF,yDAAyD;QACzD,MAAM,IAAI,GAAG,OAAO,CAAC,OAAO,CAAC,OAAO,EAAE,EAAE,CAAC,CAAC,OAAO,CAAC,oBAAoB,EAAE,EAAE,CAAC,CAAC,IAAI,EAAE,CAAC;QACnF,MAAM,GAAG,GAAG,IAAI,CAAC,WAAW,EAAE,CAAC;QAC/B,IAAI,IAAI,CAAC,GAAG,CAAC,GAAG,CAAC;YAAE,SAAS;QAC5B,IAAI,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC;QACd,IAAI,IAAI,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,IAAI,EAAE,CAAC,CAAC,OAAO,CAAC,OAAO,EAAE,EAAE,CAAC,CAAC,IAAI,EAAE,CAAC,KAAK,CAAC,CAAC,EAAE,GAAG,CAAC,CAAC;QAClE,iEAAiE;QACjE,IAAI,UAAU,CAAC,IAAI,CAAC,IAAI,CAAC;YAAE,IAAI,GAAG,EAAE,CAAC;QACrC,GAAG,CAAC,IAAI,CAAC,EAAE,IAAI,EAAE,OAAO,EAAE,WAAW,CAAC,OAAO,CAAC,EAAE,GAAG,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,IAAI,EAAE,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,CAAC,CAAC;QAC7E,IAAI,GAAG,CAAC,MAAM,IAAI,EAAE;YAAE,MAAM,CAAC,aAAa;IAC5C,CAAC;IACD,OAAO,GAAG,CAAC;AACb,CAAC;AAED,4EAA4E;AAC5E,4EAA4E;AAC5E,yCAAyC;AACzC,MAAM,WAAW,GAAG,oCAAoC,CAAC;AAEzD,0EAA0E;AAC1E,SAAS,YAAY,CAAC,IAAY,EAAE,KAAa,EAAE,MAAM,GAAG,GAAG;IAC7D,MAAM,CAAC,GAAG,IAAI,CAAC,KAAK,CAAC,KAAK,EAAE,KAAK,GAAG,MAAM,CAAC,CAAC,KAAK,CAAC,WAAW,CAAC,CAAC;IAC/D,OAAO,CAAC,EAAE,CAAC,CAAC,CAAC,CAAC;AAChB,CAAC;AAED;;;GAGG;AACH,MAAM,UAAU,mBAAmB,CAAC,SAA6B;IAC/D,IAAI,CAAC,SAAS;QAAE,OAAO,IAAI,CAAC;IAC5B,MAAM,IAAI,GAAG,SAAS,CAAC,KAAK,CAAC,CAAC,EAAE,eAAe,CAAC,CAAC;IAEjD,0EAA0E;IAC1E,qEAAqE;IACrE,sEAAsE;IACtE,MAAM,UAAU,GAAG,IAAI,CAAC,KAAK,CAAC,4CAA4C,CAAC,CAAC;IAC5E,IAAI,CAAC,UAAU,IAAI,UAAU,CAAC,KAAK,KAAK,SAAS;QAAE,OAAO,IAAI,CAAC;IAC/D,MAAM,QAAQ,GAAG,YAAY,CAAC,IAAI,EAAE,UAAU,CAAC,KAAK,GAAG,UAAU,CAAC,CAAC,CAAC,CAAC,MAAM,CAAC,CAAC;IAC7E,IAAI,CAAC,QAAQ;QAAE,OAAO,IAAI,CAAC;IAE3B,uEAAuE;IACvE,MAAM,OAAO,GAAG,6DAA6D,CAAC;IAC9E,KAAK,MAAM,CAAC,IAAI,IAAI,CAAC,QAAQ,CAAC,OAAO,CAAC,EAAE,CAAC;QACvC,IAAI,CAAC,CAAC,KAAK,KAAK,SAAS;YAAE,SAAS;QACpC,MAAM,EAAE,GAAG,YAAY,CAAC,IAAI,EAAE,CAAC,CAAC,KAAK,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,MAAM,CAAC,CAAC;QACrD,IAAI,EAAE,IAAI,EAAE,CAAC,WAAW,EAAE,KAAK,QAAQ,CAAC,WAAW,EAAE,EAAE,CAAC;YACtD,OAAO,EAAE,QAAQ,EAAE,KAAK,EAAE,EAAE,EAAE,CAAC;QACjC,CAAC;IACH,CAAC;IAED,8EAA8E;IAC9E,MAAM,EAAE,GAAG,IAAI,CAAC,KAAK,CACnB,IAAI,MAAM,CAAC,MAAM,CAAC,GAAG,CAAA,wDAAwD,EAAE,GAAG,CAAC,CACpF,CAAC;IACF,IAAI,EAAE,EAAE,CAAC;QACP,MAAM,CAAC,CAAC,EAAE,CAAC,CAAC,GAAG,CAAC,EAAE,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,CAAC,CAAC;QAC9B,MAAM,KAAK,GAAG,CAAC,CAAC,WAAW,EAAE,KAAK,QAAQ,CAAC,WAAW,EAAE,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;QACjE,IAAI,KAAK,IAAI,KAAK,CAAC,WAAW,EAAE,KAAK,QAAQ,CAAC,WAAW,EAAE,EAAE,CAAC;YAC5D,OAAO,EAAE,QAAQ,EAAE,KAAK,EAAE,KAAK,EAAE,CAAC;QACpC,CAAC;IACH,CAAC;IACD,OAAO,IAAI,CAAC;AACd,CAAC;AAED,0EAA0E;AAC1E,MAAM,UAAU,OAAO,CAAC,EAAU;IAChC,OAAO,EAAE,CAAC,MAAM,GAAG,EAAE,CAAC,CAAC,CAAC,GAAG,EAAE,CAAC,KAAK,CAAC,CAAC,EAAE,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,CAAC;AACpD,CAAC"}
|
|
@@ -71,5 +71,30 @@ export interface MatcherResult {
|
|
|
71
71
|
trajectory_alignment_score?: number;
|
|
72
72
|
[k: string]: number | undefined;
|
|
73
73
|
};
|
|
74
|
+
/**
|
|
75
|
+
* True for a synthetic entry the runner appends when the test body threw
|
|
76
|
+
* before reaching further matcher calls in source order — distinct from
|
|
77
|
+
* `pass: false` (a matcher that DID run and failed). This row was never
|
|
78
|
+
* *attempted*; it marks that later expect()/judge()/evaluate() calls (if
|
|
79
|
+
* any) never ran because chai's bail-on-first-failure semantics stopped
|
|
80
|
+
* the body at the first throw. Always excluded from gate/pass-rate
|
|
81
|
+
* aggregation. See `appendNotReachedMarker()` in services/evaluation/
|
|
82
|
+
* index.ts and `expect.soft()` in lib/matchers/expect.ts for the mode
|
|
83
|
+
* that avoids needing this marker by not bailing at all.
|
|
84
|
+
*/
|
|
85
|
+
notReached?: boolean;
|
|
86
|
+
/**
|
|
87
|
+
* Structured, non-metric judge output beyond the typed wire fields — the
|
|
88
|
+
* SDK-side mirror of `JudgeResponse.extraFields`. Any JSON key a judge
|
|
89
|
+
* prompt emits beyond the known schema lands here (captured by
|
|
90
|
+
* `server/services/judgeResponseParser.ts`), so prompt iteration surfaces
|
|
91
|
+
* new structure without code changes. Conventional keys the UI knows how
|
|
92
|
+
* to render when present:
|
|
93
|
+
*
|
|
94
|
+
* - `facts`: Array<{ fact: string; verdict: 'stated'|'partial'|'missing'|'contradicted'; rationale?: string; credit?: number }>
|
|
95
|
+
* - `failure_causes`: Array<{ cause: string; detail?: string; dimension?: string }>
|
|
96
|
+
* - `evidence`: { expected_sources?: string[]; cited_sources?: Array<string | { id: string; title?: string }> }
|
|
97
|
+
*/
|
|
98
|
+
judgeExtraFields?: Record<string, unknown>;
|
|
74
99
|
}
|
|
75
100
|
//# sourceMappingURL=types.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"types.d.ts","sourceRoot":"","sources":["../../../matchers/types.ts"],"names":[],"mappings":"AAKA;;;;;;;;;;;GAWG;AAEH,MAAM,MAAM,aAAa,GACrB,gBAAgB,GAChB,WAAW,GACX,QAAQ,GACR,WAAW,CAAC;AAEhB,MAAM,WAAW,aAAa;IAC5B,kEAAkE;IAClE,WAAW,EAAE,MAAM,CAAC;IACpB,mCAAmC;IACnC,IAAI,EAAE,OAAO,CAAC;IACd,qCAAqC;IACrC,MAAM,EAAE,aAAa,CAAC;IACtB;;;;;;OAMG;IACH,IAAI,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IAC1B;;;;OAIG;IACH,OAAO,CAAC,EAAE,OAAO,CAAC;IAClB,0DAA0D;IAC1D,UAAU,CAAC,EAAE,MAAM,CAAC;IAGpB,6DAA6D;IAC7D,MAAM,CAAC,EAAE,OAAO,CAAC;IACjB,kEAAkE;IAClE,QAAQ,CAAC,EAAE,OAAO,CAAC;IACnB,kEAAkE;IAClE,YAAY,CAAC,EAAE,MAAM,CAAC;IAGtB,+DAA+D;IAC/D,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,iDAAiD;IACjD,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,gDAAgD;IAChD,KAAK,CAAC,EAAE,MAAM,CAAC;IASf;;;;;OAKG;IACH,qBAAqB,CAAC,EAAE,KAAK,CAAC;QAC5B,QAAQ,EAAE,MAAM,CAAC;QACjB,KAAK,EAAE,MAAM,CAAC;QACd,cAAc,EAAE,MAAM,CAAC;QACvB,QAAQ,EAAE,MAAM,GAAG,QAAQ,GAAG,KAAK,CAAC;KACrC,CAAC,CAAC;IAEH;;;;;OAKG;IACH,YAAY,CAAC,EAAE;QACb,QAAQ,CAAC,EAAE,MAAM,CAAC;QAClB,YAAY,CAAC,EAAE,MAAM,CAAC;QACtB,aAAa,CAAC,EAAE,MAAM,CAAC;QACvB,0BAA0B,CAAC,EAAE,MAAM,CAAC;QACpC,CAAC,CAAC,EAAE,MAAM,GAAG,MAAM,GAAG,SAAS,CAAC;KACjC,CAAC;
|
|
1
|
+
{"version":3,"file":"types.d.ts","sourceRoot":"","sources":["../../../matchers/types.ts"],"names":[],"mappings":"AAKA;;;;;;;;;;;GAWG;AAEH,MAAM,MAAM,aAAa,GACrB,gBAAgB,GAChB,WAAW,GACX,QAAQ,GACR,WAAW,CAAC;AAEhB,MAAM,WAAW,aAAa;IAC5B,kEAAkE;IAClE,WAAW,EAAE,MAAM,CAAC;IACpB,mCAAmC;IACnC,IAAI,EAAE,OAAO,CAAC;IACd,qCAAqC;IACrC,MAAM,EAAE,aAAa,CAAC;IACtB;;;;;;OAMG;IACH,IAAI,CAAC,EAAE,MAAM,GAAG,SAAS,CAAC;IAC1B;;;;OAIG;IACH,OAAO,CAAC,EAAE,OAAO,CAAC;IAClB,0DAA0D;IAC1D,UAAU,CAAC,EAAE,MAAM,CAAC;IAGpB,6DAA6D;IAC7D,MAAM,CAAC,EAAE,OAAO,CAAC;IACjB,kEAAkE;IAClE,QAAQ,CAAC,EAAE,OAAO,CAAC;IACnB,kEAAkE;IAClE,YAAY,CAAC,EAAE,MAAM,CAAC;IAGtB,+DAA+D;IAC/D,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,iDAAiD;IACjD,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,gDAAgD;IAChD,KAAK,CAAC,EAAE,MAAM,CAAC;IASf;;;;;OAKG;IACH,qBAAqB,CAAC,EAAE,KAAK,CAAC;QAC5B,QAAQ,EAAE,MAAM,CAAC;QACjB,KAAK,EAAE,MAAM,CAAC;QACd,cAAc,EAAE,MAAM,CAAC;QACvB,QAAQ,EAAE,MAAM,GAAG,QAAQ,GAAG,KAAK,CAAC;KACrC,CAAC,CAAC;IAEH;;;;;OAKG;IACH,YAAY,CAAC,EAAE;QACb,QAAQ,CAAC,EAAE,MAAM,CAAC;QAClB,YAAY,CAAC,EAAE,MAAM,CAAC;QACtB,aAAa,CAAC,EAAE,MAAM,CAAC;QACvB,0BAA0B,CAAC,EAAE,MAAM,CAAC;QACpC,CAAC,CAAC,EAAE,MAAM,GAAG,MAAM,GAAG,SAAS,CAAC;KACjC,CAAC;IAEF;;;;;;;;;;OAUG;IACH,UAAU,CAAC,EAAE,OAAO,CAAC;IAErB;;;;;;;;;;;OAWG;IACH,gBAAgB,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;CAC5C"}
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Resolve the CANONICAL run object for an id that may exist in two shapes:
|
|
3
|
+
* a legacy `BenchmarkRun` projection embedded in `benchmark.runs[]` (never
|
|
4
|
+
* carries `docType`, never kept in sync after the initial write) and a
|
|
5
|
+
* first-class `EvaluationRun` doc (`docType: 'evaluation-run'`). Runs
|
|
6
|
+
* created WITH a benchmarkId are dual-written as both (see
|
|
7
|
+
* `server/routes/storage/evaluationRuns.ts`) -- the first-class doc is
|
|
8
|
+
* always the freshest/most capable representation when it exists.
|
|
9
|
+
*
|
|
10
|
+
* Extracted out of RunInspectorPage.tsx (the first caller) so the
|
|
11
|
+
* resolution logic is reusable and independently testable rather than a
|
|
12
|
+
* page-local pattern other components/route handlers would have to
|
|
13
|
+
* reinvent (see #462's Retry-judgement work, which needs the exact same
|
|
14
|
+
* resolution to make its own EvaluationRun-only capability checks
|
|
15
|
+
* meaningful on the benchmark-scoped route).
|
|
16
|
+
*
|
|
17
|
+
* `fetchEvaluationRun` is injected (rather than importing
|
|
18
|
+
* `services/client` directly) so this stays a plain, synchronously
|
|
19
|
+
* testable function with no module-mocking required.
|
|
20
|
+
*/
|
|
21
|
+
import type { BenchmarkRun, EvaluationRun } from '../types/index.js';
|
|
22
|
+
export declare function resolveCanonicalEvaluationRun(runId: string, embeddedProjection: BenchmarkRun, fetchEvaluationRun: (id: string) => Promise<EvaluationRun>): Promise<BenchmarkRun | EvaluationRun>;
|
|
23
|
+
//# sourceMappingURL=resolveCanonicalRun.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"resolveCanonicalRun.d.ts","sourceRoot":"","sources":["../../resolveCanonicalRun.ts"],"names":[],"mappings":"AAKA;;;;;;;;;;;;;;;;;;;GAmBG;AAEH,OAAO,KAAK,EAAE,YAAY,EAAE,aAAa,EAAE,MAAM,kBAAkB,CAAC;AAEpE,wBAAsB,6BAA6B,CACjD,KAAK,EAAE,MAAM,EACb,kBAAkB,EAAE,YAAY,EAChC,kBAAkB,EAAE,CAAC,EAAE,EAAE,MAAM,KAAK,OAAO,CAAC,aAAa,CAAC,GACzD,OAAO,CAAC,YAAY,GAAG,aAAa,CAAC,CAsBvC"}
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Copyright OpenSearch Contributors
|
|
3
|
+
* SPDX-License-Identifier: Apache-2.0
|
|
4
|
+
*/
|
|
5
|
+
export async function resolveCanonicalEvaluationRun(runId, embeddedProjection, fetchEvaluationRun) {
|
|
6
|
+
try {
|
|
7
|
+
// Defensive `?? embeddedProjection`: some test doubles / API layers
|
|
8
|
+
// resolve to a falsy value on "not found" instead of throwing.
|
|
9
|
+
return (await fetchEvaluationRun(runId)) ?? embeddedProjection;
|
|
10
|
+
}
|
|
11
|
+
catch (err) {
|
|
12
|
+
// A 404 means this run only ever exists as a legacy BenchmarkRun
|
|
13
|
+
// (pre-#399, no first-class doc) -- expected, silent fallback. Any
|
|
14
|
+
// OTHER failure (500, network error, auth) must NOT be silently
|
|
15
|
+
// treated the same way: falling back is still the right availability
|
|
16
|
+
// choice for a read-only inspector page (this page already degrades
|
|
17
|
+
// gracefully elsewhere -- see loadData()'s report-summary fallback),
|
|
18
|
+
// but masking a real failure identically to "doesn't exist" would
|
|
19
|
+
// hide it from anyone debugging why results/stats look stale.
|
|
20
|
+
if (err?.status !== 404) {
|
|
21
|
+
console.warn(`[resolveCanonicalEvaluationRun] Failed to fetch first-class EvaluationRun doc for ${runId} (falling back to the embedded projection):`, err?.message ?? err);
|
|
22
|
+
}
|
|
23
|
+
return embeddedProjection;
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
//# sourceMappingURL=resolveCanonicalRun.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"resolveCanonicalRun.js","sourceRoot":"","sources":["../../resolveCanonicalRun.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAyBH,MAAM,CAAC,KAAK,UAAU,6BAA6B,CACjD,KAAa,EACb,kBAAgC,EAChC,kBAA0D;IAE1D,IAAI,CAAC;QACH,oEAAoE;QACpE,+DAA+D;QAC/D,OAAO,CAAC,MAAM,kBAAkB,CAAC,KAAK,CAAC,CAAC,IAAI,kBAAkB,CAAC;IACjE,CAAC;IAAC,OAAO,GAAQ,EAAE,CAAC;QAClB,iEAAiE;QACjE,mEAAmE;QACnE,gEAAgE;QAChE,qEAAqE;QACrE,oEAAoE;QACpE,qEAAqE;QACrE,kEAAkE;QAClE,8DAA8D;QAC9D,IAAI,GAAG,EAAE,MAAM,KAAK,GAAG,EAAE,CAAC;YACxB,OAAO,CAAC,IAAI,CACV,qFAAqF,KAAK,6CAA6C,EACvI,GAAG,EAAE,OAAO,IAAI,GAAG,CACpB,CAAC;QACJ,CAAC;QACD,OAAO,kBAAkB,CAAC;IAC5B,CAAC;AACH,CAAC"}
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Pure, isomorphic predicates for the run-lifecycle action matrix (delete /
|
|
3
|
+
* cancel / re-run / retry-judgement) shared by every run surface (runs list,
|
|
4
|
+
* benchmark runs list, run detail/report page, inspector header) AND by the
|
|
5
|
+
* server routes that enforce the same rules server-side. No storage/IO here
|
|
6
|
+
* — callers pass in the run document they already have.
|
|
7
|
+
*
|
|
8
|
+
* Action matrix (see AGENTS.md / PR description for the full writeup):
|
|
9
|
+
* - Delete: any run, any status. Always available (existing endpoint).
|
|
10
|
+
* - Cancel: only while `status === 'running'`.
|
|
11
|
+
* - Re-run: only top-level EvaluationRun docs (docType === 'evaluation-run').
|
|
12
|
+
* Legacy benchmark-embedded BenchmarkRun rows don't support the
|
|
13
|
+
* provenance-tracked rerun endpoint (pre-existing constraint — see
|
|
14
|
+
* RunConfigDialog / EvalRunsPage).
|
|
15
|
+
* - Retry judgement: only EvaluationRun docs, only when the run is
|
|
16
|
+
* terminal (not running) AND it has at least one test case whose agent
|
|
17
|
+
* execution completed but the judge produced NO verdict (a judge-failed
|
|
18
|
+
* / "errored" case — trace timeout, judge 400, "evaluator could not
|
|
19
|
+
* run" — as opposed to an agent-failed one — retrying the judge on a
|
|
20
|
+
* case the agent itself never finished has nothing to re-grade). Same
|
|
21
|
+
* predicate the retry-judgement pipeline itself selects on
|
|
22
|
+
* (services/evaluation/retryJudgement.ts `isJudgeFailedCase`, keyed on
|
|
23
|
+
* the report's `metricsStatus: 'error'`, which the runner mirrors onto
|
|
24
|
+
* the run's results map as a `completed` result with no
|
|
25
|
+
* `passFailStatus`) and that `lib/runStats` buckets as `errored`.
|
|
26
|
+
*/
|
|
27
|
+
import type { BenchmarkRun, EvaluationRun } from '../types/index.js';
|
|
28
|
+
/** Minimal shape both BenchmarkRun and EvaluationRun satisfy for these checks. */
|
|
29
|
+
export type RunLike = Pick<BenchmarkRun | EvaluationRun, 'status' | 'results'> & {
|
|
30
|
+
docType?: string;
|
|
31
|
+
};
|
|
32
|
+
/**
|
|
33
|
+
* True when `run` is a top-level EvaluationRun document (created via
|
|
34
|
+
* `POST /api/storage/evaluation-runs`), as opposed to a legacy
|
|
35
|
+
* benchmark-embedded BenchmarkRun (`benchmark.runs[]`). The two share a lot
|
|
36
|
+
* of shape but only EvaluationRun docs carry `docType: 'evaluation-run'` and
|
|
37
|
+
* support the rerun/retry-judgement endpoints.
|
|
38
|
+
*
|
|
39
|
+
* Null-tolerant wrapper over the typed predicate in `types/index.ts` (the
|
|
40
|
+
* single source of truth for the docType discriminator) — kept so callers
|
|
41
|
+
* holding a possibly-null run don't need their own guard.
|
|
42
|
+
*/
|
|
43
|
+
export declare function isEvaluationRun(run: RunLike | null | undefined): run is EvaluationRun;
|
|
44
|
+
/** True while the run has an in-progress executor that a Cancel action could stop. */
|
|
45
|
+
export declare function isRunRunning(run: RunLike | null | undefined): boolean;
|
|
46
|
+
/** True once a run has reached any terminal state (not running/pending). */
|
|
47
|
+
export declare function isRunTerminal(run: RunLike | null | undefined): boolean;
|
|
48
|
+
/**
|
|
49
|
+
* Count test cases where the AGENT finished (`status === 'completed'`) but
|
|
50
|
+
* the JUDGE produced no verdict (`passFailStatus` neither 'passed' nor
|
|
51
|
+
* 'failed') — the "errored" bucket of `lib/runStats` `bucketRunResults`
|
|
52
|
+
* (issue #242) and exactly the set `POST .../retry-judgement` (default
|
|
53
|
+
* `scope=errored`) will re-judge. Deliberately excludes:
|
|
54
|
+
* - `status !== 'completed'` (agent-failed/cancelled/pending cases — no
|
|
55
|
+
* trajectory to re-judge, or nothing ran).
|
|
56
|
+
* - a real 'failed' verdict — the judge DID run and graded the case; that
|
|
57
|
+
* is a legitimate result, not a judge failure (re-grading it is
|
|
58
|
+
* `scope=all`, opt-in from the inspector's dedicated button).
|
|
59
|
+
*
|
|
60
|
+
* `passFailStatus` isn't declared on `EvaluationRun['results']`'s static
|
|
61
|
+
* type (a pre-existing gap — evaluationRunner.ts writes it via an `as any`
|
|
62
|
+
* spread) so this reads it defensively.
|
|
63
|
+
*/
|
|
64
|
+
export declare function countJudgeFailed(run: RunLike | null | undefined): number;
|
|
65
|
+
export interface RunActionVisibility {
|
|
66
|
+
/** Delete is always available for any run in any status. */
|
|
67
|
+
canDelete: boolean;
|
|
68
|
+
/** Cancel is available only while the run is actively running. */
|
|
69
|
+
canCancel: boolean;
|
|
70
|
+
/** Re-run is available only for top-level EvaluationRun docs. */
|
|
71
|
+
canRerun: boolean;
|
|
72
|
+
/** Reason to show (e.g. as a disabled-item tooltip) when canRerun is false. */
|
|
73
|
+
rerunDisabledReason?: string;
|
|
74
|
+
/** Retry judgement: EvaluationRun, terminal, with >0 judge-failed cases. */
|
|
75
|
+
canRetryJudgement: boolean;
|
|
76
|
+
/** Reason to show when canRetryJudgement is false. */
|
|
77
|
+
retryJudgementDisabledReason?: string;
|
|
78
|
+
/** Number of judge-failed test cases (0 when not applicable/unknown). */
|
|
79
|
+
judgeFailedCount: number;
|
|
80
|
+
}
|
|
81
|
+
/**
|
|
82
|
+
* Minimum time a run must have been persisted before a Cancel request with
|
|
83
|
+
* no in-memory executor token is allowed to take the "zombie" fallback path
|
|
84
|
+
* (mark cancelled directly — see getRunActionVisibility callers in the
|
|
85
|
+
* cancel routes). Guards the narrow window right after a run is created:
|
|
86
|
+
* the doc is persisted (and therefore visible to a concurrent Cancel
|
|
87
|
+
* request) strictly before its executor registers its cancellation token,
|
|
88
|
+
* so a Cancel that lands in that gap would otherwise mark a run "cancelled"
|
|
89
|
+
* moments before its own executor starts making progress on it. A brand-new
|
|
90
|
+
* run is also the case the fallback is LEAST useful for — "zombie" (no
|
|
91
|
+
* executor anywhere) is far more plausible once a run has been running for
|
|
92
|
+
* a while than in its first couple of seconds.
|
|
93
|
+
*/
|
|
94
|
+
export declare const ZOMBIE_CANCEL_MIN_AGE_MS = 5000;
|
|
95
|
+
/**
|
|
96
|
+
* True once a run is old enough that a tokenless Cancel request can safely
|
|
97
|
+
* assume its executor (if any) would already have registered a
|
|
98
|
+
* cancellation token — i.e. it's safe to treat "no token" as "no executor"
|
|
99
|
+
* rather than "executor hasn't started yet".
|
|
100
|
+
*
|
|
101
|
+
* NOTE — known limitation, not fixed by this check: cancellation tokens are
|
|
102
|
+
* tracked in an in-memory `Map` scoped to ONE server process. In a
|
|
103
|
+
* multi-process/clustered deployment, a Cancel request routed to a
|
|
104
|
+
* DIFFERENT process than the one executing the run will always find no
|
|
105
|
+
* token there, regardless of run age, and this zombie fallback will mark
|
|
106
|
+
* the run cancelled in storage even though it's alive and progressing on
|
|
107
|
+
* another process. This mirrors a pre-existing, documented constraint of
|
|
108
|
+
* this codebase's run-execution model (see AGENTS.md's "orphan-run
|
|
109
|
+
* recovery" notes: "active is tracked per-process"); fixing it for real
|
|
110
|
+
* needs the same heartbeat-based ownership (`run.heartbeatAt`) that doc
|
|
111
|
+
* already calls out as the eventual replacement. Out of scope here.
|
|
112
|
+
*/
|
|
113
|
+
export declare function isOldEnoughForZombieCancel(createdAt: string | undefined, now?: number): boolean;
|
|
114
|
+
/**
|
|
115
|
+
* Compute the full action-visibility matrix for one run. Pure function —
|
|
116
|
+
* safe to call from both React components and server-side route validation
|
|
117
|
+
* so the two never drift.
|
|
118
|
+
*/
|
|
119
|
+
export declare function getRunActionVisibility(run: RunLike | null | undefined): RunActionVisibility;
|
|
120
|
+
//# sourceMappingURL=runActions.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"runActions.d.ts","sourceRoot":"","sources":["../../runActions.ts"],"names":[],"mappings":"AAKA;;;;;;;;;;;;;;;;;;;;;;;;;GAyBG;AAEH,OAAO,KAAK,EAAE,YAAY,EAAE,aAAa,EAAE,MAAM,SAAS,CAAC;AAG3D,kFAAkF;AAClF,MAAM,MAAM,OAAO,GAAG,IAAI,CAAC,YAAY,GAAG,aAAa,EAAE,QAAQ,GAAG,SAAS,CAAC,GAAG;IAC/E,OAAO,CAAC,EAAE,MAAM,CAAC;CAClB,CAAC;AAEF;;;;;;;;;;GAUG;AACH,wBAAgB,eAAe,CAAC,GAAG,EAAE,OAAO,GAAG,IAAI,GAAG,SAAS,GAAG,GAAG,IAAI,aAAa,CAErF;AAED,sFAAsF;AACtF,wBAAgB,YAAY,CAAC,GAAG,EAAE,OAAO,GAAG,IAAI,GAAG,SAAS,GAAG,OAAO,CAErE;AAED,4EAA4E;AAC5E,wBAAgB,aAAa,CAAC,GAAG,EAAE,OAAO,GAAG,IAAI,GAAG,SAAS,GAAG,OAAO,CAEtE;AAED;;;;;;;;;;;;;;;GAeG;AACH,wBAAgB,gBAAgB,CAAC,GAAG,EAAE,OAAO,GAAG,IAAI,GAAG,SAAS,GAAG,MAAM,CAUxE;AAED,MAAM,WAAW,mBAAmB;IAClC,4DAA4D;IAC5D,SAAS,EAAE,OAAO,CAAC;IACnB,kEAAkE;IAClE,SAAS,EAAE,OAAO,CAAC;IACnB,iEAAiE;IACjE,QAAQ,EAAE,OAAO,CAAC;IAClB,+EAA+E;IAC/E,mBAAmB,CAAC,EAAE,MAAM,CAAC;IAC7B,4EAA4E;IAC5E,iBAAiB,EAAE,OAAO,CAAC;IAC3B,sDAAsD;IACtD,4BAA4B,CAAC,EAAE,MAAM,CAAC;IACtC,yEAAyE;IACzE,gBAAgB,EAAE,MAAM,CAAC;CAC1B;AAOD;;;;;;;;;;;;GAYG;AACH,eAAO,MAAM,wBAAwB,OAAO,CAAC;AAE7C;;;;;;;;;;;;;;;;;GAiBG;AACH,wBAAgB,0BAA0B,CAAC,SAAS,EAAE,MAAM,GAAG,SAAS,EAAE,GAAG,GAAE,MAAmB,GAAG,OAAO,CAI3G;AAED;;;;GAIG;AACH,wBAAgB,sBAAsB,CAAC,GAAG,EAAE,OAAO,GAAG,IAAI,GAAG,SAAS,GAAG,mBAAmB,CAuB3F"}
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Copyright OpenSearch Contributors
|
|
3
|
+
* SPDX-License-Identifier: Apache-2.0
|
|
4
|
+
*/
|
|
5
|
+
import { isEvaluationRun as isEvaluationRunDoc } from '../types/index.js';
|
|
6
|
+
/**
|
|
7
|
+
* True when `run` is a top-level EvaluationRun document (created via
|
|
8
|
+
* `POST /api/storage/evaluation-runs`), as opposed to a legacy
|
|
9
|
+
* benchmark-embedded BenchmarkRun (`benchmark.runs[]`). The two share a lot
|
|
10
|
+
* of shape but only EvaluationRun docs carry `docType: 'evaluation-run'` and
|
|
11
|
+
* support the rerun/retry-judgement endpoints.
|
|
12
|
+
*
|
|
13
|
+
* Null-tolerant wrapper over the typed predicate in `types/index.ts` (the
|
|
14
|
+
* single source of truth for the docType discriminator) — kept so callers
|
|
15
|
+
* holding a possibly-null run don't need their own guard.
|
|
16
|
+
*/
|
|
17
|
+
export function isEvaluationRun(run) {
|
|
18
|
+
return !!run && isEvaluationRunDoc(run);
|
|
19
|
+
}
|
|
20
|
+
/** True while the run has an in-progress executor that a Cancel action could stop. */
|
|
21
|
+
export function isRunRunning(run) {
|
|
22
|
+
return run?.status === 'running';
|
|
23
|
+
}
|
|
24
|
+
/** True once a run has reached any terminal state (not running/pending). */
|
|
25
|
+
export function isRunTerminal(run) {
|
|
26
|
+
return !!run && (run.status === 'completed' || run.status === 'failed' || run.status === 'cancelled');
|
|
27
|
+
}
|
|
28
|
+
/**
|
|
29
|
+
* Count test cases where the AGENT finished (`status === 'completed'`) but
|
|
30
|
+
* the JUDGE produced no verdict (`passFailStatus` neither 'passed' nor
|
|
31
|
+
* 'failed') — the "errored" bucket of `lib/runStats` `bucketRunResults`
|
|
32
|
+
* (issue #242) and exactly the set `POST .../retry-judgement` (default
|
|
33
|
+
* `scope=errored`) will re-judge. Deliberately excludes:
|
|
34
|
+
* - `status !== 'completed'` (agent-failed/cancelled/pending cases — no
|
|
35
|
+
* trajectory to re-judge, or nothing ran).
|
|
36
|
+
* - a real 'failed' verdict — the judge DID run and graded the case; that
|
|
37
|
+
* is a legitimate result, not a judge failure (re-grading it is
|
|
38
|
+
* `scope=all`, opt-in from the inspector's dedicated button).
|
|
39
|
+
*
|
|
40
|
+
* `passFailStatus` isn't declared on `EvaluationRun['results']`'s static
|
|
41
|
+
* type (a pre-existing gap — evaluationRunner.ts writes it via an `as any`
|
|
42
|
+
* spread) so this reads it defensively.
|
|
43
|
+
*/
|
|
44
|
+
export function countJudgeFailed(run) {
|
|
45
|
+
if (!run?.results)
|
|
46
|
+
return 0;
|
|
47
|
+
let count = 0;
|
|
48
|
+
for (const r of Object.values(run.results)) {
|
|
49
|
+
const result = r;
|
|
50
|
+
if (result.status !== 'completed')
|
|
51
|
+
continue;
|
|
52
|
+
if (result.passFailStatus === 'passed' || result.passFailStatus === 'failed')
|
|
53
|
+
continue;
|
|
54
|
+
count++;
|
|
55
|
+
}
|
|
56
|
+
return count;
|
|
57
|
+
}
|
|
58
|
+
const RERUN_NOT_SUPPORTED_REASON = "Re-run isn't available for legacy benchmark-embedded runs";
|
|
59
|
+
const RETRY_JUDGEMENT_NOT_SUPPORTED_REASON = "Retry judgement isn't available for legacy benchmark-embedded runs";
|
|
60
|
+
const RETRY_JUDGEMENT_STILL_RUNNING_REASON = 'Retry judgement is only available once the run finishes';
|
|
61
|
+
const RETRY_JUDGEMENT_NONE_FAILED_REASON = 'No judge-failed test cases to retry';
|
|
62
|
+
/**
|
|
63
|
+
* Minimum time a run must have been persisted before a Cancel request with
|
|
64
|
+
* no in-memory executor token is allowed to take the "zombie" fallback path
|
|
65
|
+
* (mark cancelled directly — see getRunActionVisibility callers in the
|
|
66
|
+
* cancel routes). Guards the narrow window right after a run is created:
|
|
67
|
+
* the doc is persisted (and therefore visible to a concurrent Cancel
|
|
68
|
+
* request) strictly before its executor registers its cancellation token,
|
|
69
|
+
* so a Cancel that lands in that gap would otherwise mark a run "cancelled"
|
|
70
|
+
* moments before its own executor starts making progress on it. A brand-new
|
|
71
|
+
* run is also the case the fallback is LEAST useful for — "zombie" (no
|
|
72
|
+
* executor anywhere) is far more plausible once a run has been running for
|
|
73
|
+
* a while than in its first couple of seconds.
|
|
74
|
+
*/
|
|
75
|
+
export const ZOMBIE_CANCEL_MIN_AGE_MS = 5000;
|
|
76
|
+
/**
|
|
77
|
+
* True once a run is old enough that a tokenless Cancel request can safely
|
|
78
|
+
* assume its executor (if any) would already have registered a
|
|
79
|
+
* cancellation token — i.e. it's safe to treat "no token" as "no executor"
|
|
80
|
+
* rather than "executor hasn't started yet".
|
|
81
|
+
*
|
|
82
|
+
* NOTE — known limitation, not fixed by this check: cancellation tokens are
|
|
83
|
+
* tracked in an in-memory `Map` scoped to ONE server process. In a
|
|
84
|
+
* multi-process/clustered deployment, a Cancel request routed to a
|
|
85
|
+
* DIFFERENT process than the one executing the run will always find no
|
|
86
|
+
* token there, regardless of run age, and this zombie fallback will mark
|
|
87
|
+
* the run cancelled in storage even though it's alive and progressing on
|
|
88
|
+
* another process. This mirrors a pre-existing, documented constraint of
|
|
89
|
+
* this codebase's run-execution model (see AGENTS.md's "orphan-run
|
|
90
|
+
* recovery" notes: "active is tracked per-process"); fixing it for real
|
|
91
|
+
* needs the same heartbeat-based ownership (`run.heartbeatAt`) that doc
|
|
92
|
+
* already calls out as the eventual replacement. Out of scope here.
|
|
93
|
+
*/
|
|
94
|
+
export function isOldEnoughForZombieCancel(createdAt, now = Date.now()) {
|
|
95
|
+
const created = createdAt ? Date.parse(createdAt) : NaN;
|
|
96
|
+
if (Number.isNaN(created))
|
|
97
|
+
return true; // no timestamp to compare against — don't block on it
|
|
98
|
+
return now - created >= ZOMBIE_CANCEL_MIN_AGE_MS;
|
|
99
|
+
}
|
|
100
|
+
/**
|
|
101
|
+
* Compute the full action-visibility matrix for one run. Pure function —
|
|
102
|
+
* safe to call from both React components and server-side route validation
|
|
103
|
+
* so the two never drift.
|
|
104
|
+
*/
|
|
105
|
+
export function getRunActionVisibility(run) {
|
|
106
|
+
const evalRun = isEvaluationRun(run);
|
|
107
|
+
const running = isRunRunning(run);
|
|
108
|
+
const terminal = isRunTerminal(run);
|
|
109
|
+
const judgeFailedCount = evalRun ? countJudgeFailed(run) : 0;
|
|
110
|
+
const canRetryJudgement = evalRun && terminal && judgeFailedCount > 0;
|
|
111
|
+
let retryJudgementDisabledReason;
|
|
112
|
+
if (!canRetryJudgement) {
|
|
113
|
+
if (!evalRun)
|
|
114
|
+
retryJudgementDisabledReason = RETRY_JUDGEMENT_NOT_SUPPORTED_REASON;
|
|
115
|
+
else if (!terminal)
|
|
116
|
+
retryJudgementDisabledReason = RETRY_JUDGEMENT_STILL_RUNNING_REASON;
|
|
117
|
+
else
|
|
118
|
+
retryJudgementDisabledReason = RETRY_JUDGEMENT_NONE_FAILED_REASON;
|
|
119
|
+
}
|
|
120
|
+
return {
|
|
121
|
+
canDelete: true,
|
|
122
|
+
canCancel: running,
|
|
123
|
+
canRerun: evalRun,
|
|
124
|
+
rerunDisabledReason: evalRun ? undefined : RERUN_NOT_SUPPORTED_REASON,
|
|
125
|
+
canRetryJudgement,
|
|
126
|
+
retryJudgementDisabledReason,
|
|
127
|
+
judgeFailedCount,
|
|
128
|
+
};
|
|
129
|
+
}
|
|
130
|
+
//# sourceMappingURL=runActions.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"runActions.js","sourceRoot":"","sources":["../../runActions.ts"],"names":[],"mappings":"AAAA;;;GAGG;AA8BH,OAAO,EAAE,eAAe,IAAI,kBAAkB,EAAE,MAAM,SAAS,CAAC;AAOhE;;;;;;;;;;GAUG;AACH,MAAM,UAAU,eAAe,CAAC,GAA+B;IAC7D,OAAO,CAAC,CAAC,GAAG,IAAI,kBAAkB,CAAC,GAAmC,CAAC,CAAC;AAC1E,CAAC;AAED,sFAAsF;AACtF,MAAM,UAAU,YAAY,CAAC,GAA+B;IAC1D,OAAO,GAAG,EAAE,MAAM,KAAK,SAAS,CAAC;AACnC,CAAC;AAED,4EAA4E;AAC5E,MAAM,UAAU,aAAa,CAAC,GAA+B;IAC3D,OAAO,CAAC,CAAC,GAAG,IAAI,CAAC,GAAG,CAAC,MAAM,KAAK,WAAW,IAAI,GAAG,CAAC,MAAM,KAAK,QAAQ,IAAI,GAAG,CAAC,MAAM,KAAK,WAAW,CAAC,CAAC;AACxG,CAAC;AAED;;;;;;;;;;;;;;;GAeG;AACH,MAAM,UAAU,gBAAgB,CAAC,GAA+B;IAC9D,IAAI,CAAC,GAAG,EAAE,OAAO;QAAE,OAAO,CAAC,CAAC;IAC5B,IAAI,KAAK,GAAG,CAAC,CAAC;IACd,KAAK,MAAM,CAAC,IAAI,MAAM,CAAC,MAAM,CAAC,GAAG,CAAC,OAAO,CAAC,EAAE,CAAC;QAC3C,MAAM,MAAM,GAAG,CAAwD,CAAC;QACxE,IAAI,MAAM,CAAC,MAAM,KAAK,WAAW;YAAE,SAAS;QAC5C,IAAI,MAAM,CAAC,cAAc,KAAK,QAAQ,IAAI,MAAM,CAAC,cAAc,KAAK,QAAQ;YAAE,SAAS;QACvF,KAAK,EAAE,CAAC;IACV,CAAC;IACD,OAAO,KAAK,CAAC;AACf,CAAC;AAmBD,MAAM,0BAA0B,GAAG,2DAA2D,CAAC;AAC/F,MAAM,oCAAoC,GAAG,oEAAoE,CAAC;AAClH,MAAM,oCAAoC,GAAG,yDAAyD,CAAC;AACvG,MAAM,kCAAkC,GAAG,qCAAqC,CAAC;AAEjF;;;;;;;;;;;;GAYG;AACH,MAAM,CAAC,MAAM,wBAAwB,GAAG,IAAI,CAAC;AAE7C;;;;;;;;;;;;;;;;;GAiBG;AACH,MAAM,UAAU,0BAA0B,CAAC,SAA6B,EAAE,MAAc,IAAI,CAAC,GAAG,EAAE;IAChG,MAAM,OAAO,GAAG,SAAS,CAAC,CAAC,CAAC,IAAI,CAAC,KAAK,CAAC,SAAS,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC;IACxD,IAAI,MAAM,CAAC,KAAK,CAAC,OAAO,CAAC;QAAE,OAAO,IAAI,CAAC,CAAC,sDAAsD;IAC9F,OAAO,GAAG,GAAG,OAAO,IAAI,wBAAwB,CAAC;AACnD,CAAC;AAED;;;;GAIG;AACH,MAAM,UAAU,sBAAsB,CAAC,GAA+B;IACpE,MAAM,OAAO,GAAG,eAAe,CAAC,GAAG,CAAC,CAAC;IACrC,MAAM,OAAO,GAAG,YAAY,CAAC,GAAG,CAAC,CAAC;IAClC,MAAM,QAAQ,GAAG,aAAa,CAAC,GAAG,CAAC,CAAC;IACpC,MAAM,gBAAgB,GAAG,OAAO,CAAC,CAAC,CAAC,gBAAgB,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;IAE7D,MAAM,iBAAiB,GAAG,OAAO,IAAI,QAAQ,IAAI,gBAAgB,GAAG,CAAC,CAAC;IACtE,IAAI,4BAAgD,CAAC;IACrD,IAAI,CAAC,iBAAiB,EAAE,CAAC;QACvB,IAAI,CAAC,OAAO;YAAE,4BAA4B,GAAG,oCAAoC,CAAC;aAC7E,IAAI,CAAC,QAAQ;YAAE,4BAA4B,GAAG,oCAAoC,CAAC;;YACnF,4BAA4B,GAAG,kCAAkC,CAAC;IACzE,CAAC;IAED,OAAO;QACL,SAAS,EAAE,IAAI;QACf,SAAS,EAAE,OAAO;QAClB,QAAQ,EAAE,OAAO;QACjB,mBAAmB,EAAE,OAAO,CAAC,CAAC,CAAC,SAAS,CAAC,CAAC,CAAC,0BAA0B;QACrE,iBAAiB;QACjB,4BAA4B;QAC5B,gBAAgB;KACjB,CAAC;AACJ,CAAC"}
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Deterministic (no-LLM) aggregation helpers for the run-report "insights"
|
|
3
|
+
* pane (RunInsightsPane.tsx), shown on the bare
|
|
4
|
+
* `/benchmarks/:benchmarkId/runs/:runId` route when no test case is
|
|
5
|
+
* selected — see owner feedback on goyamegh/run-report-redesign: "if no
|
|
6
|
+
* test case is selected, the right side can show an aggregated view ...
|
|
7
|
+
* why did the failing tests fail — something that is complete info."
|
|
8
|
+
*
|
|
9
|
+
* Everything here is a pure function over already-fetched data (report
|
|
10
|
+
* summaries + test-case categories). No network calls, no LLM calls — v1
|
|
11
|
+
* is explicitly deterministic-only per the product ask.
|
|
12
|
+
*/
|
|
13
|
+
export interface CategoryStatusRow {
|
|
14
|
+
category: string;
|
|
15
|
+
/** Any ResultStatus value; only 'passed' / 'failed' / 'errored' are counted distinctly, everything else falls into the bar's `total` only (pending/running cases). */
|
|
16
|
+
status: string;
|
|
17
|
+
}
|
|
18
|
+
export interface CategoryBar {
|
|
19
|
+
category: string;
|
|
20
|
+
passed: number;
|
|
21
|
+
failed: number;
|
|
22
|
+
errored: number;
|
|
23
|
+
total: number;
|
|
24
|
+
}
|
|
25
|
+
/**
|
|
26
|
+
* Group rows by test-case category and tally pass/fail/errored/total.
|
|
27
|
+
* Deterministic order: largest category first, ties broken alphabetically.
|
|
28
|
+
*/
|
|
29
|
+
export declare function computeCategoryBars(rows: CategoryStatusRow[]): CategoryBar[];
|
|
30
|
+
/**
|
|
31
|
+
* Normalize a judge-reasoning string down to its first sentence,
|
|
32
|
+
* lowercased, punctuation-stripped, whitespace-collapsed. Used both as the
|
|
33
|
+
* clustering input and as the theme's stable `key`.
|
|
34
|
+
*/
|
|
35
|
+
export declare function normalizeReasoningKey(reasoning: string): string;
|
|
36
|
+
export interface FailureThemeInput {
|
|
37
|
+
testCaseId: string;
|
|
38
|
+
reasoning: string;
|
|
39
|
+
}
|
|
40
|
+
export interface FailureTheme {
|
|
41
|
+
/** Stable cluster key — the normalized first sentence of the theme's representative case. */
|
|
42
|
+
key: string;
|
|
43
|
+
count: number;
|
|
44
|
+
/** Trimmed, human-readable first sentence sampled from the theme's most common exact phrasing. */
|
|
45
|
+
sampleSnippet: string;
|
|
46
|
+
testCaseIds: string[];
|
|
47
|
+
}
|
|
48
|
+
/**
|
|
49
|
+
* Cluster failing test cases into "why they failed" themes using a
|
|
50
|
+
* deterministic, LLM-free heuristic: normalize each case's judge-reasoning
|
|
51
|
+
* first sentence, then union-find cases that share at least one contiguous
|
|
52
|
+
* N-word shingle. This is robust to minor paraphrasing (a judge saying
|
|
53
|
+
* "unable to retrieve" vs "failed to retrieve" the same underlying tool
|
|
54
|
+
* connectivity failure) while still keeping genuinely distinct failure
|
|
55
|
+
* modes (e.g. "missing required facts" vs "MCP server unavailable")
|
|
56
|
+
* separate, because they share no contiguous phrase.
|
|
57
|
+
*
|
|
58
|
+
* Verified against a real production run (418-verify, 64 failing cases):
|
|
59
|
+
* 57 of 64 connectivity-flavored reasonings collapse into ONE dominant
|
|
60
|
+
* theme; the remaining 7 ("Required facts evaluation: ...", a genuinely
|
|
61
|
+
* different failure shape) form a second, correctly separate theme.
|
|
62
|
+
*
|
|
63
|
+
* Output order: largest theme first, ties broken by the theme's
|
|
64
|
+
* lowest-sorting testCaseId (deterministic, no dependency on input order).
|
|
65
|
+
*/
|
|
66
|
+
export declare function clusterFailureThemes(items: FailureThemeInput[], shingleSize?: number): FailureTheme[];
|
|
67
|
+
/**
|
|
68
|
+
* "Based on N of M failing cases" note shown under the theme list when the
|
|
69
|
+
* reasoning fetch was capped (RunInsightsPane caps at the first 100 failing
|
|
70
|
+
* cases). Returns null when nothing was capped (fetchedCount >= totalCount).
|
|
71
|
+
*/
|
|
72
|
+
export declare function formatCappedNote(fetchedCount: number, totalFailingCount: number): string | null;
|
|
73
|
+
export interface RankedCase {
|
|
74
|
+
testCaseId: string;
|
|
75
|
+
value: number;
|
|
76
|
+
}
|
|
77
|
+
/**
|
|
78
|
+
* Top-N cases by a numeric value (duration, cost, ...), descending.
|
|
79
|
+
* Cases with a null/undefined/NaN value are excluded. Deterministic tie
|
|
80
|
+
* break: lower testCaseId first.
|
|
81
|
+
*/
|
|
82
|
+
export declare function pickTopN(cases: {
|
|
83
|
+
testCaseId: string;
|
|
84
|
+
value: number | null | undefined;
|
|
85
|
+
}[], n: number): RankedCase[];
|
|
86
|
+
//# sourceMappingURL=runInsights.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"runInsights.d.ts","sourceRoot":"","sources":["../../runInsights.ts"],"names":[],"mappings":"AAKA;;;;;;;;;;;GAWG;AAIH,MAAM,WAAW,iBAAiB;IAChC,QAAQ,EAAE,MAAM,CAAC;IACjB,sKAAsK;IACtK,MAAM,EAAE,MAAM,CAAC;CAChB;AAED,MAAM,WAAW,WAAW;IAC1B,QAAQ,EAAE,MAAM,CAAC;IACjB,MAAM,EAAE,MAAM,CAAC;IACf,MAAM,EAAE,MAAM,CAAC;IACf,OAAO,EAAE,MAAM,CAAC;IAChB,KAAK,EAAE,MAAM,CAAC;CACf;AAID;;;GAGG;AACH,wBAAgB,mBAAmB,CAAC,IAAI,EAAE,iBAAiB,EAAE,GAAG,WAAW,EAAE,CAiB5E;AAID;;;;GAIG;AACH,wBAAgB,qBAAqB,CAAC,SAAS,EAAE,MAAM,GAAG,MAAM,CAM/D;AAqBD,MAAM,WAAW,iBAAiB;IAChC,UAAU,EAAE,MAAM,CAAC;IACnB,SAAS,EAAE,MAAM,CAAC;CACnB;AAED,MAAM,WAAW,YAAY;IAC3B,6FAA6F;IAC7F,GAAG,EAAE,MAAM,CAAC;IACZ,KAAK,EAAE,MAAM,CAAC;IACd,kGAAkG;IAClG,aAAa,EAAE,MAAM,CAAC;IACtB,WAAW,EAAE,MAAM,EAAE,CAAC;CACvB;AAOD;;;;;;;;;;;;;;;;;GAiBG;AACH,wBAAgB,oBAAoB,CAClC,KAAK,EAAE,iBAAiB,EAAE,EAC1B,WAAW,GAAE,MAA6B,GACzC,YAAY,EAAE,CAwEhB;AAED;;;;GAIG;AACH,wBAAgB,gBAAgB,CAAC,YAAY,EAAE,MAAM,EAAE,iBAAiB,EAAE,MAAM,GAAG,MAAM,GAAG,IAAI,CAG/F;AAID,MAAM,WAAW,UAAU;IACzB,UAAU,EAAE,MAAM,CAAC;IACnB,KAAK,EAAE,MAAM,CAAC;CACf;AAED;;;;GAIG;AACH,wBAAgB,QAAQ,CAAC,KAAK,EAAE;IAAE,UAAU,EAAE,MAAM,CAAC;IAAC,KAAK,EAAE,MAAM,GAAG,IAAI,GAAG,SAAS,CAAA;CAAE,EAAE,EAAE,CAAC,EAAE,MAAM,GAAG,UAAU,EAAE,CAKnH"}
|