@opensearch-project/agent-health 0.6.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/cli/dist/index.js +722 -330
- package/dist/assets/index-D-Np_l_T.js +246 -0
- package/dist/assets/index-vZt9QZKf.css +1 -0
- package/dist/index.html +2 -2
- package/docs/CLI.md +138 -1
- package/docs/CONNECTORS.md +1 -1
- package/docs/SDK.md +126 -2
- package/docs/skills/add-connector/SKILL.md +5 -1
- package/lib/dist/lib/agentTrends.d.ts +210 -0
- package/lib/dist/lib/agentTrends.d.ts.map +1 -0
- package/lib/dist/lib/agentTrends.js +360 -0
- package/lib/dist/lib/agentTrends.js.map +1 -0
- package/lib/dist/lib/benchmarkCaseReview.d.ts +114 -0
- package/lib/dist/lib/benchmarkCaseReview.d.ts.map +1 -0
- package/lib/dist/lib/benchmarkCaseReview.js +177 -0
- package/lib/dist/lib/benchmarkCaseReview.js.map +1 -0
- package/lib/dist/lib/benchmarkRunsTable.d.ts +109 -0
- package/lib/dist/lib/benchmarkRunsTable.d.ts.map +1 -0
- package/lib/dist/lib/benchmarkRunsTable.js +212 -0
- package/lib/dist/lib/benchmarkRunsTable.js.map +1 -0
- package/lib/dist/lib/comparisonInsights.d.ts +49 -2
- package/lib/dist/lib/comparisonInsights.d.ts.map +1 -1
- package/lib/dist/lib/comparisonInsights.js +65 -7
- package/lib/dist/lib/comparisonInsights.js.map +1 -1
- package/lib/dist/lib/config/loader.d.ts.map +1 -1
- package/lib/dist/lib/config/loader.js +11 -1
- package/lib/dist/lib/config/loader.js.map +1 -1
- package/lib/dist/lib/dashboardMetrics.d.ts +11 -2
- package/lib/dist/lib/dashboardMetrics.d.ts.map +1 -1
- package/lib/dist/lib/dashboardMetrics.js +38 -3
- package/lib/dist/lib/dashboardMetrics.js.map +1 -1
- package/lib/dist/lib/evaluationRerun.d.ts +39 -0
- package/lib/dist/lib/evaluationRerun.d.ts.map +1 -1
- package/lib/dist/lib/evaluationRerun.js +49 -0
- package/lib/dist/lib/evaluationRerun.js.map +1 -1
- package/lib/dist/lib/judgeFailureSummary.d.ts +66 -0
- package/lib/dist/lib/judgeFailureSummary.d.ts.map +1 -0
- package/lib/dist/lib/judgeFailureSummary.js +68 -0
- package/lib/dist/lib/judgeFailureSummary.js.map +1 -0
- package/lib/dist/lib/judgeStrategies.d.ts +108 -0
- package/lib/dist/lib/judgeStrategies.d.ts.map +1 -0
- package/lib/dist/lib/judgeStrategies.js +135 -0
- package/lib/dist/lib/judgeStrategies.js.map +1 -0
- package/lib/dist/lib/matchers/expect.d.ts +21 -1
- package/lib/dist/lib/matchers/expect.d.ts.map +1 -1
- package/lib/dist/lib/matchers/expect.js +51 -0
- package/lib/dist/lib/matchers/expect.js.map +1 -1
- package/lib/dist/lib/matchers/judgeAccessor.d.ts +4 -0
- package/lib/dist/lib/matchers/judgeAccessor.d.ts.map +1 -1
- package/lib/dist/lib/matchers/judgeAccessor.js +13 -2
- package/lib/dist/lib/matchers/judgeAccessor.js.map +1 -1
- package/lib/dist/lib/matchers/judgeReasoningParse.d.ts +49 -0
- package/lib/dist/lib/matchers/judgeReasoningParse.d.ts.map +1 -0
- package/lib/dist/lib/matchers/judgeReasoningParse.js +128 -0
- package/lib/dist/lib/matchers/judgeReasoningParse.js.map +1 -0
- package/lib/dist/lib/matchers/types.d.ts +25 -0
- package/lib/dist/lib/matchers/types.d.ts.map +1 -1
- package/lib/dist/lib/resolveCanonicalRun.d.ts +23 -0
- package/lib/dist/lib/resolveCanonicalRun.d.ts.map +1 -0
- package/lib/dist/lib/resolveCanonicalRun.js +26 -0
- package/lib/dist/lib/resolveCanonicalRun.js.map +1 -0
- package/lib/dist/lib/runActions.d.ts +120 -0
- package/lib/dist/lib/runActions.d.ts.map +1 -0
- package/lib/dist/lib/runActions.js +130 -0
- package/lib/dist/lib/runActions.js.map +1 -0
- package/lib/dist/lib/runInsights.d.ts +86 -0
- package/lib/dist/lib/runInsights.d.ts.map +1 -0
- package/lib/dist/lib/runInsights.js +185 -0
- package/lib/dist/lib/runInsights.js.map +1 -0
- package/lib/dist/lib/runName.d.ts +29 -0
- package/lib/dist/lib/runName.d.ts.map +1 -0
- package/lib/dist/lib/runName.js +38 -0
- package/lib/dist/lib/runName.js.map +1 -0
- package/lib/dist/lib/runReportPath.d.ts +16 -0
- package/lib/dist/lib/runReportPath.d.ts.map +1 -0
- package/lib/dist/lib/runReportPath.js +22 -0
- package/lib/dist/lib/runReportPath.js.map +1 -0
- package/lib/dist/lib/runSort.d.ts +27 -0
- package/lib/dist/lib/runSort.d.ts.map +1 -0
- package/lib/dist/lib/runSort.js +31 -0
- package/lib/dist/lib/runSort.js.map +1 -0
- package/lib/dist/lib/runStats.d.ts +86 -6
- package/lib/dist/lib/runStats.d.ts.map +1 -1
- package/lib/dist/lib/runStats.js +170 -19
- package/lib/dist/lib/runStats.js.map +1 -1
- package/lib/dist/lib/testCases/define.d.ts.map +1 -1
- package/lib/dist/lib/testCases/define.js +93 -46
- package/lib/dist/lib/testCases/define.js.map +1 -1
- package/lib/dist/lib/testCases/judge.d.ts.map +1 -1
- package/lib/dist/lib/testCases/judge.js +22 -4
- package/lib/dist/lib/testCases/judge.js.map +1 -1
- package/lib/dist/lib/testCases/loader.d.ts +24 -0
- package/lib/dist/lib/testCases/loader.d.ts.map +1 -1
- package/lib/dist/lib/testCases/loader.js +253 -36
- package/lib/dist/lib/testCases/loader.js.map +1 -1
- package/lib/dist/lib/trajectoryStepDisplay.d.ts +25 -0
- package/lib/dist/lib/trajectoryStepDisplay.d.ts.map +1 -0
- package/lib/dist/lib/trajectoryStepDisplay.js +42 -0
- package/lib/dist/lib/trajectoryStepDisplay.js.map +1 -0
- package/lib/dist/lib/utils.d.ts +19 -0
- package/lib/dist/lib/utils.d.ts.map +1 -1
- package/lib/dist/lib/utils.js +27 -0
- package/lib/dist/lib/utils.js.map +1 -1
- package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.d.ts +68 -32
- package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.d.ts.map +1 -1
- package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.js +148 -125
- package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.js.map +1 -1
- package/lib/dist/services/connectors/kiro/KiroConnector.d.ts +21 -13
- package/lib/dist/services/connectors/kiro/KiroConnector.d.ts.map +1 -1
- package/lib/dist/services/connectors/kiro/KiroConnector.js +22 -25
- package/lib/dist/services/connectors/kiro/KiroConnector.js.map +1 -1
- package/lib/dist/services/connectors/pi/PiConnector.d.ts +49 -10
- package/lib/dist/services/connectors/pi/PiConnector.d.ts.map +1 -1
- package/lib/dist/services/connectors/pi/PiConnector.js +102 -86
- package/lib/dist/services/connectors/pi/PiConnector.js.map +1 -1
- package/lib/dist/services/connectors/subprocess/SubprocessConnector.d.ts +59 -14
- package/lib/dist/services/connectors/subprocess/SubprocessConnector.d.ts.map +1 -1
- package/lib/dist/services/connectors/subprocess/SubprocessConnector.js +87 -61
- package/lib/dist/services/connectors/subprocess/SubprocessConnector.js.map +1 -1
- package/lib/dist/services/evaluation/bedrockJudge.d.ts +11 -0
- package/lib/dist/services/evaluation/bedrockJudge.d.ts.map +1 -1
- package/lib/dist/services/evaluation/bedrockJudge.js +2 -0
- package/lib/dist/services/evaluation/bedrockJudge.js.map +1 -1
- package/lib/dist/services/evaluation/index.d.ts +18 -1
- package/lib/dist/services/evaluation/index.d.ts.map +1 -1
- package/lib/dist/services/evaluation/index.js +156 -19
- package/lib/dist/services/evaluation/index.js.map +1 -1
- package/lib/dist/services/metrics.d.ts +55 -0
- package/lib/dist/services/metrics.d.ts.map +1 -0
- package/lib/dist/services/metrics.js +89 -0
- package/lib/dist/services/metrics.js.map +1 -0
- package/lib/dist/services/storage/asyncBenchmarkStorage.d.ts.map +1 -1
- package/lib/dist/services/storage/asyncBenchmarkStorage.js +16 -0
- package/lib/dist/services/storage/asyncBenchmarkStorage.js.map +1 -1
- package/lib/dist/services/storage/asyncRunStorage.d.ts +11 -0
- package/lib/dist/services/storage/asyncRunStorage.d.ts.map +1 -1
- package/lib/dist/services/storage/asyncRunStorage.js +49 -2
- package/lib/dist/services/storage/asyncRunStorage.js.map +1 -1
- package/lib/dist/services/storage/asyncTestCaseStorage.d.ts +1 -1
- package/lib/dist/services/storage/asyncTestCaseStorage.d.ts.map +1 -1
- package/lib/dist/services/storage/asyncTestCaseStorage.js +2 -1
- package/lib/dist/services/storage/asyncTestCaseStorage.js.map +1 -1
- package/lib/dist/services/storage/opensearchClient.d.ts +12 -0
- package/lib/dist/services/storage/opensearchClient.d.ts.map +1 -1
- package/lib/dist/services/storage/opensearchClient.js.map +1 -1
- package/lib/dist/services/traces/browserRecovery.d.ts.map +1 -1
- package/lib/dist/services/traces/browserRecovery.js +3 -0
- package/lib/dist/services/traces/browserRecovery.js.map +1 -1
- package/lib/dist/services/traces/index.d.ts +8 -1
- package/lib/dist/services/traces/index.d.ts.map +1 -1
- package/lib/dist/services/traces/index.js +33 -12
- package/lib/dist/services/traces/index.js.map +1 -1
- package/lib/dist/services/traces/judgeAgentsHints.d.ts +98 -3
- package/lib/dist/services/traces/judgeAgentsHints.d.ts.map +1 -1
- package/lib/dist/services/traces/judgeAgentsHints.js +144 -3
- package/lib/dist/services/traces/judgeAgentsHints.js.map +1 -1
- package/lib/dist/services/traces/spansToTrajectory.js +4 -4
- package/lib/dist/services/traces/spansToTrajectory.js.map +1 -1
- package/lib/dist/services/traces/tracePoller.d.ts.map +1 -1
- package/lib/dist/services/traces/tracePoller.js +17 -20
- package/lib/dist/services/traces/tracePoller.js.map +1 -1
- package/lib/dist/services/traces/trajectoryMerge.d.ts +79 -0
- package/lib/dist/services/traces/trajectoryMerge.d.ts.map +1 -0
- package/lib/dist/services/traces/trajectoryMerge.js +109 -0
- package/lib/dist/services/traces/trajectoryMerge.js.map +1 -0
- package/lib/dist/types/index.d.ts +195 -2
- package/lib/dist/types/index.d.ts.map +1 -1
- package/lib/dist/types/index.js +24 -0
- package/lib/dist/types/index.js.map +1 -1
- package/package.json +6 -5
- package/server/dist/app.js +3398 -926
- package/server/dist/index.js +3401 -929
- package/dist/assets/index-BfxtxmKc.css +0 -1
- package/dist/assets/index-CrjAfDHu.js +0 -243
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Copyright OpenSearch Contributors
|
|
3
|
+
* SPDX-License-Identifier: Apache-2.0
|
|
4
|
+
*/
|
|
5
|
+
const UNCATEGORIZED = 'Uncategorized';
|
|
6
|
+
/**
|
|
7
|
+
* Group rows by test-case category and tally pass/fail/errored/total.
|
|
8
|
+
* Deterministic order: largest category first, ties broken alphabetically.
|
|
9
|
+
*/
|
|
10
|
+
export function computeCategoryBars(rows) {
|
|
11
|
+
const map = new Map();
|
|
12
|
+
for (const row of rows) {
|
|
13
|
+
const category = (row.category || '').trim() || UNCATEGORIZED;
|
|
14
|
+
let bar = map.get(category);
|
|
15
|
+
if (!bar) {
|
|
16
|
+
bar = { category, passed: 0, failed: 0, errored: 0, total: 0 };
|
|
17
|
+
map.set(category, bar);
|
|
18
|
+
}
|
|
19
|
+
bar.total++;
|
|
20
|
+
if (row.status === 'passed')
|
|
21
|
+
bar.passed++;
|
|
22
|
+
else if (row.status === 'errored')
|
|
23
|
+
bar.errored++;
|
|
24
|
+
else if (row.status === 'failed')
|
|
25
|
+
bar.failed++;
|
|
26
|
+
}
|
|
27
|
+
return Array.from(map.values()).sort((a, b) => b.total - a.total || a.category.localeCompare(b.category));
|
|
28
|
+
}
|
|
29
|
+
// ── Failure-theme clustering ───────────────────────────────────────────────
|
|
30
|
+
/**
|
|
31
|
+
* Normalize a judge-reasoning string down to its first sentence,
|
|
32
|
+
* lowercased, punctuation-stripped, whitespace-collapsed. Used both as the
|
|
33
|
+
* clustering input and as the theme's stable `key`.
|
|
34
|
+
*/
|
|
35
|
+
export function normalizeReasoningKey(reasoning) {
|
|
36
|
+
const trimmed = (reasoning || '').trim();
|
|
37
|
+
if (!trimmed)
|
|
38
|
+
return '';
|
|
39
|
+
const sentenceMatch = trimmed.match(/^[^.!?\n]+[.!?]?/);
|
|
40
|
+
const firstSentence = (sentenceMatch ? sentenceMatch[0] : trimmed).toLowerCase();
|
|
41
|
+
return firstSentence.replace(/[^a-z0-9\s]/g, ' ').replace(/\s+/g, ' ').trim();
|
|
42
|
+
}
|
|
43
|
+
/** Returns the literal (non-lowercased) first sentence, trimmed — used as the human-facing sample snippet. */
|
|
44
|
+
function literalFirstSentence(reasoning) {
|
|
45
|
+
const trimmed = (reasoning || '').trim();
|
|
46
|
+
if (!trimmed)
|
|
47
|
+
return '';
|
|
48
|
+
const sentenceMatch = trimmed.match(/^[^.!?\n]+[.!?]?/);
|
|
49
|
+
return (sentenceMatch ? sentenceMatch[0] : trimmed).trim();
|
|
50
|
+
}
|
|
51
|
+
function shingles(normalized, size) {
|
|
52
|
+
const words = normalized.split(' ').filter(Boolean);
|
|
53
|
+
if (words.length === 0)
|
|
54
|
+
return [];
|
|
55
|
+
if (words.length <= size)
|
|
56
|
+
return [words.join(' ')];
|
|
57
|
+
const out = [];
|
|
58
|
+
for (let i = 0; i <= words.length - size; i++) {
|
|
59
|
+
out.push(words.slice(i, i + size).join(' '));
|
|
60
|
+
}
|
|
61
|
+
return out;
|
|
62
|
+
}
|
|
63
|
+
/** Contiguous-word window size used for near-duplicate matching. Paraphrases of the same failure ("agent was unable to retrieve..." vs "agent failed to retrieve...") reliably share at least one 6-word run even when their leading words differ. */
|
|
64
|
+
const DEFAULT_SHINGLE_SIZE = 6;
|
|
65
|
+
/** Reasoning first-sentences shorter than this many words don't carry enough signal to safely dedupe against unrelated failures — left as their own singleton theme rather than risk clustering unrelated short reasons together (e.g. "The agent failed to retrieve the required information."). */
|
|
66
|
+
const MIN_WORDS_FOR_SHINGLING = 3;
|
|
67
|
+
/**
|
|
68
|
+
* Cluster failing test cases into "why they failed" themes using a
|
|
69
|
+
* deterministic, LLM-free heuristic: normalize each case's judge-reasoning
|
|
70
|
+
* first sentence, then union-find cases that share at least one contiguous
|
|
71
|
+
* N-word shingle. This is robust to minor paraphrasing (a judge saying
|
|
72
|
+
* "unable to retrieve" vs "failed to retrieve" the same underlying tool
|
|
73
|
+
* connectivity failure) while still keeping genuinely distinct failure
|
|
74
|
+
* modes (e.g. "missing required facts" vs "MCP server unavailable")
|
|
75
|
+
* separate, because they share no contiguous phrase.
|
|
76
|
+
*
|
|
77
|
+
* Verified against a real production run (418-verify, 64 failing cases):
|
|
78
|
+
* 57 of 64 connectivity-flavored reasonings collapse into ONE dominant
|
|
79
|
+
* theme; the remaining 7 ("Required facts evaluation: ...", a genuinely
|
|
80
|
+
* different failure shape) form a second, correctly separate theme.
|
|
81
|
+
*
|
|
82
|
+
* Output order: largest theme first, ties broken by the theme's
|
|
83
|
+
* lowest-sorting testCaseId (deterministic, no dependency on input order).
|
|
84
|
+
*/
|
|
85
|
+
export function clusterFailureThemes(items, shingleSize = DEFAULT_SHINGLE_SIZE) {
|
|
86
|
+
const n = items.length;
|
|
87
|
+
if (n === 0)
|
|
88
|
+
return [];
|
|
89
|
+
const parent = Array.from({ length: n }, (_, i) => i);
|
|
90
|
+
function find(i) {
|
|
91
|
+
while (parent[i] !== i) {
|
|
92
|
+
parent[i] = parent[parent[i]];
|
|
93
|
+
i = parent[i];
|
|
94
|
+
}
|
|
95
|
+
return i;
|
|
96
|
+
}
|
|
97
|
+
function union(a, b) {
|
|
98
|
+
const ra = find(a);
|
|
99
|
+
const rb = find(b);
|
|
100
|
+
if (ra !== rb)
|
|
101
|
+
parent[Math.max(ra, rb)] = Math.min(ra, rb);
|
|
102
|
+
}
|
|
103
|
+
const normalized = items.map(it => normalizeReasoningKey(it.reasoning));
|
|
104
|
+
const shingleBuckets = new Map(); // shingle -> first index seen
|
|
105
|
+
normalized.forEach((norm, idx) => {
|
|
106
|
+
const words = norm.split(' ').filter(Boolean);
|
|
107
|
+
if (words.length < MIN_WORDS_FOR_SHINGLING)
|
|
108
|
+
return;
|
|
109
|
+
for (const sh of shingles(norm, shingleSize)) {
|
|
110
|
+
const firstIdx = shingleBuckets.get(sh);
|
|
111
|
+
if (firstIdx === undefined)
|
|
112
|
+
shingleBuckets.set(sh, idx);
|
|
113
|
+
else
|
|
114
|
+
union(firstIdx, idx);
|
|
115
|
+
}
|
|
116
|
+
});
|
|
117
|
+
const groups = new Map();
|
|
118
|
+
for (let i = 0; i < n; i++) {
|
|
119
|
+
const root = find(i);
|
|
120
|
+
const arr = groups.get(root);
|
|
121
|
+
if (arr)
|
|
122
|
+
arr.push(i);
|
|
123
|
+
else
|
|
124
|
+
groups.set(root, [i]);
|
|
125
|
+
}
|
|
126
|
+
const themes = [];
|
|
127
|
+
for (const idxs of groups.values()) {
|
|
128
|
+
// Representative snippet: the most frequent literal first-sentence
|
|
129
|
+
// phrasing within the cluster (ties broken by first occurrence) - so a
|
|
130
|
+
// theme with 4 identical sentences + a handful of near-duplicate
|
|
131
|
+
// paraphrases surfaces the exact sentence, not an arbitrary member.
|
|
132
|
+
const counts = new Map();
|
|
133
|
+
for (const idx of idxs) {
|
|
134
|
+
const key = normalized[idx];
|
|
135
|
+
const snippet = literalFirstSentence(items[idx].reasoning);
|
|
136
|
+
const existing = counts.get(key);
|
|
137
|
+
if (existing)
|
|
138
|
+
existing.count++;
|
|
139
|
+
else
|
|
140
|
+
counts.set(key, { count: 1, firstIdx: idx, snippet });
|
|
141
|
+
}
|
|
142
|
+
let best = null;
|
|
143
|
+
for (const v of counts.values()) {
|
|
144
|
+
if (!best || v.count > best.count || (v.count === best.count && v.firstIdx < best.firstIdx))
|
|
145
|
+
best = v;
|
|
146
|
+
}
|
|
147
|
+
const testCaseIds = idxs.map(idx => items[idx].testCaseId);
|
|
148
|
+
themes.push({
|
|
149
|
+
key: normalized[idxs[0]] || `theme-${idxs[0]}`,
|
|
150
|
+
count: idxs.length,
|
|
151
|
+
sampleSnippet: best?.snippet || '',
|
|
152
|
+
testCaseIds,
|
|
153
|
+
});
|
|
154
|
+
}
|
|
155
|
+
themes.sort((a, b) => {
|
|
156
|
+
if (b.count !== a.count)
|
|
157
|
+
return b.count - a.count;
|
|
158
|
+
const aMin = [...a.testCaseIds].sort()[0] || '';
|
|
159
|
+
const bMin = [...b.testCaseIds].sort()[0] || '';
|
|
160
|
+
return aMin < bMin ? -1 : aMin > bMin ? 1 : 0;
|
|
161
|
+
});
|
|
162
|
+
return themes;
|
|
163
|
+
}
|
|
164
|
+
/**
|
|
165
|
+
* "Based on N of M failing cases" note shown under the theme list when the
|
|
166
|
+
* reasoning fetch was capped (RunInsightsPane caps at the first 100 failing
|
|
167
|
+
* cases). Returns null when nothing was capped (fetchedCount >= totalCount).
|
|
168
|
+
*/
|
|
169
|
+
export function formatCappedNote(fetchedCount, totalFailingCount) {
|
|
170
|
+
if (totalFailingCount <= fetchedCount)
|
|
171
|
+
return null;
|
|
172
|
+
return `Based on ${fetchedCount} of ${totalFailingCount} failing cases`;
|
|
173
|
+
}
|
|
174
|
+
/**
|
|
175
|
+
* Top-N cases by a numeric value (duration, cost, ...), descending.
|
|
176
|
+
* Cases with a null/undefined/NaN value are excluded. Deterministic tie
|
|
177
|
+
* break: lower testCaseId first.
|
|
178
|
+
*/
|
|
179
|
+
export function pickTopN(cases, n) {
|
|
180
|
+
return cases
|
|
181
|
+
.filter((c) => typeof c.value === 'number' && Number.isFinite(c.value))
|
|
182
|
+
.sort((a, b) => (b.value !== a.value ? b.value - a.value : a.testCaseId.localeCompare(b.testCaseId)))
|
|
183
|
+
.slice(0, n);
|
|
184
|
+
}
|
|
185
|
+
//# sourceMappingURL=runInsights.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"runInsights.js","sourceRoot":"","sources":["../../runInsights.ts"],"names":[],"mappings":"AAAA;;;GAGG;AA+BH,MAAM,aAAa,GAAG,eAAe,CAAC;AAEtC;;;GAGG;AACH,MAAM,UAAU,mBAAmB,CAAC,IAAyB;IAC3D,MAAM,GAAG,GAAG,IAAI,GAAG,EAAuB,CAAC;IAC3C,KAAK,MAAM,GAAG,IAAI,IAAI,EAAE,CAAC;QACvB,MAAM,QAAQ,GAAG,CAAC,GAAG,CAAC,QAAQ,IAAI,EAAE,CAAC,CAAC,IAAI,EAAE,IAAI,aAAa,CAAC;QAC9D,IAAI,GAAG,GAAG,GAAG,CAAC,GAAG,CAAC,QAAQ,CAAC,CAAC;QAC5B,IAAI,CAAC,GAAG,EAAE,CAAC;YACT,GAAG,GAAG,EAAE,QAAQ,EAAE,MAAM,EAAE,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,OAAO,EAAE,CAAC,EAAE,KAAK,EAAE,CAAC,EAAE,CAAC;YAC/D,GAAG,CAAC,GAAG,CAAC,QAAQ,EAAE,GAAG,CAAC,CAAC;QACzB,CAAC;QACD,GAAG,CAAC,KAAK,EAAE,CAAC;QACZ,IAAI,GAAG,CAAC,MAAM,KAAK,QAAQ;YAAE,GAAG,CAAC,MAAM,EAAE,CAAC;aACrC,IAAI,GAAG,CAAC,MAAM,KAAK,SAAS;YAAE,GAAG,CAAC,OAAO,EAAE,CAAC;aAC5C,IAAI,GAAG,CAAC,MAAM,KAAK,QAAQ;YAAE,GAAG,CAAC,MAAM,EAAE,CAAC;IACjD,CAAC;IACD,OAAO,KAAK,CAAC,IAAI,CAAC,GAAG,CAAC,MAAM,EAAE,CAAC,CAAC,IAAI,CAClC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,KAAK,GAAG,CAAC,CAAC,KAAK,IAAI,CAAC,CAAC,QAAQ,CAAC,aAAa,CAAC,CAAC,CAAC,QAAQ,CAAC,CACpE,CAAC;AACJ,CAAC;AAED,8EAA8E;AAE9E;;;;GAIG;AACH,MAAM,UAAU,qBAAqB,CAAC,SAAiB;IACrD,MAAM,OAAO,GAAG,CAAC,SAAS,IAAI,EAAE,CAAC,CAAC,IAAI,EAAE,CAAC;IACzC,IAAI,CAAC,OAAO;QAAE,OAAO,EAAE,CAAC;IACxB,MAAM,aAAa,GAAG,OAAO,CAAC,KAAK,CAAC,kBAAkB,CAAC,CAAC;IACxD,MAAM,aAAa,GAAG,CAAC,aAAa,CAAC,CAAC,CAAC,aAAa,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,WAAW,EAAE,CAAC;IACjF,OAAO,aAAa,CAAC,OAAO,CAAC,cAAc,EAAE,GAAG,CAAC,CAAC,OAAO,CAAC,MAAM,EAAE,GAAG,CAAC,CAAC,IAAI,EAAE,CAAC;AAChF,CAAC;AAED,8GAA8G;AAC9G,SAAS,oBAAoB,CAAC,SAAiB;IAC7C,MAAM,OAAO,GAAG,CAAC,SAAS,IAAI,EAAE,CAAC,CAAC,IAAI,EAAE,CAAC;IACzC,IAAI,CAAC,OAAO;QAAE,OAAO,EAAE,CAAC;IACxB,MAAM,aAAa,GAAG,OAAO,CAAC,KAAK,CAAC,kBAAkB,CAAC,CAAC;IACxD,OAAO,CAAC,aAAa,CAAC,CAAC,CAAC,aAAa,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,IAAI,EAAE,CAAC;AAC7D,CAAC;AAED,SAAS,QAAQ,CAAC,UAAkB,EAAE,IAAY;IAChD,MAAM,KAAK,GAAG,UAAU,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,MAAM,CAAC,OAAO,CAAC,CAAC;IACpD,IAAI,KAAK,CAAC,MAAM,KAAK,CAAC;QAAE,OAAO,EAAE,CAAC;IAClC,IAAI,KAAK,CAAC,MAAM,IAAI,IAAI;QAAE,OAAO,CAAC,KAAK,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,CAAC;IACnD,MAAM,GAAG,GAAa,EAAE,CAAC;IACzB,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,IAAI,KAAK,CAAC,MAAM,GAAG,IAAI,EAAE,CAAC,EAAE,EAAE,CAAC;QAC9C,GAAG,CAAC,IAAI,CAAC,KAAK,CAAC,KAAK,CAAC,CAAC,EAAE,CAAC,GAAG,IAAI,CAAC,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,CAAC;IAC/C,CAAC;IACD,OAAO,GAAG,CAAC;AACb,CAAC;AAgBD,sPAAsP;AACtP,MAAM,oBAAoB,GAAG,CAAC,CAAC;AAC/B,qSAAqS;AACrS,MAAM,uBAAuB,GAAG,CAAC,CAAC;AAElC;;;;;;;;;;;;;;;;;GAiBG;AACH,MAAM,UAAU,oBAAoB,CAClC,KAA0B,EAC1B,cAAsB,oBAAoB;IAE1C,MAAM,CAAC,GAAG,KAAK,CAAC,MAAM,CAAC;IACvB,IAAI,CAAC,KAAK,CAAC;QAAE,OAAO,EAAE,CAAC;IAEvB,MAAM,MAAM,GAAG,KAAK,CAAC,IAAI,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,CAAC;IACtD,SAAS,IAAI,CAAC,CAAS;QACrB,OAAO,MAAM,CAAC,CAAC,CAAC,KAAK,CAAC,EAAE,CAAC;YACvB,MAAM,CAAC,CAAC,CAAC,GAAG,MAAM,CAAC,MAAM,CAAC,CAAC,CAAC,CAAC,CAAC;YAC9B,CAAC,GAAG,MAAM,CAAC,CAAC,CAAC,CAAC;QAChB,CAAC;QACD,OAAO,CAAC,CAAC;IACX,CAAC;IACD,SAAS,KAAK,CAAC,CAAS,EAAE,CAAS;QACjC,MAAM,EAAE,GAAG,IAAI,CAAC,CAAC,CAAC,CAAC;QACnB,MAAM,EAAE,GAAG,IAAI,CAAC,CAAC,CAAC,CAAC;QACnB,IAAI,EAAE,KAAK,EAAE;YAAE,MAAM,CAAC,IAAI,CAAC,GAAG,CAAC,EAAE,EAAE,EAAE,CAAC,CAAC,GAAG,IAAI,CAAC,GAAG,CAAC,EAAE,EAAE,EAAE,CAAC,CAAC;IAC7D,CAAC;IAED,MAAM,UAAU,GAAG,KAAK,CAAC,GAAG,CAAC,EAAE,CAAC,EAAE,CAAC,qBAAqB,CAAC,EAAE,CAAC,SAAS,CAAC,CAAC,CAAC;IACxE,MAAM,cAAc,GAAG,IAAI,GAAG,EAAkB,CAAC,CAAC,8BAA8B;IAChF,UAAU,CAAC,OAAO,CAAC,CAAC,IAAI,EAAE,GAAG,EAAE,EAAE;QAC/B,MAAM,KAAK,GAAG,IAAI,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,MAAM,CAAC,OAAO,CAAC,CAAC;QAC9C,IAAI,KAAK,CAAC,MAAM,GAAG,uBAAuB;YAAE,OAAO;QACnD,KAAK,MAAM,EAAE,IAAI,QAAQ,CAAC,IAAI,EAAE,WAAW,CAAC,EAAE,CAAC;YAC7C,MAAM,QAAQ,GAAG,cAAc,CAAC,GAAG,CAAC,EAAE,CAAC,CAAC;YACxC,IAAI,QAAQ,KAAK,SAAS;gBAAE,cAAc,CAAC,GAAG,CAAC,EAAE,EAAE,GAAG,CAAC,CAAC;;gBACnD,KAAK,CAAC,QAAQ,EAAE,GAAG,CAAC,CAAC;QAC5B,CAAC;IACH,CAAC,CAAC,CAAC;IAEH,MAAM,MAAM,GAAG,IAAI,GAAG,EAAoB,CAAC;IAC3C,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC;QAC3B,MAAM,IAAI,GAAG,IAAI,CAAC,CAAC,CAAC,CAAC;QACrB,MAAM,GAAG,GAAG,MAAM,CAAC,GAAG,CAAC,IAAI,CAAC,CAAC;QAC7B,IAAI,GAAG;YAAE,GAAG,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC;;YAChB,MAAM,CAAC,GAAG,CAAC,IAAI,EAAE,CAAC,CAAC,CAAC,CAAC,CAAC;IAC7B,CAAC;IAED,MAAM,MAAM,GAAmB,EAAE,CAAC;IAClC,KAAK,MAAM,IAAI,IAAI,MAAM,CAAC,MAAM,EAAE,EAAE,CAAC;QACnC,mEAAmE;QACnE,uEAAuE;QACvE,iEAAiE;QACjE,oEAAoE;QACpE,MAAM,MAAM,GAAG,IAAI,GAAG,EAAgE,CAAC;QACvF,KAAK,MAAM,GAAG,IAAI,IAAI,EAAE,CAAC;YACvB,MAAM,GAAG,GAAG,UAAU,CAAC,GAAG,CAAC,CAAC;YAC5B,MAAM,OAAO,GAAG,oBAAoB,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,SAAS,CAAC,CAAC;YAC3D,MAAM,QAAQ,GAAG,MAAM,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC;YACjC,IAAI,QAAQ;gBAAE,QAAQ,CAAC,KAAK,EAAE,CAAC;;gBAC1B,MAAM,CAAC,GAAG,CAAC,GAAG,EAAE,EAAE,KAAK,EAAE,CAAC,EAAE,QAAQ,EAAE,GAAG,EAAE,OAAO,EAAE,CAAC,CAAC;QAC7D,CAAC;QACD,IAAI,IAAI,GAAgE,IAAI,CAAC;QAC7E,KAAK,MAAM,CAAC,IAAI,MAAM,CAAC,MAAM,EAAE,EAAE,CAAC;YAChC,IAAI,CAAC,IAAI,IAAI,CAAC,CAAC,KAAK,GAAG,IAAI,CAAC,KAAK,IAAI,CAAC,CAAC,CAAC,KAAK,KAAK,IAAI,CAAC,KAAK,IAAI,CAAC,CAAC,QAAQ,GAAG,IAAI,CAAC,QAAQ,CAAC;gBAAE,IAAI,GAAG,CAAC,CAAC;QACxG,CAAC;QACD,MAAM,WAAW,GAAG,IAAI,CAAC,GAAG,CAAC,GAAG,CAAC,EAAE,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,UAAU,CAAC,CAAC;QAC3D,MAAM,CAAC,IAAI,CAAC;YACV,GAAG,EAAE,UAAU,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC,IAAI,SAAS,IAAI,CAAC,CAAC,CAAC,EAAE;YAC9C,KAAK,EAAE,IAAI,CAAC,MAAM;YAClB,aAAa,EAAE,IAAI,EAAE,OAAO,IAAI,EAAE;YAClC,WAAW;SACZ,CAAC,CAAC;IACL,CAAC;IAED,MAAM,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE;QACnB,IAAI,CAAC,CAAC,KAAK,KAAK,CAAC,CAAC,KAAK;YAAE,OAAO,CAAC,CAAC,KAAK,GAAG,CAAC,CAAC,KAAK,CAAC;QAClD,MAAM,IAAI,GAAG,CAAC,GAAG,CAAC,CAAC,WAAW,CAAC,CAAC,IAAI,EAAE,CAAC,CAAC,CAAC,IAAI,EAAE,CAAC;QAChD,MAAM,IAAI,GAAG,CAAC,GAAG,CAAC,CAAC,WAAW,CAAC,CAAC,IAAI,EAAE,CAAC,CAAC,CAAC,IAAI,EAAE,CAAC;QAChD,OAAO,IAAI,GAAG,IAAI,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,IAAI,GAAG,IAAI,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;IAChD,CAAC,CAAC,CAAC;IACH,OAAO,MAAM,CAAC;AAChB,CAAC;AAED;;;;GAIG;AACH,MAAM,UAAU,gBAAgB,CAAC,YAAoB,EAAE,iBAAyB;IAC9E,IAAI,iBAAiB,IAAI,YAAY;QAAE,OAAO,IAAI,CAAC;IACnD,OAAO,YAAY,YAAY,OAAO,iBAAiB,gBAAgB,CAAC;AAC1E,CAAC;AASD;;;;GAIG;AACH,MAAM,UAAU,QAAQ,CAAC,KAAiE,EAAE,CAAS;IACnG,OAAO,KAAK;SACT,MAAM,CAAC,CAAC,CAAC,EAAmB,EAAE,CAAC,OAAO,CAAC,CAAC,KAAK,KAAK,QAAQ,IAAI,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC;SACvF,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,CAAC,KAAK,KAAK,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC,KAAK,GAAG,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC,UAAU,CAAC,aAAa,CAAC,CAAC,CAAC,UAAU,CAAC,CAAC,CAAC;SACpG,KAAK,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC;AACjB,CAAC"}
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Validation for renaming an evaluation run (PATCH /api/storage/evaluation-runs/:id
|
|
3
|
+
* `{ name }`). This is the server route's authoritative check.
|
|
4
|
+
*
|
|
5
|
+
* Rename is intentionally narrow: it only ever touches `name`. No version bump,
|
|
6
|
+
* no stats recompute — this file has no opinion on anything but the string.
|
|
7
|
+
*/
|
|
8
|
+
/** Renamed evaluation runs are capped at this length (arbitrary but generous —
|
|
9
|
+
* long enough for any real run name, short enough to keep list/table UIs sane). */
|
|
10
|
+
export declare const MAX_RUN_NAME_LENGTH = 200;
|
|
11
|
+
export type RunNameValidation = {
|
|
12
|
+
ok: true;
|
|
13
|
+
value: string;
|
|
14
|
+
} | {
|
|
15
|
+
ok: false;
|
|
16
|
+
error: string;
|
|
17
|
+
};
|
|
18
|
+
/**
|
|
19
|
+
* Validate + normalize a candidate evaluation-run name.
|
|
20
|
+
*
|
|
21
|
+
* - Must be a string.
|
|
22
|
+
* - Trimmed value must be non-empty.
|
|
23
|
+
* - Trimmed value must be at most {@link MAX_RUN_NAME_LENGTH} characters.
|
|
24
|
+
*
|
|
25
|
+
* Returns the trimmed value on success so callers persist a normalized name
|
|
26
|
+
* (no leading/trailing whitespace) rather than the raw input.
|
|
27
|
+
*/
|
|
28
|
+
export declare function validateRunNameUpdate(name: unknown): RunNameValidation;
|
|
29
|
+
//# sourceMappingURL=runName.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"runName.d.ts","sourceRoot":"","sources":["../../runName.ts"],"names":[],"mappings":"AAKA;;;;;;GAMG;AAEH;oFACoF;AACpF,eAAO,MAAM,mBAAmB,MAAM,CAAC;AAEvC,MAAM,MAAM,iBAAiB,GACzB;IAAE,EAAE,EAAE,IAAI,CAAC;IAAC,KAAK,EAAE,MAAM,CAAA;CAAE,GAC3B;IAAE,EAAE,EAAE,KAAK,CAAC;IAAC,KAAK,EAAE,MAAM,CAAA;CAAE,CAAC;AAEjC;;;;;;;;;GASG;AACH,wBAAgB,qBAAqB,CAAC,IAAI,EAAE,OAAO,GAAG,iBAAiB,CAYtE"}
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Copyright OpenSearch Contributors
|
|
3
|
+
* SPDX-License-Identifier: Apache-2.0
|
|
4
|
+
*/
|
|
5
|
+
/**
|
|
6
|
+
* Validation for renaming an evaluation run (PATCH /api/storage/evaluation-runs/:id
|
|
7
|
+
* `{ name }`). This is the server route's authoritative check.
|
|
8
|
+
*
|
|
9
|
+
* Rename is intentionally narrow: it only ever touches `name`. No version bump,
|
|
10
|
+
* no stats recompute — this file has no opinion on anything but the string.
|
|
11
|
+
*/
|
|
12
|
+
/** Renamed evaluation runs are capped at this length (arbitrary but generous —
|
|
13
|
+
* long enough for any real run name, short enough to keep list/table UIs sane). */
|
|
14
|
+
export const MAX_RUN_NAME_LENGTH = 200;
|
|
15
|
+
/**
|
|
16
|
+
* Validate + normalize a candidate evaluation-run name.
|
|
17
|
+
*
|
|
18
|
+
* - Must be a string.
|
|
19
|
+
* - Trimmed value must be non-empty.
|
|
20
|
+
* - Trimmed value must be at most {@link MAX_RUN_NAME_LENGTH} characters.
|
|
21
|
+
*
|
|
22
|
+
* Returns the trimmed value on success so callers persist a normalized name
|
|
23
|
+
* (no leading/trailing whitespace) rather than the raw input.
|
|
24
|
+
*/
|
|
25
|
+
export function validateRunNameUpdate(name) {
|
|
26
|
+
if (typeof name !== 'string') {
|
|
27
|
+
return { ok: false, error: 'name must be a string' };
|
|
28
|
+
}
|
|
29
|
+
const trimmed = name.trim();
|
|
30
|
+
if (!trimmed) {
|
|
31
|
+
return { ok: false, error: 'name must not be empty' };
|
|
32
|
+
}
|
|
33
|
+
if (trimmed.length > MAX_RUN_NAME_LENGTH) {
|
|
34
|
+
return { ok: false, error: `name must be ${MAX_RUN_NAME_LENGTH} characters or fewer` };
|
|
35
|
+
}
|
|
36
|
+
return { ok: true, value: trimmed };
|
|
37
|
+
}
|
|
38
|
+
//# sourceMappingURL=runName.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"runName.js","sourceRoot":"","sources":["../../runName.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAEH;;;;;;GAMG;AAEH;oFACoF;AACpF,MAAM,CAAC,MAAM,mBAAmB,GAAG,GAAG,CAAC;AAMvC;;;;;;;;;GASG;AACH,MAAM,UAAU,qBAAqB,CAAC,IAAa;IACjD,IAAI,OAAO,IAAI,KAAK,QAAQ,EAAE,CAAC;QAC7B,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,uBAAuB,EAAE,CAAC;IACvD,CAAC;IACD,MAAM,OAAO,GAAG,IAAI,CAAC,IAAI,EAAE,CAAC;IAC5B,IAAI,CAAC,OAAO,EAAE,CAAC;QACb,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,wBAAwB,EAAE,CAAC;IACxD,CAAC;IACD,IAAI,OAAO,CAAC,MAAM,GAAG,mBAAmB,EAAE,CAAC;QACzC,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,gBAAgB,mBAAmB,sBAAsB,EAAE,CAAC;IACzF,CAAC;IACD,OAAO,EAAE,EAAE,EAAE,IAAI,EAAE,KAAK,EAAE,OAAO,EAAE,CAAC;AACtC,CAAC"}
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Canonical run-report path for a run surfaced on the compare page.
|
|
3
|
+
*
|
|
4
|
+
* Benchmark runs (embedded in a benchmark's `runs[]`) resolve at
|
|
5
|
+
* `/evaluations/benchmarks/:benchmarkId/runs/:runId`; everything else —
|
|
6
|
+
* ad-hoc / SDK eval-runs, or runs whose `benchmarkId` is merely a label — goes
|
|
7
|
+
* to the bare `/evaluations/runs/:runId` route (the benchmark route would
|
|
8
|
+
* 404/redirect for those, and the bare route 404s for benchmark run ids).
|
|
9
|
+
*
|
|
10
|
+
* Shared by the scoreboard's run-name link, its "Open run" icon, and the
|
|
11
|
+
* per-case table's run headers so the three can never disagree. Callers pass
|
|
12
|
+
* the benchmarkId from `ComparisonPage`'s `runBenchmarkIdById` map (which is
|
|
13
|
+
* `undefined` unless the run is a real benchmark member).
|
|
14
|
+
*/
|
|
15
|
+
export declare const runReportPath: (runId: string, benchmarkId?: string) => string;
|
|
16
|
+
//# sourceMappingURL=runReportPath.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"runReportPath.d.ts","sourceRoot":"","sources":["../../runReportPath.ts"],"names":[],"mappings":"AAKA;;;;;;;;;;;;;GAaG;AACH,eAAO,MAAM,aAAa,GAAI,OAAO,MAAM,EAAE,cAAc,MAAM,KAAG,MAGd,CAAC"}
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Copyright OpenSearch Contributors
|
|
3
|
+
* SPDX-License-Identifier: Apache-2.0
|
|
4
|
+
*/
|
|
5
|
+
/**
|
|
6
|
+
* Canonical run-report path for a run surfaced on the compare page.
|
|
7
|
+
*
|
|
8
|
+
* Benchmark runs (embedded in a benchmark's `runs[]`) resolve at
|
|
9
|
+
* `/evaluations/benchmarks/:benchmarkId/runs/:runId`; everything else —
|
|
10
|
+
* ad-hoc / SDK eval-runs, or runs whose `benchmarkId` is merely a label — goes
|
|
11
|
+
* to the bare `/evaluations/runs/:runId` route (the benchmark route would
|
|
12
|
+
* 404/redirect for those, and the bare route 404s for benchmark run ids).
|
|
13
|
+
*
|
|
14
|
+
* Shared by the scoreboard's run-name link, its "Open run" icon, and the
|
|
15
|
+
* per-case table's run headers so the three can never disagree. Callers pass
|
|
16
|
+
* the benchmarkId from `ComparisonPage`'s `runBenchmarkIdById` map (which is
|
|
17
|
+
* `undefined` unless the run is a real benchmark member).
|
|
18
|
+
*/
|
|
19
|
+
export const runReportPath = (runId, benchmarkId) => benchmarkId
|
|
20
|
+
? `/evaluations/benchmarks/${encodeURIComponent(benchmarkId)}/runs/${encodeURIComponent(runId)}`
|
|
21
|
+
: `/evaluations/runs/${encodeURIComponent(runId)}`;
|
|
22
|
+
//# sourceMappingURL=runReportPath.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"runReportPath.js","sourceRoot":"","sources":["../../runReportPath.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAEH;;;;;;;;;;;;;GAaG;AACH,MAAM,CAAC,MAAM,aAAa,GAAG,CAAC,KAAa,EAAE,WAAoB,EAAU,EAAE,CAC3E,WAAW;IACT,CAAC,CAAC,2BAA2B,kBAAkB,CAAC,WAAW,CAAC,SAAS,kBAAkB,CAAC,KAAK,CAAC,EAAE;IAChG,CAAC,CAAC,qBAAqB,kBAAkB,CAAC,KAAK,CAAC,EAAE,CAAC"}
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Comparators for sorting evaluation-run rows and run-groups newest-first.
|
|
3
|
+
*
|
|
4
|
+
* Extracted from EvalRunsPage so the "grouped view sorts groups by most
|
|
5
|
+
* recent run, not alphabetically by benchmark name" rule is a plain,
|
|
6
|
+
* unit-testable function instead of logic buried inside a `useMemo`.
|
|
7
|
+
*/
|
|
8
|
+
export interface HasCreatedAt {
|
|
9
|
+
createdAt: string;
|
|
10
|
+
}
|
|
11
|
+
/**
|
|
12
|
+
* Most recent `createdAt` timestamp (ms since epoch) across a list of runs.
|
|
13
|
+
* Empty/all-invalid input returns 0 so a group with no valid timestamps
|
|
14
|
+
* sorts to the oldest position rather than throwing or sorting first.
|
|
15
|
+
*/
|
|
16
|
+
export declare function mostRecentTime(items: HasCreatedAt[]): number;
|
|
17
|
+
/**
|
|
18
|
+
* Sort groups (e.g. "runs grouped by benchmark") by each group's most
|
|
19
|
+
* recent run — descending, so the group with the latest activity is first.
|
|
20
|
+
* Ties (including empty groups, which compare equal at 0) preserve the
|
|
21
|
+
* original relative order (stable sort).
|
|
22
|
+
*
|
|
23
|
+
* @param groups - the groups to sort
|
|
24
|
+
* @param getItems - extracts the timestamped items (e.g. runs) from a group
|
|
25
|
+
*/
|
|
26
|
+
export declare function sortGroupsByRecency<T>(groups: T[], getItems: (group: T) => HasCreatedAt[]): T[];
|
|
27
|
+
//# sourceMappingURL=runSort.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"runSort.d.ts","sourceRoot":"","sources":["../../runSort.ts"],"names":[],"mappings":"AAKA;;;;;;GAMG;AAEH,MAAM,WAAW,YAAY;IAC3B,SAAS,EAAE,MAAM,CAAC;CACnB;AAED;;;;GAIG;AACH,wBAAgB,cAAc,CAAC,KAAK,EAAE,YAAY,EAAE,GAAG,MAAM,CAO5D;AAED;;;;;;;;GAQG;AACH,wBAAgB,mBAAmB,CAAC,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,EAAE,QAAQ,EAAE,CAAC,KAAK,EAAE,CAAC,KAAK,YAAY,EAAE,GAAG,CAAC,EAAE,CAE/F"}
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Copyright OpenSearch Contributors
|
|
3
|
+
* SPDX-License-Identifier: Apache-2.0
|
|
4
|
+
*/
|
|
5
|
+
/**
|
|
6
|
+
* Most recent `createdAt` timestamp (ms since epoch) across a list of runs.
|
|
7
|
+
* Empty/all-invalid input returns 0 so a group with no valid timestamps
|
|
8
|
+
* sorts to the oldest position rather than throwing or sorting first.
|
|
9
|
+
*/
|
|
10
|
+
export function mostRecentTime(items) {
|
|
11
|
+
let max = 0;
|
|
12
|
+
for (const item of items) {
|
|
13
|
+
const t = new Date(item.createdAt).getTime();
|
|
14
|
+
if (Number.isFinite(t) && t > max)
|
|
15
|
+
max = t;
|
|
16
|
+
}
|
|
17
|
+
return max;
|
|
18
|
+
}
|
|
19
|
+
/**
|
|
20
|
+
* Sort groups (e.g. "runs grouped by benchmark") by each group's most
|
|
21
|
+
* recent run — descending, so the group with the latest activity is first.
|
|
22
|
+
* Ties (including empty groups, which compare equal at 0) preserve the
|
|
23
|
+
* original relative order (stable sort).
|
|
24
|
+
*
|
|
25
|
+
* @param groups - the groups to sort
|
|
26
|
+
* @param getItems - extracts the timestamped items (e.g. runs) from a group
|
|
27
|
+
*/
|
|
28
|
+
export function sortGroupsByRecency(groups, getItems) {
|
|
29
|
+
return [...groups].sort((a, b) => mostRecentTime(getItems(b)) - mostRecentTime(getItems(a)));
|
|
30
|
+
}
|
|
31
|
+
//# sourceMappingURL=runSort.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"runSort.js","sourceRoot":"","sources":["../../runSort.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAcH;;;;GAIG;AACH,MAAM,UAAU,cAAc,CAAC,KAAqB;IAClD,IAAI,GAAG,GAAG,CAAC,CAAC;IACZ,KAAK,MAAM,IAAI,IAAI,KAAK,EAAE,CAAC;QACzB,MAAM,CAAC,GAAG,IAAI,IAAI,CAAC,IAAI,CAAC,SAAS,CAAC,CAAC,OAAO,EAAE,CAAC;QAC7C,IAAI,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAC,IAAI,CAAC,GAAG,GAAG;YAAE,GAAG,GAAG,CAAC,CAAC;IAC7C,CAAC;IACD,OAAO,GAAG,CAAC;AACb,CAAC;AAED;;;;;;;;GAQG;AACH,MAAM,UAAU,mBAAmB,CAAI,MAAW,EAAE,QAAsC;IACxF,OAAO,CAAC,GAAG,MAAM,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,cAAc,CAAC,QAAQ,CAAC,CAAC,CAAC,CAAC,GAAG,cAAc,CAAC,QAAQ,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;AAC/F,CAAC"}
|
|
@@ -12,7 +12,7 @@
|
|
|
12
12
|
* For performance optimization, stats are denormalized onto BenchmarkRun.stats
|
|
13
13
|
* and updated incrementally as test cases complete.
|
|
14
14
|
*/
|
|
15
|
-
import type { BenchmarkRun, EvaluationReport, RunStats as RunStatsType } from '../types/index.js';
|
|
15
|
+
import type { BenchmarkRun, EvaluationReport, RunStats as RunStatsType, BenchmarkRunStatus } from '../types/index.js';
|
|
16
16
|
/**
|
|
17
17
|
* Statistics for a benchmark run
|
|
18
18
|
*/
|
|
@@ -30,15 +30,27 @@ export interface RunStats {
|
|
|
30
30
|
* pass rates. Issue #242.
|
|
31
31
|
*/
|
|
32
32
|
errored: number;
|
|
33
|
+
/**
|
|
34
|
+
* Number of planned test cases that never executed (or never finished)
|
|
35
|
+
* because the run reached a TERMINAL status first — cancelled mid-way, or
|
|
36
|
+
* the executor crashed. Neither a pass, a fail, nor "pending": nothing is
|
|
37
|
+
* ever going to happen to them. Excluded from the pass-rate denominator
|
|
38
|
+
* and rendered as "n not run" (never with an in-progress spinner).
|
|
39
|
+
*/
|
|
40
|
+
notRun: number;
|
|
33
41
|
/** Total number of test cases in the run */
|
|
34
42
|
total: number;
|
|
35
43
|
/**
|
|
36
|
-
* Pass rate as a percentage (0-100). Computed over `total - errored
|
|
37
|
-
* (the *evaluable* set), not over `total`, so a non-retryable
|
|
38
|
-
* failure can't masquerade as the agent
|
|
44
|
+
* Pass rate as a percentage (0-100). Computed over `total - errored -
|
|
45
|
+
* notRun` (the *evaluable* set), not over `total`, so a non-retryable
|
|
46
|
+
* judge failure — or a cancellation — can't masquerade as the agent
|
|
47
|
+
* scoring 0%.
|
|
39
48
|
*/
|
|
40
49
|
passRate: number;
|
|
41
50
|
}
|
|
51
|
+
/** Run statuses after which no further per-case progress can happen. */
|
|
52
|
+
export declare const TERMINAL_RUN_STATUSES: ReadonlySet<BenchmarkRunStatus>;
|
|
53
|
+
export declare function isTerminalRunStatus(status: BenchmarkRunStatus | undefined): boolean;
|
|
42
54
|
/**
|
|
43
55
|
* Bucket a run's per-test-case results into passed/failed/errored/pending using
|
|
44
56
|
* ONLY the persisted result fields (status + passFailStatus) — no reports
|
|
@@ -55,11 +67,49 @@ export interface RunStats {
|
|
|
55
67
|
* the evaluator couldn't produce one (judge validation error, trace timeout).
|
|
56
68
|
* Excluded from passed/failed — exactly as calculateRunStats does via the
|
|
57
69
|
* report's metricsStatus.
|
|
70
|
+
*
|
|
71
|
+
* `plannedTotal` (bug: runs list showed no in-flight indication, 2026-09-01):
|
|
72
|
+
* for a genuinely in-progress run, `results` only gets an entry once a test
|
|
73
|
+
* case has actually STARTED — a run 9 cases into a planned 62 has a
|
|
74
|
+
* `results` map of length 9, so `total` here would report 9, not 62. Pass
|
|
75
|
+
* the run's snapshotted test-case count (`testCaseSnapshots.length`) as
|
|
76
|
+
* `plannedTotal` and the shortfall (planned - observed) is folded into
|
|
77
|
+
* `pending` so the invariant `total === passed+failed+errored+pending+notRun`
|
|
78
|
+
* still holds and callers see the true "53 more to go" total instead of a
|
|
79
|
+
* misleadingly small, already-100%-accounted-for total that looks finished.
|
|
80
|
+
*
|
|
81
|
+
* `runStatus` (bug: cancelled/failed runs rendered a phantom "n pending ⟳"
|
|
82
|
+
* forever, 2026-09-04): the shortfall above is only *pending* while the run
|
|
83
|
+
* can still make progress. Once the run is TERMINAL (cancelled / failed /
|
|
84
|
+
* completed) nothing will ever start those cases, so they are bucketed as
|
|
85
|
+
* `notRun` instead — a cancelled 34/62 run reads "34 executed · 28 not run",
|
|
86
|
+
* not "28 pending" with a spinner. The same applies to per-case entries a
|
|
87
|
+
* terminal run left behind in `pending`/`running` (an executor that died
|
|
88
|
+
* mid-case) and to entries explicitly marked `status: 'cancelled'` (written
|
|
89
|
+
* for never-started cases when a run is cancelled — see
|
|
90
|
+
* `EvaluationRun.results` in types/index.ts). Without `runStatus` the legacy
|
|
91
|
+
* behaviour (shortfall = pending) is preserved.
|
|
58
92
|
*/
|
|
59
93
|
export declare function bucketRunResults(results: Record<string, {
|
|
60
94
|
status?: string;
|
|
61
95
|
passFailStatus?: string;
|
|
62
|
-
}> | undefined): Pick<RunStats, 'passed' | 'failed' | 'errored' | 'pending' | 'total'>;
|
|
96
|
+
}> | undefined, plannedTotal?: number, runStatus?: BenchmarkRunStatus): Pick<RunStats, 'passed' | 'failed' | 'errored' | 'pending' | 'notRun' | 'total'>;
|
|
97
|
+
/**
|
|
98
|
+
* Pass rate (0-100) over the JUDGED set only: `passed / (passed + failed)`.
|
|
99
|
+
* Excluded: errored (executed, but the evaluator produced no verdict — the
|
|
100
|
+
* repo-wide #242 convention, same as the benchmark runs table and the run
|
|
101
|
+
* inspector), pending (not finished) and notRun (never started). So a run
|
|
102
|
+
* cancelled at 34/62 reports the pass rate of the cases that were actually
|
|
103
|
+
* judged, and an in-flight run doesn't read as "12%" because 50 cases
|
|
104
|
+
* haven't happened yet. Named for what it is — callers that surface it
|
|
105
|
+
* should footnote the excluded errored/notRun counts (the detail page does).
|
|
106
|
+
* Returns `null` when nothing has been judged so callers can render "—"
|
|
107
|
+
* rather than a fabricated 0%.
|
|
108
|
+
*/
|
|
109
|
+
export declare function passRateOverJudged(stats: {
|
|
110
|
+
passed: number;
|
|
111
|
+
failed: number;
|
|
112
|
+
}): number | null;
|
|
63
113
|
/**
|
|
64
114
|
* Calculate statistics for a benchmark run.
|
|
65
115
|
*
|
|
@@ -106,11 +156,41 @@ export declare function getReportIdsFromRun(run: BenchmarkRun): string[];
|
|
|
106
156
|
* simply ignore the extra `pending` field.
|
|
107
157
|
*/
|
|
108
158
|
export declare function computeRunStats(run: {
|
|
159
|
+
status?: BenchmarkRunStatus;
|
|
109
160
|
results?: Record<string, {
|
|
110
161
|
status?: string;
|
|
111
162
|
passFailStatus?: string;
|
|
112
163
|
}>;
|
|
113
164
|
stats?: Partial<RunStatsType> | null;
|
|
114
|
-
|
|
165
|
+
testCaseSnapshots?: unknown[];
|
|
166
|
+
}): Pick<RunStatsType, 'passed' | 'failed' | 'errored' | 'pending' | 'total'> & {
|
|
167
|
+
notRun: number;
|
|
168
|
+
};
|
|
115
169
|
export declare function computeRunStatsFromReports(run: BenchmarkRun, reports: EvaluationReport[]): RunStatsType;
|
|
170
|
+
/**
|
|
171
|
+
* Single canonical "is this run actively in progress" check, shared by every
|
|
172
|
+
* runs-list surface (bug: the Evaluation Runs page and the benchmark-scoped
|
|
173
|
+
* Runs panel each grew their own copy of this, and only one of them actually
|
|
174
|
+
* rendered a running indicator). `status` is authoritative when present
|
|
175
|
+
* (always true for `EvaluationRun`; `BenchmarkRun.status` is only undefined
|
|
176
|
+
* for legacy pre-status data). Falls back to inspecting `results` so legacy
|
|
177
|
+
* runs without a top-level `status` still get a sensible effective status.
|
|
178
|
+
*/
|
|
179
|
+
export declare function getEffectiveRunStatus(run: {
|
|
180
|
+
status?: BenchmarkRunStatus;
|
|
181
|
+
results?: Record<string, {
|
|
182
|
+
status?: string;
|
|
183
|
+
}>;
|
|
184
|
+
}): BenchmarkRunStatus;
|
|
185
|
+
/**
|
|
186
|
+
* True when a run has neither reached a terminal status nor accounted for
|
|
187
|
+
* every planned test case yet. Used to decide whether a runs-list page
|
|
188
|
+
* should keep polling for live updates.
|
|
189
|
+
*/
|
|
190
|
+
export declare function isRunInProgress(run: {
|
|
191
|
+
status?: BenchmarkRunStatus;
|
|
192
|
+
results?: Record<string, {
|
|
193
|
+
status?: string;
|
|
194
|
+
}>;
|
|
195
|
+
}): boolean;
|
|
116
196
|
//# sourceMappingURL=runStats.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"runStats.d.ts","sourceRoot":"","sources":["../../runStats.ts"],"names":[],"mappings":"AAKA;;;;;;;;;;;;;GAaG;AAEH,OAAO,KAAK,EAAE,YAAY,EAAE,gBAAgB,EAAE,QAAQ,IAAI,YAAY,EAAE,MAAM,kBAAkB,CAAC;
|
|
1
|
+
{"version":3,"file":"runStats.d.ts","sourceRoot":"","sources":["../../runStats.ts"],"names":[],"mappings":"AAKA;;;;;;;;;;;;;GAaG;AAEH,OAAO,KAAK,EAAE,YAAY,EAAE,gBAAgB,EAAE,QAAQ,IAAI,YAAY,EAAE,kBAAkB,EAAE,MAAM,kBAAkB,CAAC;AAErH;;GAEG;AACH,MAAM,WAAW,QAAQ;IACvB,qEAAqE;IACrE,MAAM,EAAE,MAAM,CAAC;IACf,yFAAyF;IACzF,MAAM,EAAE,MAAM,CAAC;IACf,gFAAgF;IAChF,OAAO,EAAE,MAAM,CAAC;IAChB;;;;;OAKG;IACH,OAAO,EAAE,MAAM,CAAC;IAChB;;;;;;OAMG;IACH,MAAM,EAAE,MAAM,CAAC;IACf,4CAA4C;IAC5C,KAAK,EAAE,MAAM,CAAC;IACd;;;;;OAKG;IACH,QAAQ,EAAE,MAAM,CAAC;CAClB;AAED,wEAAwE;AACxE,eAAO,MAAM,qBAAqB,EAAE,WAAW,CAAC,kBAAkB,CAAiD,CAAC;AAEpH,wBAAgB,mBAAmB,CAAC,MAAM,EAAE,kBAAkB,GAAG,SAAS,GAAG,OAAO,CAEnF;AAED;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAsCG;AACH,wBAAgB,gBAAgB,CAC9B,OAAO,EAAE,MAAM,CAAC,MAAM,EAAE;IAAE,MAAM,CAAC,EAAE,MAAM,CAAC;IAAC,cAAc,CAAC,EAAE,MAAM,CAAA;CAAE,CAAC,GAAG,SAAS,EACjF,YAAY,CAAC,EAAE,MAAM,EACrB,SAAS,CAAC,EAAE,kBAAkB,GAC7B,IAAI,CAAC,QAAQ,EAAE,QAAQ,GAAG,QAAQ,GAAG,SAAS,GAAG,SAAS,GAAG,QAAQ,GAAG,OAAO,CAAC,CA6BlF;AAED;;;;;;;;;;;GAWG;AACH,wBAAgB,kBAAkB,CAAC,KAAK,EAAE;IAAE,MAAM,EAAE,MAAM,CAAC;IAAC,MAAM,EAAE,MAAM,CAAA;CAAE,GAAG,MAAM,GAAG,IAAI,CAG3F;AAED;;;;;;;;;;;GAWG;AACH,wBAAgB,iBAAiB,CAC/B,GAAG,EAAE,YAAY,EACjB,OAAO,EAAE,MAAM,CAAC,MAAM,EAAE,gBAAgB,GAAG,IAAI,CAAC,GAC/C,QAAQ,CA+EV;AAED;;;;;GAKG;AACH,wBAAgB,mBAAmB,CAAC,GAAG,EAAE,YAAY,GAAG,MAAM,EAAE,CAU/D;AAED;;;;;;;GAOG;AACH;;;;;;;;;;;;;;;;GAgBG;AACH,wBAAgB,eAAe,CAC7B,GAAG,EAAE;IACH,MAAM,CAAC,EAAE,kBAAkB,CAAC;IAC5B,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE;QAAE,MAAM,CAAC,EAAE,MAAM,CAAC;QAAC,cAAc,CAAC,EAAE,MAAM,CAAA;KAAE,CAAC,CAAC;IACvE,KAAK,CAAC,EAAE,OAAO,CAAC,YAAY,CAAC,GAAG,IAAI,CAAC;IACrC,iBAAiB,CAAC,EAAE,OAAO,EAAE,CAAC;CAC/B,GACA,IAAI,CAAC,YAAY,EAAE,QAAQ,GAAG,QAAQ,GAAG,SAAS,GAAG,SAAS,GAAG,OAAO,CAAC,GAAG;IAAE,MAAM,EAAE,MAAM,CAAA;CAAE,CA6ChG;AA0BD,wBAAgB,0BAA0B,CACxC,GAAG,EAAE,YAAY,EACjB,OAAO,EAAE,gBAAgB,EAAE,GAC1B,YAAY,CAgBd;AAED;;;;;;;;GAQG;AACH,wBAAgB,qBAAqB,CACnC,GAAG,EAAE;IAAE,MAAM,CAAC,EAAE,kBAAkB,CAAC;IAAC,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE;QAAE,MAAM,CAAC,EAAE,MAAM,CAAA;KAAE,CAAC,CAAA;CAAE,GAClF,kBAAkB,CAQpB;AAED;;;;GAIG;AACH,wBAAgB,eAAe,CAC7B,GAAG,EAAE;IAAE,MAAM,CAAC,EAAE,kBAAkB,CAAC;IAAC,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE;QAAE,MAAM,CAAC,EAAE,MAAM,CAAA;KAAE,CAAC,CAAA;CAAE,GAClF,OAAO,CAET"}
|