@opensearch-project/agent-health 0.5.2 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -0
- package/cli/dist/index.js +2325 -961
- package/dist/assets/index-D-Np_l_T.js +246 -0
- package/dist/assets/index-vZt9QZKf.css +1 -0
- package/dist/index.html +2 -2
- package/docs/CLI.md +183 -3
- package/docs/CONFIGURATION.md +1 -1
- package/docs/CONNECTORS.md +1 -1
- package/docs/INSTRUMENT_WITH_OTEL.md +7 -0
- package/docs/SDK.md +178 -5
- package/docs/STORAGE_INDEX_FIELD_LIMITS.md +216 -0
- package/docs/skills/AGENT_HEALTH.md +55 -1
- package/docs/skills/add-connector/SKILL.md +5 -1
- package/examples/eval-files/demo.eval.js +1 -1
- package/examples/eval-files/ops-rca-classification.eval.js +71 -0
- package/examples/eval-files/ops-rca-evaluator.json +15 -0
- package/examples/eval-files/sdk-demo.eval.js +72 -0
- package/examples/eval-files/sdk-describe-demo.eval.js +50 -0
- package/examples/eval-files/sdk-hooks-demo.eval.js +1 -1
- package/lib/dist/lib/agentTrends.d.ts +210 -0
- package/lib/dist/lib/agentTrends.d.ts.map +1 -0
- package/lib/dist/lib/agentTrends.js +360 -0
- package/lib/dist/lib/agentTrends.js.map +1 -0
- package/lib/dist/lib/bedrockCompat.d.ts +27 -0
- package/lib/dist/lib/bedrockCompat.d.ts.map +1 -0
- package/lib/dist/lib/bedrockCompat.js +83 -0
- package/lib/dist/lib/bedrockCompat.js.map +1 -0
- package/lib/dist/lib/benchmarkCaseReview.d.ts +114 -0
- package/lib/dist/lib/benchmarkCaseReview.d.ts.map +1 -0
- package/lib/dist/lib/benchmarkCaseReview.js +177 -0
- package/lib/dist/lib/benchmarkCaseReview.js.map +1 -0
- package/lib/dist/lib/benchmarkImage.d.ts +52 -0
- package/lib/dist/lib/benchmarkImage.d.ts.map +1 -0
- package/lib/dist/lib/benchmarkImage.js +113 -0
- package/lib/dist/lib/benchmarkImage.js.map +1 -0
- package/lib/dist/lib/benchmarkRunsTable.d.ts +109 -0
- package/lib/dist/lib/benchmarkRunsTable.d.ts.map +1 -0
- package/lib/dist/lib/benchmarkRunsTable.js +212 -0
- package/lib/dist/lib/benchmarkRunsTable.js.map +1 -0
- package/lib/dist/lib/benchmarkVersionUtils.d.ts +13 -0
- package/lib/dist/lib/benchmarkVersionUtils.d.ts.map +1 -1
- package/lib/dist/lib/benchmarkVersionUtils.js +21 -0
- package/lib/dist/lib/benchmarkVersionUtils.js.map +1 -1
- package/lib/dist/lib/chunkedFetch.d.ts +18 -0
- package/lib/dist/lib/chunkedFetch.d.ts.map +1 -0
- package/lib/dist/lib/chunkedFetch.js +40 -0
- package/lib/dist/lib/chunkedFetch.js.map +1 -0
- package/lib/dist/lib/comparisonInsights.d.ts +151 -0
- package/lib/dist/lib/comparisonInsights.d.ts.map +1 -0
- package/lib/dist/lib/comparisonInsights.js +270 -0
- package/lib/dist/lib/comparisonInsights.js.map +1 -0
- package/lib/dist/lib/config/loader.d.ts.map +1 -1
- package/lib/dist/lib/config/loader.js +16 -1
- package/lib/dist/lib/config/loader.js.map +1 -1
- package/lib/dist/lib/config/types.d.ts +14 -0
- package/lib/dist/lib/config/types.d.ts.map +1 -1
- package/lib/dist/lib/constants.d.ts +11 -0
- package/lib/dist/lib/constants.d.ts.map +1 -1
- package/lib/dist/lib/constants.js +10 -1
- package/lib/dist/lib/constants.js.map +1 -1
- package/lib/dist/lib/contextFormat.d.ts +26 -0
- package/lib/dist/lib/contextFormat.d.ts.map +1 -0
- package/lib/dist/lib/contextFormat.js +28 -0
- package/lib/dist/lib/contextFormat.js.map +1 -0
- package/lib/dist/lib/dashboardMetrics.d.ts +11 -2
- package/lib/dist/lib/dashboardMetrics.d.ts.map +1 -1
- package/lib/dist/lib/dashboardMetrics.js +38 -3
- package/lib/dist/lib/dashboardMetrics.js.map +1 -1
- package/lib/dist/lib/envCompat.d.ts.map +1 -1
- package/lib/dist/lib/envCompat.js +14 -5
- package/lib/dist/lib/envCompat.js.map +1 -1
- package/lib/dist/lib/evaluationRerun.d.ts +102 -0
- package/lib/dist/lib/evaluationRerun.d.ts.map +1 -0
- package/lib/dist/lib/evaluationRerun.js +134 -0
- package/lib/dist/lib/evaluationRerun.js.map +1 -0
- package/lib/dist/lib/judgeFailureSummary.d.ts +66 -0
- package/lib/dist/lib/judgeFailureSummary.d.ts.map +1 -0
- package/lib/dist/lib/judgeFailureSummary.js +68 -0
- package/lib/dist/lib/judgeFailureSummary.js.map +1 -0
- package/lib/dist/lib/judgeStrategies.d.ts +108 -0
- package/lib/dist/lib/judgeStrategies.d.ts.map +1 -0
- package/lib/dist/lib/judgeStrategies.js +135 -0
- package/lib/dist/lib/judgeStrategies.js.map +1 -0
- package/lib/dist/lib/matchers/expect.d.ts +21 -1
- package/lib/dist/lib/matchers/expect.d.ts.map +1 -1
- package/lib/dist/lib/matchers/expect.js +51 -0
- package/lib/dist/lib/matchers/expect.js.map +1 -1
- package/lib/dist/lib/matchers/judgeAccessor.d.ts +4 -0
- package/lib/dist/lib/matchers/judgeAccessor.d.ts.map +1 -1
- package/lib/dist/lib/matchers/judgeAccessor.js +13 -2
- package/lib/dist/lib/matchers/judgeAccessor.js.map +1 -1
- package/lib/dist/lib/matchers/judgeReasoningParse.d.ts +49 -0
- package/lib/dist/lib/matchers/judgeReasoningParse.d.ts.map +1 -0
- package/lib/dist/lib/matchers/judgeReasoningParse.js +128 -0
- package/lib/dist/lib/matchers/judgeReasoningParse.js.map +1 -0
- package/lib/dist/lib/matchers/traces.d.ts +17 -2
- package/lib/dist/lib/matchers/traces.d.ts.map +1 -1
- package/lib/dist/lib/matchers/traces.js +136 -16
- package/lib/dist/lib/matchers/traces.js.map +1 -1
- package/lib/dist/lib/matchers/tracesPricing.d.ts +38 -0
- package/lib/dist/lib/matchers/tracesPricing.d.ts.map +1 -0
- package/lib/dist/lib/matchers/tracesPricing.js +64 -0
- package/lib/dist/lib/matchers/tracesPricing.js.map +1 -0
- package/lib/dist/lib/matchers/types.d.ts +25 -0
- package/lib/dist/lib/matchers/types.d.ts.map +1 -1
- package/lib/dist/lib/resolveCanonicalRun.d.ts +23 -0
- package/lib/dist/lib/resolveCanonicalRun.d.ts.map +1 -0
- package/lib/dist/lib/resolveCanonicalRun.js +26 -0
- package/lib/dist/lib/resolveCanonicalRun.js.map +1 -0
- package/lib/dist/lib/runActions.d.ts +120 -0
- package/lib/dist/lib/runActions.d.ts.map +1 -0
- package/lib/dist/lib/runActions.js +130 -0
- package/lib/dist/lib/runActions.js.map +1 -0
- package/lib/dist/lib/runInsights.d.ts +86 -0
- package/lib/dist/lib/runInsights.d.ts.map +1 -0
- package/lib/dist/lib/runInsights.js +185 -0
- package/lib/dist/lib/runInsights.js.map +1 -0
- package/lib/dist/lib/runName.d.ts +29 -0
- package/lib/dist/lib/runName.d.ts.map +1 -0
- package/lib/dist/lib/runName.js +38 -0
- package/lib/dist/lib/runName.js.map +1 -0
- package/lib/dist/lib/runReportPath.d.ts +16 -0
- package/lib/dist/lib/runReportPath.d.ts.map +1 -0
- package/lib/dist/lib/runReportPath.js +22 -0
- package/lib/dist/lib/runReportPath.js.map +1 -0
- package/lib/dist/lib/runSort.d.ts +27 -0
- package/lib/dist/lib/runSort.d.ts.map +1 -0
- package/lib/dist/lib/runSort.js +31 -0
- package/lib/dist/lib/runSort.js.map +1 -0
- package/lib/dist/lib/runStats.d.ts +109 -5
- package/lib/dist/lib/runStats.d.ts.map +1 -1
- package/lib/dist/lib/runStats.js +193 -10
- package/lib/dist/lib/runStats.js.map +1 -1
- package/lib/dist/lib/testCases/define.d.ts.map +1 -1
- package/lib/dist/lib/testCases/define.js +93 -46
- package/lib/dist/lib/testCases/define.js.map +1 -1
- package/lib/dist/lib/testCases/judge.d.ts.map +1 -1
- package/lib/dist/lib/testCases/judge.js +22 -4
- package/lib/dist/lib/testCases/judge.js.map +1 -1
- package/lib/dist/lib/testCases/loader.d.ts +35 -1
- package/lib/dist/lib/testCases/loader.d.ts.map +1 -1
- package/lib/dist/lib/testCases/loader.js +278 -31
- package/lib/dist/lib/testCases/loader.js.map +1 -1
- package/lib/dist/lib/trajectoryStepDisplay.d.ts +25 -0
- package/lib/dist/lib/trajectoryStepDisplay.d.ts.map +1 -0
- package/lib/dist/lib/trajectoryStepDisplay.js +42 -0
- package/lib/dist/lib/trajectoryStepDisplay.js.map +1 -0
- package/lib/dist/lib/utils.d.ts +34 -0
- package/lib/dist/lib/utils.d.ts.map +1 -1
- package/lib/dist/lib/utils.js +49 -0
- package/lib/dist/lib/utils.js.map +1 -1
- package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.d.ts +68 -22
- package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.d.ts.map +1 -1
- package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.js +148 -111
- package/lib/dist/services/connectors/claude-code/ClaudeCodeConnector.js.map +1 -1
- package/lib/dist/services/connectors/kiro/KiroConnector.d.ts +21 -13
- package/lib/dist/services/connectors/kiro/KiroConnector.d.ts.map +1 -1
- package/lib/dist/services/connectors/kiro/KiroConnector.js +22 -25
- package/lib/dist/services/connectors/kiro/KiroConnector.js.map +1 -1
- package/lib/dist/services/connectors/pi/PiConnector.d.ts +49 -10
- package/lib/dist/services/connectors/pi/PiConnector.d.ts.map +1 -1
- package/lib/dist/services/connectors/pi/PiConnector.js +102 -86
- package/lib/dist/services/connectors/pi/PiConnector.js.map +1 -1
- package/lib/dist/services/connectors/subprocess/SubprocessConnector.d.ts +59 -14
- package/lib/dist/services/connectors/subprocess/SubprocessConnector.d.ts.map +1 -1
- package/lib/dist/services/connectors/subprocess/SubprocessConnector.js +95 -61
- package/lib/dist/services/connectors/subprocess/SubprocessConnector.js.map +1 -1
- package/lib/dist/services/evaluation/bedrockJudge.d.ts +18 -0
- package/lib/dist/services/evaluation/bedrockJudge.d.ts.map +1 -1
- package/lib/dist/services/evaluation/bedrockJudge.js +4 -0
- package/lib/dist/services/evaluation/bedrockJudge.js.map +1 -1
- package/lib/dist/services/evaluation/index.d.ts +18 -1
- package/lib/dist/services/evaluation/index.d.ts.map +1 -1
- package/lib/dist/services/evaluation/index.js +156 -19
- package/lib/dist/services/evaluation/index.js.map +1 -1
- package/lib/dist/services/metrics.d.ts +55 -0
- package/lib/dist/services/metrics.d.ts.map +1 -0
- package/lib/dist/services/metrics.js +89 -0
- package/lib/dist/services/metrics.js.map +1 -0
- package/lib/dist/services/storage/asyncBenchmarkStorage.d.ts +1 -1
- package/lib/dist/services/storage/asyncBenchmarkStorage.d.ts.map +1 -1
- package/lib/dist/services/storage/asyncBenchmarkStorage.js +30 -3
- package/lib/dist/services/storage/asyncBenchmarkStorage.js.map +1 -1
- package/lib/dist/services/storage/asyncRunStorage.d.ts +20 -1
- package/lib/dist/services/storage/asyncRunStorage.d.ts.map +1 -1
- package/lib/dist/services/storage/asyncRunStorage.js +125 -2
- package/lib/dist/services/storage/asyncRunStorage.js.map +1 -1
- package/lib/dist/services/storage/asyncTestCaseStorage.d.ts +18 -4
- package/lib/dist/services/storage/asyncTestCaseStorage.d.ts.map +1 -1
- package/lib/dist/services/storage/asyncTestCaseStorage.js +21 -4
- package/lib/dist/services/storage/asyncTestCaseStorage.js.map +1 -1
- package/lib/dist/services/storage/opensearchClient.d.ts +47 -12
- package/lib/dist/services/storage/opensearchClient.d.ts.map +1 -1
- package/lib/dist/services/storage/opensearchClient.js +12 -5
- package/lib/dist/services/storage/opensearchClient.js.map +1 -1
- package/lib/dist/services/traces/browserRecovery.d.ts +12 -1
- package/lib/dist/services/traces/browserRecovery.d.ts.map +1 -1
- package/lib/dist/services/traces/browserRecovery.js +35 -5
- package/lib/dist/services/traces/browserRecovery.js.map +1 -1
- package/lib/dist/services/traces/index.d.ts +8 -1
- package/lib/dist/services/traces/index.d.ts.map +1 -1
- package/lib/dist/services/traces/index.js +33 -12
- package/lib/dist/services/traces/index.js.map +1 -1
- package/lib/dist/services/traces/judgeAgentsHints.d.ts +98 -3
- package/lib/dist/services/traces/judgeAgentsHints.d.ts.map +1 -1
- package/lib/dist/services/traces/judgeAgentsHints.js +144 -3
- package/lib/dist/services/traces/judgeAgentsHints.js.map +1 -1
- package/lib/dist/services/traces/messageExtraction.d.ts.map +1 -1
- package/lib/dist/services/traces/messageExtraction.js +95 -33
- package/lib/dist/services/traces/messageExtraction.js.map +1 -1
- package/lib/dist/services/traces/spansToTrajectory.d.ts.map +1 -1
- package/lib/dist/services/traces/spansToTrajectory.js +51 -13
- package/lib/dist/services/traces/spansToTrajectory.js.map +1 -1
- package/lib/dist/services/traces/tracePoller.d.ts +27 -3
- package/lib/dist/services/traces/tracePoller.d.ts.map +1 -1
- package/lib/dist/services/traces/tracePoller.js +196 -33
- package/lib/dist/services/traces/tracePoller.js.map +1 -1
- package/lib/dist/services/traces/trajectoryMerge.d.ts +79 -0
- package/lib/dist/services/traces/trajectoryMerge.d.ts.map +1 -0
- package/lib/dist/services/traces/trajectoryMerge.js +109 -0
- package/lib/dist/services/traces/trajectoryMerge.js.map +1 -0
- package/lib/dist/types/index.d.ts +250 -3
- package/lib/dist/types/index.d.ts.map +1 -1
- package/lib/dist/types/index.js +24 -0
- package/lib/dist/types/index.js.map +1 -1
- package/package.json +8 -5
- package/server/dist/app.js +5490 -1557
- package/server/dist/index.js +5490 -1557
- package/dist/assets/index-CCQRDlO0.js +0 -243
- package/dist/assets/index-CNHQVbcj.css +0 -1
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Pure, isomorphic predicates for the run-lifecycle action matrix (delete /
|
|
3
|
+
* cancel / re-run / retry-judgement) shared by every run surface (runs list,
|
|
4
|
+
* benchmark runs list, run detail/report page, inspector header) AND by the
|
|
5
|
+
* server routes that enforce the same rules server-side. No storage/IO here
|
|
6
|
+
* — callers pass in the run document they already have.
|
|
7
|
+
*
|
|
8
|
+
* Action matrix (see AGENTS.md / PR description for the full writeup):
|
|
9
|
+
* - Delete: any run, any status. Always available (existing endpoint).
|
|
10
|
+
* - Cancel: only while `status === 'running'`.
|
|
11
|
+
* - Re-run: only top-level EvaluationRun docs (docType === 'evaluation-run').
|
|
12
|
+
* Legacy benchmark-embedded BenchmarkRun rows don't support the
|
|
13
|
+
* provenance-tracked rerun endpoint (pre-existing constraint — see
|
|
14
|
+
* RunConfigDialog / EvalRunsPage).
|
|
15
|
+
* - Retry judgement: only EvaluationRun docs, only when the run is
|
|
16
|
+
* terminal (not running) AND it has at least one test case whose agent
|
|
17
|
+
* execution completed but the judge produced NO verdict (a judge-failed
|
|
18
|
+
* / "errored" case — trace timeout, judge 400, "evaluator could not
|
|
19
|
+
* run" — as opposed to an agent-failed one — retrying the judge on a
|
|
20
|
+
* case the agent itself never finished has nothing to re-grade). Same
|
|
21
|
+
* predicate the retry-judgement pipeline itself selects on
|
|
22
|
+
* (services/evaluation/retryJudgement.ts `isJudgeFailedCase`, keyed on
|
|
23
|
+
* the report's `metricsStatus: 'error'`, which the runner mirrors onto
|
|
24
|
+
* the run's results map as a `completed` result with no
|
|
25
|
+
* `passFailStatus`) and that `lib/runStats` buckets as `errored`.
|
|
26
|
+
*/
|
|
27
|
+
import type { BenchmarkRun, EvaluationRun } from '../types/index.js';
|
|
28
|
+
/** Minimal shape both BenchmarkRun and EvaluationRun satisfy for these checks. */
|
|
29
|
+
export type RunLike = Pick<BenchmarkRun | EvaluationRun, 'status' | 'results'> & {
|
|
30
|
+
docType?: string;
|
|
31
|
+
};
|
|
32
|
+
/**
|
|
33
|
+
* True when `run` is a top-level EvaluationRun document (created via
|
|
34
|
+
* `POST /api/storage/evaluation-runs`), as opposed to a legacy
|
|
35
|
+
* benchmark-embedded BenchmarkRun (`benchmark.runs[]`). The two share a lot
|
|
36
|
+
* of shape but only EvaluationRun docs carry `docType: 'evaluation-run'` and
|
|
37
|
+
* support the rerun/retry-judgement endpoints.
|
|
38
|
+
*
|
|
39
|
+
* Null-tolerant wrapper over the typed predicate in `types/index.ts` (the
|
|
40
|
+
* single source of truth for the docType discriminator) — kept so callers
|
|
41
|
+
* holding a possibly-null run don't need their own guard.
|
|
42
|
+
*/
|
|
43
|
+
export declare function isEvaluationRun(run: RunLike | null | undefined): run is EvaluationRun;
|
|
44
|
+
/** True while the run has an in-progress executor that a Cancel action could stop. */
|
|
45
|
+
export declare function isRunRunning(run: RunLike | null | undefined): boolean;
|
|
46
|
+
/** True once a run has reached any terminal state (not running/pending). */
|
|
47
|
+
export declare function isRunTerminal(run: RunLike | null | undefined): boolean;
|
|
48
|
+
/**
|
|
49
|
+
* Count test cases where the AGENT finished (`status === 'completed'`) but
|
|
50
|
+
* the JUDGE produced no verdict (`passFailStatus` neither 'passed' nor
|
|
51
|
+
* 'failed') — the "errored" bucket of `lib/runStats` `bucketRunResults`
|
|
52
|
+
* (issue #242) and exactly the set `POST .../retry-judgement` (default
|
|
53
|
+
* `scope=errored`) will re-judge. Deliberately excludes:
|
|
54
|
+
* - `status !== 'completed'` (agent-failed/cancelled/pending cases — no
|
|
55
|
+
* trajectory to re-judge, or nothing ran).
|
|
56
|
+
* - a real 'failed' verdict — the judge DID run and graded the case; that
|
|
57
|
+
* is a legitimate result, not a judge failure (re-grading it is
|
|
58
|
+
* `scope=all`, opt-in from the inspector's dedicated button).
|
|
59
|
+
*
|
|
60
|
+
* `passFailStatus` isn't declared on `EvaluationRun['results']`'s static
|
|
61
|
+
* type (a pre-existing gap — evaluationRunner.ts writes it via an `as any`
|
|
62
|
+
* spread) so this reads it defensively.
|
|
63
|
+
*/
|
|
64
|
+
export declare function countJudgeFailed(run: RunLike | null | undefined): number;
|
|
65
|
+
export interface RunActionVisibility {
|
|
66
|
+
/** Delete is always available for any run in any status. */
|
|
67
|
+
canDelete: boolean;
|
|
68
|
+
/** Cancel is available only while the run is actively running. */
|
|
69
|
+
canCancel: boolean;
|
|
70
|
+
/** Re-run is available only for top-level EvaluationRun docs. */
|
|
71
|
+
canRerun: boolean;
|
|
72
|
+
/** Reason to show (e.g. as a disabled-item tooltip) when canRerun is false. */
|
|
73
|
+
rerunDisabledReason?: string;
|
|
74
|
+
/** Retry judgement: EvaluationRun, terminal, with >0 judge-failed cases. */
|
|
75
|
+
canRetryJudgement: boolean;
|
|
76
|
+
/** Reason to show when canRetryJudgement is false. */
|
|
77
|
+
retryJudgementDisabledReason?: string;
|
|
78
|
+
/** Number of judge-failed test cases (0 when not applicable/unknown). */
|
|
79
|
+
judgeFailedCount: number;
|
|
80
|
+
}
|
|
81
|
+
/**
|
|
82
|
+
* Minimum time a run must have been persisted before a Cancel request with
|
|
83
|
+
* no in-memory executor token is allowed to take the "zombie" fallback path
|
|
84
|
+
* (mark cancelled directly — see getRunActionVisibility callers in the
|
|
85
|
+
* cancel routes). Guards the narrow window right after a run is created:
|
|
86
|
+
* the doc is persisted (and therefore visible to a concurrent Cancel
|
|
87
|
+
* request) strictly before its executor registers its cancellation token,
|
|
88
|
+
* so a Cancel that lands in that gap would otherwise mark a run "cancelled"
|
|
89
|
+
* moments before its own executor starts making progress on it. A brand-new
|
|
90
|
+
* run is also the case the fallback is LEAST useful for — "zombie" (no
|
|
91
|
+
* executor anywhere) is far more plausible once a run has been running for
|
|
92
|
+
* a while than in its first couple of seconds.
|
|
93
|
+
*/
|
|
94
|
+
export declare const ZOMBIE_CANCEL_MIN_AGE_MS = 5000;
|
|
95
|
+
/**
|
|
96
|
+
* True once a run is old enough that a tokenless Cancel request can safely
|
|
97
|
+
* assume its executor (if any) would already have registered a
|
|
98
|
+
* cancellation token — i.e. it's safe to treat "no token" as "no executor"
|
|
99
|
+
* rather than "executor hasn't started yet".
|
|
100
|
+
*
|
|
101
|
+
* NOTE — known limitation, not fixed by this check: cancellation tokens are
|
|
102
|
+
* tracked in an in-memory `Map` scoped to ONE server process. In a
|
|
103
|
+
* multi-process/clustered deployment, a Cancel request routed to a
|
|
104
|
+
* DIFFERENT process than the one executing the run will always find no
|
|
105
|
+
* token there, regardless of run age, and this zombie fallback will mark
|
|
106
|
+
* the run cancelled in storage even though it's alive and progressing on
|
|
107
|
+
* another process. This mirrors a pre-existing, documented constraint of
|
|
108
|
+
* this codebase's run-execution model (see AGENTS.md's "orphan-run
|
|
109
|
+
* recovery" notes: "active is tracked per-process"); fixing it for real
|
|
110
|
+
* needs the same heartbeat-based ownership (`run.heartbeatAt`) that doc
|
|
111
|
+
* already calls out as the eventual replacement. Out of scope here.
|
|
112
|
+
*/
|
|
113
|
+
export declare function isOldEnoughForZombieCancel(createdAt: string | undefined, now?: number): boolean;
|
|
114
|
+
/**
|
|
115
|
+
* Compute the full action-visibility matrix for one run. Pure function —
|
|
116
|
+
* safe to call from both React components and server-side route validation
|
|
117
|
+
* so the two never drift.
|
|
118
|
+
*/
|
|
119
|
+
export declare function getRunActionVisibility(run: RunLike | null | undefined): RunActionVisibility;
|
|
120
|
+
//# sourceMappingURL=runActions.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"runActions.d.ts","sourceRoot":"","sources":["../../runActions.ts"],"names":[],"mappings":"AAKA;;;;;;;;;;;;;;;;;;;;;;;;;GAyBG;AAEH,OAAO,KAAK,EAAE,YAAY,EAAE,aAAa,EAAE,MAAM,SAAS,CAAC;AAG3D,kFAAkF;AAClF,MAAM,MAAM,OAAO,GAAG,IAAI,CAAC,YAAY,GAAG,aAAa,EAAE,QAAQ,GAAG,SAAS,CAAC,GAAG;IAC/E,OAAO,CAAC,EAAE,MAAM,CAAC;CAClB,CAAC;AAEF;;;;;;;;;;GAUG;AACH,wBAAgB,eAAe,CAAC,GAAG,EAAE,OAAO,GAAG,IAAI,GAAG,SAAS,GAAG,GAAG,IAAI,aAAa,CAErF;AAED,sFAAsF;AACtF,wBAAgB,YAAY,CAAC,GAAG,EAAE,OAAO,GAAG,IAAI,GAAG,SAAS,GAAG,OAAO,CAErE;AAED,4EAA4E;AAC5E,wBAAgB,aAAa,CAAC,GAAG,EAAE,OAAO,GAAG,IAAI,GAAG,SAAS,GAAG,OAAO,CAEtE;AAED;;;;;;;;;;;;;;;GAeG;AACH,wBAAgB,gBAAgB,CAAC,GAAG,EAAE,OAAO,GAAG,IAAI,GAAG,SAAS,GAAG,MAAM,CAUxE;AAED,MAAM,WAAW,mBAAmB;IAClC,4DAA4D;IAC5D,SAAS,EAAE,OAAO,CAAC;IACnB,kEAAkE;IAClE,SAAS,EAAE,OAAO,CAAC;IACnB,iEAAiE;IACjE,QAAQ,EAAE,OAAO,CAAC;IAClB,+EAA+E;IAC/E,mBAAmB,CAAC,EAAE,MAAM,CAAC;IAC7B,4EAA4E;IAC5E,iBAAiB,EAAE,OAAO,CAAC;IAC3B,sDAAsD;IACtD,4BAA4B,CAAC,EAAE,MAAM,CAAC;IACtC,yEAAyE;IACzE,gBAAgB,EAAE,MAAM,CAAC;CAC1B;AAOD;;;;;;;;;;;;GAYG;AACH,eAAO,MAAM,wBAAwB,OAAO,CAAC;AAE7C;;;;;;;;;;;;;;;;;GAiBG;AACH,wBAAgB,0BAA0B,CAAC,SAAS,EAAE,MAAM,GAAG,SAAS,EAAE,GAAG,GAAE,MAAmB,GAAG,OAAO,CAI3G;AAED;;;;GAIG;AACH,wBAAgB,sBAAsB,CAAC,GAAG,EAAE,OAAO,GAAG,IAAI,GAAG,SAAS,GAAG,mBAAmB,CAuB3F"}
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Copyright OpenSearch Contributors
|
|
3
|
+
* SPDX-License-Identifier: Apache-2.0
|
|
4
|
+
*/
|
|
5
|
+
import { isEvaluationRun as isEvaluationRunDoc } from '../types/index.js';
|
|
6
|
+
/**
|
|
7
|
+
* True when `run` is a top-level EvaluationRun document (created via
|
|
8
|
+
* `POST /api/storage/evaluation-runs`), as opposed to a legacy
|
|
9
|
+
* benchmark-embedded BenchmarkRun (`benchmark.runs[]`). The two share a lot
|
|
10
|
+
* of shape but only EvaluationRun docs carry `docType: 'evaluation-run'` and
|
|
11
|
+
* support the rerun/retry-judgement endpoints.
|
|
12
|
+
*
|
|
13
|
+
* Null-tolerant wrapper over the typed predicate in `types/index.ts` (the
|
|
14
|
+
* single source of truth for the docType discriminator) — kept so callers
|
|
15
|
+
* holding a possibly-null run don't need their own guard.
|
|
16
|
+
*/
|
|
17
|
+
export function isEvaluationRun(run) {
|
|
18
|
+
return !!run && isEvaluationRunDoc(run);
|
|
19
|
+
}
|
|
20
|
+
/** True while the run has an in-progress executor that a Cancel action could stop. */
|
|
21
|
+
export function isRunRunning(run) {
|
|
22
|
+
return run?.status === 'running';
|
|
23
|
+
}
|
|
24
|
+
/** True once a run has reached any terminal state (not running/pending). */
|
|
25
|
+
export function isRunTerminal(run) {
|
|
26
|
+
return !!run && (run.status === 'completed' || run.status === 'failed' || run.status === 'cancelled');
|
|
27
|
+
}
|
|
28
|
+
/**
|
|
29
|
+
* Count test cases where the AGENT finished (`status === 'completed'`) but
|
|
30
|
+
* the JUDGE produced no verdict (`passFailStatus` neither 'passed' nor
|
|
31
|
+
* 'failed') — the "errored" bucket of `lib/runStats` `bucketRunResults`
|
|
32
|
+
* (issue #242) and exactly the set `POST .../retry-judgement` (default
|
|
33
|
+
* `scope=errored`) will re-judge. Deliberately excludes:
|
|
34
|
+
* - `status !== 'completed'` (agent-failed/cancelled/pending cases — no
|
|
35
|
+
* trajectory to re-judge, or nothing ran).
|
|
36
|
+
* - a real 'failed' verdict — the judge DID run and graded the case; that
|
|
37
|
+
* is a legitimate result, not a judge failure (re-grading it is
|
|
38
|
+
* `scope=all`, opt-in from the inspector's dedicated button).
|
|
39
|
+
*
|
|
40
|
+
* `passFailStatus` isn't declared on `EvaluationRun['results']`'s static
|
|
41
|
+
* type (a pre-existing gap — evaluationRunner.ts writes it via an `as any`
|
|
42
|
+
* spread) so this reads it defensively.
|
|
43
|
+
*/
|
|
44
|
+
export function countJudgeFailed(run) {
|
|
45
|
+
if (!run?.results)
|
|
46
|
+
return 0;
|
|
47
|
+
let count = 0;
|
|
48
|
+
for (const r of Object.values(run.results)) {
|
|
49
|
+
const result = r;
|
|
50
|
+
if (result.status !== 'completed')
|
|
51
|
+
continue;
|
|
52
|
+
if (result.passFailStatus === 'passed' || result.passFailStatus === 'failed')
|
|
53
|
+
continue;
|
|
54
|
+
count++;
|
|
55
|
+
}
|
|
56
|
+
return count;
|
|
57
|
+
}
|
|
58
|
+
const RERUN_NOT_SUPPORTED_REASON = "Re-run isn't available for legacy benchmark-embedded runs";
|
|
59
|
+
const RETRY_JUDGEMENT_NOT_SUPPORTED_REASON = "Retry judgement isn't available for legacy benchmark-embedded runs";
|
|
60
|
+
const RETRY_JUDGEMENT_STILL_RUNNING_REASON = 'Retry judgement is only available once the run finishes';
|
|
61
|
+
const RETRY_JUDGEMENT_NONE_FAILED_REASON = 'No judge-failed test cases to retry';
|
|
62
|
+
/**
|
|
63
|
+
* Minimum time a run must have been persisted before a Cancel request with
|
|
64
|
+
* no in-memory executor token is allowed to take the "zombie" fallback path
|
|
65
|
+
* (mark cancelled directly — see getRunActionVisibility callers in the
|
|
66
|
+
* cancel routes). Guards the narrow window right after a run is created:
|
|
67
|
+
* the doc is persisted (and therefore visible to a concurrent Cancel
|
|
68
|
+
* request) strictly before its executor registers its cancellation token,
|
|
69
|
+
* so a Cancel that lands in that gap would otherwise mark a run "cancelled"
|
|
70
|
+
* moments before its own executor starts making progress on it. A brand-new
|
|
71
|
+
* run is also the case the fallback is LEAST useful for — "zombie" (no
|
|
72
|
+
* executor anywhere) is far more plausible once a run has been running for
|
|
73
|
+
* a while than in its first couple of seconds.
|
|
74
|
+
*/
|
|
75
|
+
export const ZOMBIE_CANCEL_MIN_AGE_MS = 5000;
|
|
76
|
+
/**
|
|
77
|
+
* True once a run is old enough that a tokenless Cancel request can safely
|
|
78
|
+
* assume its executor (if any) would already have registered a
|
|
79
|
+
* cancellation token — i.e. it's safe to treat "no token" as "no executor"
|
|
80
|
+
* rather than "executor hasn't started yet".
|
|
81
|
+
*
|
|
82
|
+
* NOTE — known limitation, not fixed by this check: cancellation tokens are
|
|
83
|
+
* tracked in an in-memory `Map` scoped to ONE server process. In a
|
|
84
|
+
* multi-process/clustered deployment, a Cancel request routed to a
|
|
85
|
+
* DIFFERENT process than the one executing the run will always find no
|
|
86
|
+
* token there, regardless of run age, and this zombie fallback will mark
|
|
87
|
+
* the run cancelled in storage even though it's alive and progressing on
|
|
88
|
+
* another process. This mirrors a pre-existing, documented constraint of
|
|
89
|
+
* this codebase's run-execution model (see AGENTS.md's "orphan-run
|
|
90
|
+
* recovery" notes: "active is tracked per-process"); fixing it for real
|
|
91
|
+
* needs the same heartbeat-based ownership (`run.heartbeatAt`) that doc
|
|
92
|
+
* already calls out as the eventual replacement. Out of scope here.
|
|
93
|
+
*/
|
|
94
|
+
export function isOldEnoughForZombieCancel(createdAt, now = Date.now()) {
|
|
95
|
+
const created = createdAt ? Date.parse(createdAt) : NaN;
|
|
96
|
+
if (Number.isNaN(created))
|
|
97
|
+
return true; // no timestamp to compare against — don't block on it
|
|
98
|
+
return now - created >= ZOMBIE_CANCEL_MIN_AGE_MS;
|
|
99
|
+
}
|
|
100
|
+
/**
|
|
101
|
+
* Compute the full action-visibility matrix for one run. Pure function —
|
|
102
|
+
* safe to call from both React components and server-side route validation
|
|
103
|
+
* so the two never drift.
|
|
104
|
+
*/
|
|
105
|
+
export function getRunActionVisibility(run) {
|
|
106
|
+
const evalRun = isEvaluationRun(run);
|
|
107
|
+
const running = isRunRunning(run);
|
|
108
|
+
const terminal = isRunTerminal(run);
|
|
109
|
+
const judgeFailedCount = evalRun ? countJudgeFailed(run) : 0;
|
|
110
|
+
const canRetryJudgement = evalRun && terminal && judgeFailedCount > 0;
|
|
111
|
+
let retryJudgementDisabledReason;
|
|
112
|
+
if (!canRetryJudgement) {
|
|
113
|
+
if (!evalRun)
|
|
114
|
+
retryJudgementDisabledReason = RETRY_JUDGEMENT_NOT_SUPPORTED_REASON;
|
|
115
|
+
else if (!terminal)
|
|
116
|
+
retryJudgementDisabledReason = RETRY_JUDGEMENT_STILL_RUNNING_REASON;
|
|
117
|
+
else
|
|
118
|
+
retryJudgementDisabledReason = RETRY_JUDGEMENT_NONE_FAILED_REASON;
|
|
119
|
+
}
|
|
120
|
+
return {
|
|
121
|
+
canDelete: true,
|
|
122
|
+
canCancel: running,
|
|
123
|
+
canRerun: evalRun,
|
|
124
|
+
rerunDisabledReason: evalRun ? undefined : RERUN_NOT_SUPPORTED_REASON,
|
|
125
|
+
canRetryJudgement,
|
|
126
|
+
retryJudgementDisabledReason,
|
|
127
|
+
judgeFailedCount,
|
|
128
|
+
};
|
|
129
|
+
}
|
|
130
|
+
//# sourceMappingURL=runActions.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"runActions.js","sourceRoot":"","sources":["../../runActions.ts"],"names":[],"mappings":"AAAA;;;GAGG;AA8BH,OAAO,EAAE,eAAe,IAAI,kBAAkB,EAAE,MAAM,SAAS,CAAC;AAOhE;;;;;;;;;;GAUG;AACH,MAAM,UAAU,eAAe,CAAC,GAA+B;IAC7D,OAAO,CAAC,CAAC,GAAG,IAAI,kBAAkB,CAAC,GAAmC,CAAC,CAAC;AAC1E,CAAC;AAED,sFAAsF;AACtF,MAAM,UAAU,YAAY,CAAC,GAA+B;IAC1D,OAAO,GAAG,EAAE,MAAM,KAAK,SAAS,CAAC;AACnC,CAAC;AAED,4EAA4E;AAC5E,MAAM,UAAU,aAAa,CAAC,GAA+B;IAC3D,OAAO,CAAC,CAAC,GAAG,IAAI,CAAC,GAAG,CAAC,MAAM,KAAK,WAAW,IAAI,GAAG,CAAC,MAAM,KAAK,QAAQ,IAAI,GAAG,CAAC,MAAM,KAAK,WAAW,CAAC,CAAC;AACxG,CAAC;AAED;;;;;;;;;;;;;;;GAeG;AACH,MAAM,UAAU,gBAAgB,CAAC,GAA+B;IAC9D,IAAI,CAAC,GAAG,EAAE,OAAO;QAAE,OAAO,CAAC,CAAC;IAC5B,IAAI,KAAK,GAAG,CAAC,CAAC;IACd,KAAK,MAAM,CAAC,IAAI,MAAM,CAAC,MAAM,CAAC,GAAG,CAAC,OAAO,CAAC,EAAE,CAAC;QAC3C,MAAM,MAAM,GAAG,CAAwD,CAAC;QACxE,IAAI,MAAM,CAAC,MAAM,KAAK,WAAW;YAAE,SAAS;QAC5C,IAAI,MAAM,CAAC,cAAc,KAAK,QAAQ,IAAI,MAAM,CAAC,cAAc,KAAK,QAAQ;YAAE,SAAS;QACvF,KAAK,EAAE,CAAC;IACV,CAAC;IACD,OAAO,KAAK,CAAC;AACf,CAAC;AAmBD,MAAM,0BAA0B,GAAG,2DAA2D,CAAC;AAC/F,MAAM,oCAAoC,GAAG,oEAAoE,CAAC;AAClH,MAAM,oCAAoC,GAAG,yDAAyD,CAAC;AACvG,MAAM,kCAAkC,GAAG,qCAAqC,CAAC;AAEjF;;;;;;;;;;;;GAYG;AACH,MAAM,CAAC,MAAM,wBAAwB,GAAG,IAAI,CAAC;AAE7C;;;;;;;;;;;;;;;;;GAiBG;AACH,MAAM,UAAU,0BAA0B,CAAC,SAA6B,EAAE,MAAc,IAAI,CAAC,GAAG,EAAE;IAChG,MAAM,OAAO,GAAG,SAAS,CAAC,CAAC,CAAC,IAAI,CAAC,KAAK,CAAC,SAAS,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC;IACxD,IAAI,MAAM,CAAC,KAAK,CAAC,OAAO,CAAC;QAAE,OAAO,IAAI,CAAC,CAAC,sDAAsD;IAC9F,OAAO,GAAG,GAAG,OAAO,IAAI,wBAAwB,CAAC;AACnD,CAAC;AAED;;;;GAIG;AACH,MAAM,UAAU,sBAAsB,CAAC,GAA+B;IACpE,MAAM,OAAO,GAAG,eAAe,CAAC,GAAG,CAAC,CAAC;IACrC,MAAM,OAAO,GAAG,YAAY,CAAC,GAAG,CAAC,CAAC;IAClC,MAAM,QAAQ,GAAG,aAAa,CAAC,GAAG,CAAC,CAAC;IACpC,MAAM,gBAAgB,GAAG,OAAO,CAAC,CAAC,CAAC,gBAAgB,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;IAE7D,MAAM,iBAAiB,GAAG,OAAO,IAAI,QAAQ,IAAI,gBAAgB,GAAG,CAAC,CAAC;IACtE,IAAI,4BAAgD,CAAC;IACrD,IAAI,CAAC,iBAAiB,EAAE,CAAC;QACvB,IAAI,CAAC,OAAO;YAAE,4BAA4B,GAAG,oCAAoC,CAAC;aAC7E,IAAI,CAAC,QAAQ;YAAE,4BAA4B,GAAG,oCAAoC,CAAC;;YACnF,4BAA4B,GAAG,kCAAkC,CAAC;IACzE,CAAC;IAED,OAAO;QACL,SAAS,EAAE,IAAI;QACf,SAAS,EAAE,OAAO;QAClB,QAAQ,EAAE,OAAO;QACjB,mBAAmB,EAAE,OAAO,CAAC,CAAC,CAAC,SAAS,CAAC,CAAC,CAAC,0BAA0B;QACrE,iBAAiB;QACjB,4BAA4B;QAC5B,gBAAgB;KACjB,CAAC;AACJ,CAAC"}
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Deterministic (no-LLM) aggregation helpers for the run-report "insights"
|
|
3
|
+
* pane (RunInsightsPane.tsx), shown on the bare
|
|
4
|
+
* `/benchmarks/:benchmarkId/runs/:runId` route when no test case is
|
|
5
|
+
* selected — see owner feedback on goyamegh/run-report-redesign: "if no
|
|
6
|
+
* test case is selected, the right side can show an aggregated view ...
|
|
7
|
+
* why did the failing tests fail — something that is complete info."
|
|
8
|
+
*
|
|
9
|
+
* Everything here is a pure function over already-fetched data (report
|
|
10
|
+
* summaries + test-case categories). No network calls, no LLM calls — v1
|
|
11
|
+
* is explicitly deterministic-only per the product ask.
|
|
12
|
+
*/
|
|
13
|
+
export interface CategoryStatusRow {
|
|
14
|
+
category: string;
|
|
15
|
+
/** Any ResultStatus value; only 'passed' / 'failed' / 'errored' are counted distinctly, everything else falls into the bar's `total` only (pending/running cases). */
|
|
16
|
+
status: string;
|
|
17
|
+
}
|
|
18
|
+
export interface CategoryBar {
|
|
19
|
+
category: string;
|
|
20
|
+
passed: number;
|
|
21
|
+
failed: number;
|
|
22
|
+
errored: number;
|
|
23
|
+
total: number;
|
|
24
|
+
}
|
|
25
|
+
/**
|
|
26
|
+
* Group rows by test-case category and tally pass/fail/errored/total.
|
|
27
|
+
* Deterministic order: largest category first, ties broken alphabetically.
|
|
28
|
+
*/
|
|
29
|
+
export declare function computeCategoryBars(rows: CategoryStatusRow[]): CategoryBar[];
|
|
30
|
+
/**
|
|
31
|
+
* Normalize a judge-reasoning string down to its first sentence,
|
|
32
|
+
* lowercased, punctuation-stripped, whitespace-collapsed. Used both as the
|
|
33
|
+
* clustering input and as the theme's stable `key`.
|
|
34
|
+
*/
|
|
35
|
+
export declare function normalizeReasoningKey(reasoning: string): string;
|
|
36
|
+
export interface FailureThemeInput {
|
|
37
|
+
testCaseId: string;
|
|
38
|
+
reasoning: string;
|
|
39
|
+
}
|
|
40
|
+
export interface FailureTheme {
|
|
41
|
+
/** Stable cluster key — the normalized first sentence of the theme's representative case. */
|
|
42
|
+
key: string;
|
|
43
|
+
count: number;
|
|
44
|
+
/** Trimmed, human-readable first sentence sampled from the theme's most common exact phrasing. */
|
|
45
|
+
sampleSnippet: string;
|
|
46
|
+
testCaseIds: string[];
|
|
47
|
+
}
|
|
48
|
+
/**
|
|
49
|
+
* Cluster failing test cases into "why they failed" themes using a
|
|
50
|
+
* deterministic, LLM-free heuristic: normalize each case's judge-reasoning
|
|
51
|
+
* first sentence, then union-find cases that share at least one contiguous
|
|
52
|
+
* N-word shingle. This is robust to minor paraphrasing (a judge saying
|
|
53
|
+
* "unable to retrieve" vs "failed to retrieve" the same underlying tool
|
|
54
|
+
* connectivity failure) while still keeping genuinely distinct failure
|
|
55
|
+
* modes (e.g. "missing required facts" vs "MCP server unavailable")
|
|
56
|
+
* separate, because they share no contiguous phrase.
|
|
57
|
+
*
|
|
58
|
+
* Verified against a real production run (418-verify, 64 failing cases):
|
|
59
|
+
* 57 of 64 connectivity-flavored reasonings collapse into ONE dominant
|
|
60
|
+
* theme; the remaining 7 ("Required facts evaluation: ...", a genuinely
|
|
61
|
+
* different failure shape) form a second, correctly separate theme.
|
|
62
|
+
*
|
|
63
|
+
* Output order: largest theme first, ties broken by the theme's
|
|
64
|
+
* lowest-sorting testCaseId (deterministic, no dependency on input order).
|
|
65
|
+
*/
|
|
66
|
+
export declare function clusterFailureThemes(items: FailureThemeInput[], shingleSize?: number): FailureTheme[];
|
|
67
|
+
/**
|
|
68
|
+
* "Based on N of M failing cases" note shown under the theme list when the
|
|
69
|
+
* reasoning fetch was capped (RunInsightsPane caps at the first 100 failing
|
|
70
|
+
* cases). Returns null when nothing was capped (fetchedCount >= totalCount).
|
|
71
|
+
*/
|
|
72
|
+
export declare function formatCappedNote(fetchedCount: number, totalFailingCount: number): string | null;
|
|
73
|
+
export interface RankedCase {
|
|
74
|
+
testCaseId: string;
|
|
75
|
+
value: number;
|
|
76
|
+
}
|
|
77
|
+
/**
|
|
78
|
+
* Top-N cases by a numeric value (duration, cost, ...), descending.
|
|
79
|
+
* Cases with a null/undefined/NaN value are excluded. Deterministic tie
|
|
80
|
+
* break: lower testCaseId first.
|
|
81
|
+
*/
|
|
82
|
+
export declare function pickTopN(cases: {
|
|
83
|
+
testCaseId: string;
|
|
84
|
+
value: number | null | undefined;
|
|
85
|
+
}[], n: number): RankedCase[];
|
|
86
|
+
//# sourceMappingURL=runInsights.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"runInsights.d.ts","sourceRoot":"","sources":["../../runInsights.ts"],"names":[],"mappings":"AAKA;;;;;;;;;;;GAWG;AAIH,MAAM,WAAW,iBAAiB;IAChC,QAAQ,EAAE,MAAM,CAAC;IACjB,sKAAsK;IACtK,MAAM,EAAE,MAAM,CAAC;CAChB;AAED,MAAM,WAAW,WAAW;IAC1B,QAAQ,EAAE,MAAM,CAAC;IACjB,MAAM,EAAE,MAAM,CAAC;IACf,MAAM,EAAE,MAAM,CAAC;IACf,OAAO,EAAE,MAAM,CAAC;IAChB,KAAK,EAAE,MAAM,CAAC;CACf;AAID;;;GAGG;AACH,wBAAgB,mBAAmB,CAAC,IAAI,EAAE,iBAAiB,EAAE,GAAG,WAAW,EAAE,CAiB5E;AAID;;;;GAIG;AACH,wBAAgB,qBAAqB,CAAC,SAAS,EAAE,MAAM,GAAG,MAAM,CAM/D;AAqBD,MAAM,WAAW,iBAAiB;IAChC,UAAU,EAAE,MAAM,CAAC;IACnB,SAAS,EAAE,MAAM,CAAC;CACnB;AAED,MAAM,WAAW,YAAY;IAC3B,6FAA6F;IAC7F,GAAG,EAAE,MAAM,CAAC;IACZ,KAAK,EAAE,MAAM,CAAC;IACd,kGAAkG;IAClG,aAAa,EAAE,MAAM,CAAC;IACtB,WAAW,EAAE,MAAM,EAAE,CAAC;CACvB;AAOD;;;;;;;;;;;;;;;;;GAiBG;AACH,wBAAgB,oBAAoB,CAClC,KAAK,EAAE,iBAAiB,EAAE,EAC1B,WAAW,GAAE,MAA6B,GACzC,YAAY,EAAE,CAwEhB;AAED;;;;GAIG;AACH,wBAAgB,gBAAgB,CAAC,YAAY,EAAE,MAAM,EAAE,iBAAiB,EAAE,MAAM,GAAG,MAAM,GAAG,IAAI,CAG/F;AAID,MAAM,WAAW,UAAU;IACzB,UAAU,EAAE,MAAM,CAAC;IACnB,KAAK,EAAE,MAAM,CAAC;CACf;AAED;;;;GAIG;AACH,wBAAgB,QAAQ,CAAC,KAAK,EAAE;IAAE,UAAU,EAAE,MAAM,CAAC;IAAC,KAAK,EAAE,MAAM,GAAG,IAAI,GAAG,SAAS,CAAA;CAAE,EAAE,EAAE,CAAC,EAAE,MAAM,GAAG,UAAU,EAAE,CAKnH"}
|
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Copyright OpenSearch Contributors
|
|
3
|
+
* SPDX-License-Identifier: Apache-2.0
|
|
4
|
+
*/
|
|
5
|
+
const UNCATEGORIZED = 'Uncategorized';
|
|
6
|
+
/**
|
|
7
|
+
* Group rows by test-case category and tally pass/fail/errored/total.
|
|
8
|
+
* Deterministic order: largest category first, ties broken alphabetically.
|
|
9
|
+
*/
|
|
10
|
+
export function computeCategoryBars(rows) {
|
|
11
|
+
const map = new Map();
|
|
12
|
+
for (const row of rows) {
|
|
13
|
+
const category = (row.category || '').trim() || UNCATEGORIZED;
|
|
14
|
+
let bar = map.get(category);
|
|
15
|
+
if (!bar) {
|
|
16
|
+
bar = { category, passed: 0, failed: 0, errored: 0, total: 0 };
|
|
17
|
+
map.set(category, bar);
|
|
18
|
+
}
|
|
19
|
+
bar.total++;
|
|
20
|
+
if (row.status === 'passed')
|
|
21
|
+
bar.passed++;
|
|
22
|
+
else if (row.status === 'errored')
|
|
23
|
+
bar.errored++;
|
|
24
|
+
else if (row.status === 'failed')
|
|
25
|
+
bar.failed++;
|
|
26
|
+
}
|
|
27
|
+
return Array.from(map.values()).sort((a, b) => b.total - a.total || a.category.localeCompare(b.category));
|
|
28
|
+
}
|
|
29
|
+
// ── Failure-theme clustering ───────────────────────────────────────────────
|
|
30
|
+
/**
|
|
31
|
+
* Normalize a judge-reasoning string down to its first sentence,
|
|
32
|
+
* lowercased, punctuation-stripped, whitespace-collapsed. Used both as the
|
|
33
|
+
* clustering input and as the theme's stable `key`.
|
|
34
|
+
*/
|
|
35
|
+
export function normalizeReasoningKey(reasoning) {
|
|
36
|
+
const trimmed = (reasoning || '').trim();
|
|
37
|
+
if (!trimmed)
|
|
38
|
+
return '';
|
|
39
|
+
const sentenceMatch = trimmed.match(/^[^.!?\n]+[.!?]?/);
|
|
40
|
+
const firstSentence = (sentenceMatch ? sentenceMatch[0] : trimmed).toLowerCase();
|
|
41
|
+
return firstSentence.replace(/[^a-z0-9\s]/g, ' ').replace(/\s+/g, ' ').trim();
|
|
42
|
+
}
|
|
43
|
+
/** Returns the literal (non-lowercased) first sentence, trimmed — used as the human-facing sample snippet. */
|
|
44
|
+
function literalFirstSentence(reasoning) {
|
|
45
|
+
const trimmed = (reasoning || '').trim();
|
|
46
|
+
if (!trimmed)
|
|
47
|
+
return '';
|
|
48
|
+
const sentenceMatch = trimmed.match(/^[^.!?\n]+[.!?]?/);
|
|
49
|
+
return (sentenceMatch ? sentenceMatch[0] : trimmed).trim();
|
|
50
|
+
}
|
|
51
|
+
function shingles(normalized, size) {
|
|
52
|
+
const words = normalized.split(' ').filter(Boolean);
|
|
53
|
+
if (words.length === 0)
|
|
54
|
+
return [];
|
|
55
|
+
if (words.length <= size)
|
|
56
|
+
return [words.join(' ')];
|
|
57
|
+
const out = [];
|
|
58
|
+
for (let i = 0; i <= words.length - size; i++) {
|
|
59
|
+
out.push(words.slice(i, i + size).join(' '));
|
|
60
|
+
}
|
|
61
|
+
return out;
|
|
62
|
+
}
|
|
63
|
+
/** Contiguous-word window size used for near-duplicate matching. Paraphrases of the same failure ("agent was unable to retrieve..." vs "agent failed to retrieve...") reliably share at least one 6-word run even when their leading words differ. */
|
|
64
|
+
const DEFAULT_SHINGLE_SIZE = 6;
|
|
65
|
+
/** Reasoning first-sentences shorter than this many words don't carry enough signal to safely dedupe against unrelated failures — left as their own singleton theme rather than risk clustering unrelated short reasons together (e.g. "The agent failed to retrieve the required information."). */
|
|
66
|
+
const MIN_WORDS_FOR_SHINGLING = 3;
|
|
67
|
+
/**
|
|
68
|
+
* Cluster failing test cases into "why they failed" themes using a
|
|
69
|
+
* deterministic, LLM-free heuristic: normalize each case's judge-reasoning
|
|
70
|
+
* first sentence, then union-find cases that share at least one contiguous
|
|
71
|
+
* N-word shingle. This is robust to minor paraphrasing (a judge saying
|
|
72
|
+
* "unable to retrieve" vs "failed to retrieve" the same underlying tool
|
|
73
|
+
* connectivity failure) while still keeping genuinely distinct failure
|
|
74
|
+
* modes (e.g. "missing required facts" vs "MCP server unavailable")
|
|
75
|
+
* separate, because they share no contiguous phrase.
|
|
76
|
+
*
|
|
77
|
+
* Verified against a real production run (418-verify, 64 failing cases):
|
|
78
|
+
* 57 of 64 connectivity-flavored reasonings collapse into ONE dominant
|
|
79
|
+
* theme; the remaining 7 ("Required facts evaluation: ...", a genuinely
|
|
80
|
+
* different failure shape) form a second, correctly separate theme.
|
|
81
|
+
*
|
|
82
|
+
* Output order: largest theme first, ties broken by the theme's
|
|
83
|
+
* lowest-sorting testCaseId (deterministic, no dependency on input order).
|
|
84
|
+
*/
|
|
85
|
+
export function clusterFailureThemes(items, shingleSize = DEFAULT_SHINGLE_SIZE) {
|
|
86
|
+
const n = items.length;
|
|
87
|
+
if (n === 0)
|
|
88
|
+
return [];
|
|
89
|
+
const parent = Array.from({ length: n }, (_, i) => i);
|
|
90
|
+
function find(i) {
|
|
91
|
+
while (parent[i] !== i) {
|
|
92
|
+
parent[i] = parent[parent[i]];
|
|
93
|
+
i = parent[i];
|
|
94
|
+
}
|
|
95
|
+
return i;
|
|
96
|
+
}
|
|
97
|
+
function union(a, b) {
|
|
98
|
+
const ra = find(a);
|
|
99
|
+
const rb = find(b);
|
|
100
|
+
if (ra !== rb)
|
|
101
|
+
parent[Math.max(ra, rb)] = Math.min(ra, rb);
|
|
102
|
+
}
|
|
103
|
+
const normalized = items.map(it => normalizeReasoningKey(it.reasoning));
|
|
104
|
+
const shingleBuckets = new Map(); // shingle -> first index seen
|
|
105
|
+
normalized.forEach((norm, idx) => {
|
|
106
|
+
const words = norm.split(' ').filter(Boolean);
|
|
107
|
+
if (words.length < MIN_WORDS_FOR_SHINGLING)
|
|
108
|
+
return;
|
|
109
|
+
for (const sh of shingles(norm, shingleSize)) {
|
|
110
|
+
const firstIdx = shingleBuckets.get(sh);
|
|
111
|
+
if (firstIdx === undefined)
|
|
112
|
+
shingleBuckets.set(sh, idx);
|
|
113
|
+
else
|
|
114
|
+
union(firstIdx, idx);
|
|
115
|
+
}
|
|
116
|
+
});
|
|
117
|
+
const groups = new Map();
|
|
118
|
+
for (let i = 0; i < n; i++) {
|
|
119
|
+
const root = find(i);
|
|
120
|
+
const arr = groups.get(root);
|
|
121
|
+
if (arr)
|
|
122
|
+
arr.push(i);
|
|
123
|
+
else
|
|
124
|
+
groups.set(root, [i]);
|
|
125
|
+
}
|
|
126
|
+
const themes = [];
|
|
127
|
+
for (const idxs of groups.values()) {
|
|
128
|
+
// Representative snippet: the most frequent literal first-sentence
|
|
129
|
+
// phrasing within the cluster (ties broken by first occurrence) - so a
|
|
130
|
+
// theme with 4 identical sentences + a handful of near-duplicate
|
|
131
|
+
// paraphrases surfaces the exact sentence, not an arbitrary member.
|
|
132
|
+
const counts = new Map();
|
|
133
|
+
for (const idx of idxs) {
|
|
134
|
+
const key = normalized[idx];
|
|
135
|
+
const snippet = literalFirstSentence(items[idx].reasoning);
|
|
136
|
+
const existing = counts.get(key);
|
|
137
|
+
if (existing)
|
|
138
|
+
existing.count++;
|
|
139
|
+
else
|
|
140
|
+
counts.set(key, { count: 1, firstIdx: idx, snippet });
|
|
141
|
+
}
|
|
142
|
+
let best = null;
|
|
143
|
+
for (const v of counts.values()) {
|
|
144
|
+
if (!best || v.count > best.count || (v.count === best.count && v.firstIdx < best.firstIdx))
|
|
145
|
+
best = v;
|
|
146
|
+
}
|
|
147
|
+
const testCaseIds = idxs.map(idx => items[idx].testCaseId);
|
|
148
|
+
themes.push({
|
|
149
|
+
key: normalized[idxs[0]] || `theme-${idxs[0]}`,
|
|
150
|
+
count: idxs.length,
|
|
151
|
+
sampleSnippet: best?.snippet || '',
|
|
152
|
+
testCaseIds,
|
|
153
|
+
});
|
|
154
|
+
}
|
|
155
|
+
themes.sort((a, b) => {
|
|
156
|
+
if (b.count !== a.count)
|
|
157
|
+
return b.count - a.count;
|
|
158
|
+
const aMin = [...a.testCaseIds].sort()[0] || '';
|
|
159
|
+
const bMin = [...b.testCaseIds].sort()[0] || '';
|
|
160
|
+
return aMin < bMin ? -1 : aMin > bMin ? 1 : 0;
|
|
161
|
+
});
|
|
162
|
+
return themes;
|
|
163
|
+
}
|
|
164
|
+
/**
|
|
165
|
+
* "Based on N of M failing cases" note shown under the theme list when the
|
|
166
|
+
* reasoning fetch was capped (RunInsightsPane caps at the first 100 failing
|
|
167
|
+
* cases). Returns null when nothing was capped (fetchedCount >= totalCount).
|
|
168
|
+
*/
|
|
169
|
+
export function formatCappedNote(fetchedCount, totalFailingCount) {
|
|
170
|
+
if (totalFailingCount <= fetchedCount)
|
|
171
|
+
return null;
|
|
172
|
+
return `Based on ${fetchedCount} of ${totalFailingCount} failing cases`;
|
|
173
|
+
}
|
|
174
|
+
/**
|
|
175
|
+
* Top-N cases by a numeric value (duration, cost, ...), descending.
|
|
176
|
+
* Cases with a null/undefined/NaN value are excluded. Deterministic tie
|
|
177
|
+
* break: lower testCaseId first.
|
|
178
|
+
*/
|
|
179
|
+
export function pickTopN(cases, n) {
|
|
180
|
+
return cases
|
|
181
|
+
.filter((c) => typeof c.value === 'number' && Number.isFinite(c.value))
|
|
182
|
+
.sort((a, b) => (b.value !== a.value ? b.value - a.value : a.testCaseId.localeCompare(b.testCaseId)))
|
|
183
|
+
.slice(0, n);
|
|
184
|
+
}
|
|
185
|
+
//# sourceMappingURL=runInsights.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"runInsights.js","sourceRoot":"","sources":["../../runInsights.ts"],"names":[],"mappings":"AAAA;;;GAGG;AA+BH,MAAM,aAAa,GAAG,eAAe,CAAC;AAEtC;;;GAGG;AACH,MAAM,UAAU,mBAAmB,CAAC,IAAyB;IAC3D,MAAM,GAAG,GAAG,IAAI,GAAG,EAAuB,CAAC;IAC3C,KAAK,MAAM,GAAG,IAAI,IAAI,EAAE,CAAC;QACvB,MAAM,QAAQ,GAAG,CAAC,GAAG,CAAC,QAAQ,IAAI,EAAE,CAAC,CAAC,IAAI,EAAE,IAAI,aAAa,CAAC;QAC9D,IAAI,GAAG,GAAG,GAAG,CAAC,GAAG,CAAC,QAAQ,CAAC,CAAC;QAC5B,IAAI,CAAC,GAAG,EAAE,CAAC;YACT,GAAG,GAAG,EAAE,QAAQ,EAAE,MAAM,EAAE,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,OAAO,EAAE,CAAC,EAAE,KAAK,EAAE,CAAC,EAAE,CAAC;YAC/D,GAAG,CAAC,GAAG,CAAC,QAAQ,EAAE,GAAG,CAAC,CAAC;QACzB,CAAC;QACD,GAAG,CAAC,KAAK,EAAE,CAAC;QACZ,IAAI,GAAG,CAAC,MAAM,KAAK,QAAQ;YAAE,GAAG,CAAC,MAAM,EAAE,CAAC;aACrC,IAAI,GAAG,CAAC,MAAM,KAAK,SAAS;YAAE,GAAG,CAAC,OAAO,EAAE,CAAC;aAC5C,IAAI,GAAG,CAAC,MAAM,KAAK,QAAQ;YAAE,GAAG,CAAC,MAAM,EAAE,CAAC;IACjD,CAAC;IACD,OAAO,KAAK,CAAC,IAAI,CAAC,GAAG,CAAC,MAAM,EAAE,CAAC,CAAC,IAAI,CAClC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,KAAK,GAAG,CAAC,CAAC,KAAK,IAAI,CAAC,CAAC,QAAQ,CAAC,aAAa,CAAC,CAAC,CAAC,QAAQ,CAAC,CACpE,CAAC;AACJ,CAAC;AAED,8EAA8E;AAE9E;;;;GAIG;AACH,MAAM,UAAU,qBAAqB,CAAC,SAAiB;IACrD,MAAM,OAAO,GAAG,CAAC,SAAS,IAAI,EAAE,CAAC,CAAC,IAAI,EAAE,CAAC;IACzC,IAAI,CAAC,OAAO;QAAE,OAAO,EAAE,CAAC;IACxB,MAAM,aAAa,GAAG,OAAO,CAAC,KAAK,CAAC,kBAAkB,CAAC,CAAC;IACxD,MAAM,aAAa,GAAG,CAAC,aAAa,CAAC,CAAC,CAAC,aAAa,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,WAAW,EAAE,CAAC;IACjF,OAAO,aAAa,CAAC,OAAO,CAAC,cAAc,EAAE,GAAG,CAAC,CAAC,OAAO,CAAC,MAAM,EAAE,GAAG,CAAC,CAAC,IAAI,EAAE,CAAC;AAChF,CAAC;AAED,8GAA8G;AAC9G,SAAS,oBAAoB,CAAC,SAAiB;IAC7C,MAAM,OAAO,GAAG,CAAC,SAAS,IAAI,EAAE,CAAC,CAAC,IAAI,EAAE,CAAC;IACzC,IAAI,CAAC,OAAO;QAAE,OAAO,EAAE,CAAC;IACxB,MAAM,aAAa,GAAG,OAAO,CAAC,KAAK,CAAC,kBAAkB,CAAC,CAAC;IACxD,OAAO,CAAC,aAAa,CAAC,CAAC,CAAC,aAAa,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,IAAI,EAAE,CAAC;AAC7D,CAAC;AAED,SAAS,QAAQ,CAAC,UAAkB,EAAE,IAAY;IAChD,MAAM,KAAK,GAAG,UAAU,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,MAAM,CAAC,OAAO,CAAC,CAAC;IACpD,IAAI,KAAK,CAAC,MAAM,KAAK,CAAC;QAAE,OAAO,EAAE,CAAC;IAClC,IAAI,KAAK,CAAC,MAAM,IAAI,IAAI;QAAE,OAAO,CAAC,KAAK,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,CAAC;IACnD,MAAM,GAAG,GAAa,EAAE,CAAC;IACzB,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,IAAI,KAAK,CAAC,MAAM,GAAG,IAAI,EAAE,CAAC,EAAE,EAAE,CAAC;QAC9C,GAAG,CAAC,IAAI,CAAC,KAAK,CAAC,KAAK,CAAC,CAAC,EAAE,CAAC,GAAG,IAAI,CAAC,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,CAAC;IAC/C,CAAC;IACD,OAAO,GAAG,CAAC;AACb,CAAC;AAgBD,sPAAsP;AACtP,MAAM,oBAAoB,GAAG,CAAC,CAAC;AAC/B,qSAAqS;AACrS,MAAM,uBAAuB,GAAG,CAAC,CAAC;AAElC;;;;;;;;;;;;;;;;;GAiBG;AACH,MAAM,UAAU,oBAAoB,CAClC,KAA0B,EAC1B,cAAsB,oBAAoB;IAE1C,MAAM,CAAC,GAAG,KAAK,CAAC,MAAM,CAAC;IACvB,IAAI,CAAC,KAAK,CAAC;QAAE,OAAO,EAAE,CAAC;IAEvB,MAAM,MAAM,GAAG,KAAK,CAAC,IAAI,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,CAAC;IACtD,SAAS,IAAI,CAAC,CAAS;QACrB,OAAO,MAAM,CAAC,CAAC,CAAC,KAAK,CAAC,EAAE,CAAC;YACvB,MAAM,CAAC,CAAC,CAAC,GAAG,MAAM,CAAC,MAAM,CAAC,CAAC,CAAC,CAAC,CAAC;YAC9B,CAAC,GAAG,MAAM,CAAC,CAAC,CAAC,CAAC;QAChB,CAAC;QACD,OAAO,CAAC,CAAC;IACX,CAAC;IACD,SAAS,KAAK,CAAC,CAAS,EAAE,CAAS;QACjC,MAAM,EAAE,GAAG,IAAI,CAAC,CAAC,CAAC,CAAC;QACnB,MAAM,EAAE,GAAG,IAAI,CAAC,CAAC,CAAC,CAAC;QACnB,IAAI,EAAE,KAAK,EAAE;YAAE,MAAM,CAAC,IAAI,CAAC,GAAG,CAAC,EAAE,EAAE,EAAE,CAAC,CAAC,GAAG,IAAI,CAAC,GAAG,CAAC,EAAE,EAAE,EAAE,CAAC,CAAC;IAC7D,CAAC;IAED,MAAM,UAAU,GAAG,KAAK,CAAC,GAAG,CAAC,EAAE,CAAC,EAAE,CAAC,qBAAqB,CAAC,EAAE,CAAC,SAAS,CAAC,CAAC,CAAC;IACxE,MAAM,cAAc,GAAG,IAAI,GAAG,EAAkB,CAAC,CAAC,8BAA8B;IAChF,UAAU,CAAC,OAAO,CAAC,CAAC,IAAI,EAAE,GAAG,EAAE,EAAE;QAC/B,MAAM,KAAK,GAAG,IAAI,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,MAAM,CAAC,OAAO,CAAC,CAAC;QAC9C,IAAI,KAAK,CAAC,MAAM,GAAG,uBAAuB;YAAE,OAAO;QACnD,KAAK,MAAM,EAAE,IAAI,QAAQ,CAAC,IAAI,EAAE,WAAW,CAAC,EAAE,CAAC;YAC7C,MAAM,QAAQ,GAAG,cAAc,CAAC,GAAG,CAAC,EAAE,CAAC,CAAC;YACxC,IAAI,QAAQ,KAAK,SAAS;gBAAE,cAAc,CAAC,GAAG,CAAC,EAAE,EAAE,GAAG,CAAC,CAAC;;gBACnD,KAAK,CAAC,QAAQ,EAAE,GAAG,CAAC,CAAC;QAC5B,CAAC;IACH,CAAC,CAAC,CAAC;IAEH,MAAM,MAAM,GAAG,IAAI,GAAG,EAAoB,CAAC;IAC3C,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC;QAC3B,MAAM,IAAI,GAAG,IAAI,CAAC,CAAC,CAAC,CAAC;QACrB,MAAM,GAAG,GAAG,MAAM,CAAC,GAAG,CAAC,IAAI,CAAC,CAAC;QAC7B,IAAI,GAAG;YAAE,GAAG,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC;;YAChB,MAAM,CAAC,GAAG,CAAC,IAAI,EAAE,CAAC,CAAC,CAAC,CAAC,CAAC;IAC7B,CAAC;IAED,MAAM,MAAM,GAAmB,EAAE,CAAC;IAClC,KAAK,MAAM,IAAI,IAAI,MAAM,CAAC,MAAM,EAAE,EAAE,CAAC;QACnC,mEAAmE;QACnE,uEAAuE;QACvE,iEAAiE;QACjE,oEAAoE;QACpE,MAAM,MAAM,GAAG,IAAI,GAAG,EAAgE,CAAC;QACvF,KAAK,MAAM,GAAG,IAAI,IAAI,EAAE,CAAC;YACvB,MAAM,GAAG,GAAG,UAAU,CAAC,GAAG,CAAC,CAAC;YAC5B,MAAM,OAAO,GAAG,oBAAoB,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,SAAS,CAAC,CAAC;YAC3D,MAAM,QAAQ,GAAG,MAAM,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC;YACjC,IAAI,QAAQ;gBAAE,QAAQ,CAAC,KAAK,EAAE,CAAC;;gBAC1B,MAAM,CAAC,GAAG,CAAC,GAAG,EAAE,EAAE,KAAK,EAAE,CAAC,EAAE,QAAQ,EAAE,GAAG,EAAE,OAAO,EAAE,CAAC,CAAC;QAC7D,CAAC;QACD,IAAI,IAAI,GAAgE,IAAI,CAAC;QAC7E,KAAK,MAAM,CAAC,IAAI,MAAM,CAAC,MAAM,EAAE,EAAE,CAAC;YAChC,IAAI,CAAC,IAAI,IAAI,CAAC,CAAC,KAAK,GAAG,IAAI,CAAC,KAAK,IAAI,CAAC,CAAC,CAAC,KAAK,KAAK,IAAI,CAAC,KAAK,IAAI,CAAC,CAAC,QAAQ,GAAG,IAAI,CAAC,QAAQ,CAAC;gBAAE,IAAI,GAAG,CAAC,CAAC;QACxG,CAAC;QACD,MAAM,WAAW,GAAG,IAAI,CAAC,GAAG,CAAC,GAAG,CAAC,EAAE,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,UAAU,CAAC,CAAC;QAC3D,MAAM,CAAC,IAAI,CAAC;YACV,GAAG,EAAE,UAAU,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC,IAAI,SAAS,IAAI,CAAC,CAAC,CAAC,EAAE;YAC9C,KAAK,EAAE,IAAI,CAAC,MAAM;YAClB,aAAa,EAAE,IAAI,EAAE,OAAO,IAAI,EAAE;YAClC,WAAW;SACZ,CAAC,CAAC;IACL,CAAC;IAED,MAAM,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE;QACnB,IAAI,CAAC,CAAC,KAAK,KAAK,CAAC,CAAC,KAAK;YAAE,OAAO,CAAC,CAAC,KAAK,GAAG,CAAC,CAAC,KAAK,CAAC;QAClD,MAAM,IAAI,GAAG,CAAC,GAAG,CAAC,CAAC,WAAW,CAAC,CAAC,IAAI,EAAE,CAAC,CAAC,CAAC,IAAI,EAAE,CAAC;QAChD,MAAM,IAAI,GAAG,CAAC,GAAG,CAAC,CAAC,WAAW,CAAC,CAAC,IAAI,EAAE,CAAC,CAAC,CAAC,IAAI,EAAE,CAAC;QAChD,OAAO,IAAI,GAAG,IAAI,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,IAAI,GAAG,IAAI,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;IAChD,CAAC,CAAC,CAAC;IACH,OAAO,MAAM,CAAC;AAChB,CAAC;AAED;;;;GAIG;AACH,MAAM,UAAU,gBAAgB,CAAC,YAAoB,EAAE,iBAAyB;IAC9E,IAAI,iBAAiB,IAAI,YAAY;QAAE,OAAO,IAAI,CAAC;IACnD,OAAO,YAAY,YAAY,OAAO,iBAAiB,gBAAgB,CAAC;AAC1E,CAAC;AASD;;;;GAIG;AACH,MAAM,UAAU,QAAQ,CAAC,KAAiE,EAAE,CAAS;IACnG,OAAO,KAAK;SACT,MAAM,CAAC,CAAC,CAAC,EAAmB,EAAE,CAAC,OAAO,CAAC,CAAC,KAAK,KAAK,QAAQ,IAAI,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC;SACvF,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,CAAC,KAAK,KAAK,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC,KAAK,GAAG,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC,UAAU,CAAC,aAAa,CAAC,CAAC,CAAC,UAAU,CAAC,CAAC,CAAC;SACpG,KAAK,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC;AACjB,CAAC"}
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Validation for renaming an evaluation run (PATCH /api/storage/evaluation-runs/:id
|
|
3
|
+
* `{ name }`). This is the server route's authoritative check.
|
|
4
|
+
*
|
|
5
|
+
* Rename is intentionally narrow: it only ever touches `name`. No version bump,
|
|
6
|
+
* no stats recompute — this file has no opinion on anything but the string.
|
|
7
|
+
*/
|
|
8
|
+
/** Renamed evaluation runs are capped at this length (arbitrary but generous —
|
|
9
|
+
* long enough for any real run name, short enough to keep list/table UIs sane). */
|
|
10
|
+
export declare const MAX_RUN_NAME_LENGTH = 200;
|
|
11
|
+
export type RunNameValidation = {
|
|
12
|
+
ok: true;
|
|
13
|
+
value: string;
|
|
14
|
+
} | {
|
|
15
|
+
ok: false;
|
|
16
|
+
error: string;
|
|
17
|
+
};
|
|
18
|
+
/**
|
|
19
|
+
* Validate + normalize a candidate evaluation-run name.
|
|
20
|
+
*
|
|
21
|
+
* - Must be a string.
|
|
22
|
+
* - Trimmed value must be non-empty.
|
|
23
|
+
* - Trimmed value must be at most {@link MAX_RUN_NAME_LENGTH} characters.
|
|
24
|
+
*
|
|
25
|
+
* Returns the trimmed value on success so callers persist a normalized name
|
|
26
|
+
* (no leading/trailing whitespace) rather than the raw input.
|
|
27
|
+
*/
|
|
28
|
+
export declare function validateRunNameUpdate(name: unknown): RunNameValidation;
|
|
29
|
+
//# sourceMappingURL=runName.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"runName.d.ts","sourceRoot":"","sources":["../../runName.ts"],"names":[],"mappings":"AAKA;;;;;;GAMG;AAEH;oFACoF;AACpF,eAAO,MAAM,mBAAmB,MAAM,CAAC;AAEvC,MAAM,MAAM,iBAAiB,GACzB;IAAE,EAAE,EAAE,IAAI,CAAC;IAAC,KAAK,EAAE,MAAM,CAAA;CAAE,GAC3B;IAAE,EAAE,EAAE,KAAK,CAAC;IAAC,KAAK,EAAE,MAAM,CAAA;CAAE,CAAC;AAEjC;;;;;;;;;GASG;AACH,wBAAgB,qBAAqB,CAAC,IAAI,EAAE,OAAO,GAAG,iBAAiB,CAYtE"}
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Copyright OpenSearch Contributors
|
|
3
|
+
* SPDX-License-Identifier: Apache-2.0
|
|
4
|
+
*/
|
|
5
|
+
/**
|
|
6
|
+
* Validation for renaming an evaluation run (PATCH /api/storage/evaluation-runs/:id
|
|
7
|
+
* `{ name }`). This is the server route's authoritative check.
|
|
8
|
+
*
|
|
9
|
+
* Rename is intentionally narrow: it only ever touches `name`. No version bump,
|
|
10
|
+
* no stats recompute — this file has no opinion on anything but the string.
|
|
11
|
+
*/
|
|
12
|
+
/** Renamed evaluation runs are capped at this length (arbitrary but generous —
|
|
13
|
+
* long enough for any real run name, short enough to keep list/table UIs sane). */
|
|
14
|
+
export const MAX_RUN_NAME_LENGTH = 200;
|
|
15
|
+
/**
|
|
16
|
+
* Validate + normalize a candidate evaluation-run name.
|
|
17
|
+
*
|
|
18
|
+
* - Must be a string.
|
|
19
|
+
* - Trimmed value must be non-empty.
|
|
20
|
+
* - Trimmed value must be at most {@link MAX_RUN_NAME_LENGTH} characters.
|
|
21
|
+
*
|
|
22
|
+
* Returns the trimmed value on success so callers persist a normalized name
|
|
23
|
+
* (no leading/trailing whitespace) rather than the raw input.
|
|
24
|
+
*/
|
|
25
|
+
export function validateRunNameUpdate(name) {
|
|
26
|
+
if (typeof name !== 'string') {
|
|
27
|
+
return { ok: false, error: 'name must be a string' };
|
|
28
|
+
}
|
|
29
|
+
const trimmed = name.trim();
|
|
30
|
+
if (!trimmed) {
|
|
31
|
+
return { ok: false, error: 'name must not be empty' };
|
|
32
|
+
}
|
|
33
|
+
if (trimmed.length > MAX_RUN_NAME_LENGTH) {
|
|
34
|
+
return { ok: false, error: `name must be ${MAX_RUN_NAME_LENGTH} characters or fewer` };
|
|
35
|
+
}
|
|
36
|
+
return { ok: true, value: trimmed };
|
|
37
|
+
}
|
|
38
|
+
//# sourceMappingURL=runName.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"runName.js","sourceRoot":"","sources":["../../runName.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAEH;;;;;;GAMG;AAEH;oFACoF;AACpF,MAAM,CAAC,MAAM,mBAAmB,GAAG,GAAG,CAAC;AAMvC;;;;;;;;;GASG;AACH,MAAM,UAAU,qBAAqB,CAAC,IAAa;IACjD,IAAI,OAAO,IAAI,KAAK,QAAQ,EAAE,CAAC;QAC7B,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,uBAAuB,EAAE,CAAC;IACvD,CAAC;IACD,MAAM,OAAO,GAAG,IAAI,CAAC,IAAI,EAAE,CAAC;IAC5B,IAAI,CAAC,OAAO,EAAE,CAAC;QACb,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,wBAAwB,EAAE,CAAC;IACxD,CAAC;IACD,IAAI,OAAO,CAAC,MAAM,GAAG,mBAAmB,EAAE,CAAC;QACzC,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,gBAAgB,mBAAmB,sBAAsB,EAAE,CAAC;IACzF,CAAC;IACD,OAAO,EAAE,EAAE,EAAE,IAAI,EAAE,KAAK,EAAE,OAAO,EAAE,CAAC;AACtC,CAAC"}
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Canonical run-report path for a run surfaced on the compare page.
|
|
3
|
+
*
|
|
4
|
+
* Benchmark runs (embedded in a benchmark's `runs[]`) resolve at
|
|
5
|
+
* `/evaluations/benchmarks/:benchmarkId/runs/:runId`; everything else —
|
|
6
|
+
* ad-hoc / SDK eval-runs, or runs whose `benchmarkId` is merely a label — goes
|
|
7
|
+
* to the bare `/evaluations/runs/:runId` route (the benchmark route would
|
|
8
|
+
* 404/redirect for those, and the bare route 404s for benchmark run ids).
|
|
9
|
+
*
|
|
10
|
+
* Shared by the scoreboard's run-name link, its "Open run" icon, and the
|
|
11
|
+
* per-case table's run headers so the three can never disagree. Callers pass
|
|
12
|
+
* the benchmarkId from `ComparisonPage`'s `runBenchmarkIdById` map (which is
|
|
13
|
+
* `undefined` unless the run is a real benchmark member).
|
|
14
|
+
*/
|
|
15
|
+
export declare const runReportPath: (runId: string, benchmarkId?: string) => string;
|
|
16
|
+
//# sourceMappingURL=runReportPath.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"runReportPath.d.ts","sourceRoot":"","sources":["../../runReportPath.ts"],"names":[],"mappings":"AAKA;;;;;;;;;;;;;GAaG;AACH,eAAO,MAAM,aAAa,GAAI,OAAO,MAAM,EAAE,cAAc,MAAM,KAAG,MAGd,CAAC"}
|