@datalayer/core 1.1.66 → 1.2.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +43 -20
- package/lib/api/constants.d.ts +1 -1
- package/lib/api/constants.js +1 -1
- package/lib/api/contents/attachments.js +3 -7
- package/lib/api/contents/datasets.d.ts +48 -0
- package/lib/api/contents/datasets.js +44 -7
- package/lib/api/contents/runtimeMounts.js +1 -1
- package/lib/api/contents/sandboxUid.d.ts +13 -45
- package/lib/api/contents/sandboxUid.js +14 -65
- package/lib/api/evals/client.d.ts +402 -0
- package/lib/api/evals/client.js +359 -0
- package/lib/api/evals/index.d.ts +10 -0
- package/lib/api/evals/index.js +18 -0
- package/lib/api/evals/request.d.ts +43 -0
- package/lib/api/evals/request.js +67 -0
- package/lib/api/evals/status.d.ts +27 -0
- package/lib/api/evals/status.js +122 -0
- package/lib/api/evals/types.d.ts +1335 -0
- package/lib/api/evals/types.js +5 -0
- package/lib/api/iam/authentication.js +6 -4
- package/lib/api/iam/connectedAgents.d.ts +29 -0
- package/lib/api/iam/connectedAgents.js +35 -0
- package/lib/api/iam/healthz.js +2 -2
- package/lib/api/iam/identityProviders.d.ts +76 -0
- package/lib/api/iam/identityProviders.js +187 -0
- package/lib/api/iam/index.d.ts +1 -0
- package/lib/api/iam/index.js +1 -0
- package/lib/api/iam/mcpPolicy.d.ts +102 -1
- package/lib/api/iam/mcpPolicy.js +38 -2
- package/lib/api/iam/oauth2.js +18 -12
- package/lib/api/iam/profile.d.ts +17 -1
- package/lib/api/iam/profile.js +27 -0
- package/lib/api/iam/scimTokens.d.ts +79 -0
- package/lib/api/iam/scimTokens.js +141 -0
- package/lib/api/iam/secrets.js +10 -23
- package/lib/api/iam/trials.d.ts +38 -0
- package/lib/api/iam/trials.js +66 -0
- package/lib/api/index.d.ts +4 -0
- package/lib/api/index.js +4 -0
- package/lib/api/mcp/observability.d.ts +34 -11
- package/lib/api/mcp/observability.js +97 -59
- package/lib/api/orchestration/client.d.ts +53 -0
- package/lib/api/orchestration/client.js +105 -0
- package/lib/api/orchestration/events.d.ts +88 -0
- package/lib/api/orchestration/events.js +245 -0
- package/lib/api/orchestration/generated.d.ts +693 -0
- package/lib/api/orchestration/generated.js +687 -0
- package/lib/api/orchestration/index.d.ts +30 -0
- package/lib/api/orchestration/index.js +34 -0
- package/lib/api/orchestration/lifecycle.d.ts +39 -0
- package/lib/api/orchestration/lifecycle.js +57 -0
- package/lib/api/orchestration/measures.d.ts +55 -0
- package/lib/api/orchestration/measures.js +99 -0
- package/lib/api/otel/dashboards.d.ts +56 -0
- package/lib/api/otel/dashboards.js +50 -0
- package/lib/api/otel/index.d.ts +1 -0
- package/lib/api/otel/index.js +1 -0
- package/lib/api/otel/metrics.d.ts +16 -0
- package/lib/api/otel/metrics.js +15 -0
- package/lib/api/scheduler/client.d.ts +32 -0
- package/lib/api/scheduler/client.js +42 -0
- package/lib/api/scheduler/index.d.ts +8 -0
- package/lib/api/scheduler/index.js +16 -0
- package/lib/api/scheduler/request.d.ts +28 -0
- package/lib/api/scheduler/request.js +50 -0
- package/lib/api/scheduler/types.d.ts +88 -0
- package/lib/api/scheduler/types.js +5 -0
- package/lib/api/spacer/comments.d.ts +97 -0
- package/lib/api/spacer/comments.js +45 -0
- package/lib/api/spacer/index.d.ts +9 -0
- package/lib/api/spacer/index.js +17 -0
- package/lib/api/spacer/notebooks.d.ts +33 -0
- package/lib/api/spacer/notebooks.js +39 -0
- package/lib/api/spacer/request.d.ts +25 -0
- package/lib/api/spacer/request.js +46 -0
- package/lib/api/spacer/spaces.d.ts +68 -0
- package/lib/api/spacer/spaces.js +45 -0
- package/lib/api/utils/validation.d.ts +4 -4
- package/lib/api/utils/validation.js +10 -7
- package/lib/client/auth/strategies.js +1 -1
- package/lib/client/constants.d.ts +1 -0
- package/lib/client/constants.js +1 -0
- package/lib/client/mixins/IAMMixin.js +2 -2
- package/lib/components/animation/AnimatedText.d.ts +1 -44
- package/lib/components/animation/AnimatedText.js +6 -122
- package/lib/components/anonymous/AnonymousKeyExpired.d.ts +56 -0
- package/lib/components/anonymous/AnonymousKeyExpired.js +96 -0
- package/lib/components/anonymous/AnonymousKeyTimer.d.ts +63 -0
- package/lib/components/anonymous/AnonymousKeyTimer.js +140 -0
- package/lib/components/anonymous/index.d.ts +7 -0
- package/lib/components/anonymous/index.js +15 -0
- package/lib/components/avatars/UserAvatar.d.ts +7 -1
- package/lib/components/avatars/UserAvatar.js +15 -1
- package/lib/components/billing/BillingEntitySelect.d.ts +6 -0
- package/lib/components/billing/BillingEntitySelect.js +49 -10
- package/lib/components/billing/eligibility.d.ts +22 -0
- package/lib/components/billing/eligibility.js +27 -0
- package/lib/components/billing/index.d.ts +1 -0
- package/lib/components/billing/index.js +1 -0
- package/lib/components/checkout/StripeCheckout.d.ts +1 -1
- package/lib/components/checkout/StripeCheckout.js +1 -1
- package/lib/components/collaboration/LiveEditorCollaborators.d.ts +5 -0
- package/lib/components/collaboration/LiveEditorCollaborators.js +3 -1
- package/lib/components/display/DatalayerBox.d.ts +2 -3
- package/lib/components/display/DatalayerBox.js +1 -1
- package/lib/components/display/NavLink.js +3 -2
- package/lib/components/display/VisuallyHidden.d.ts +1 -2
- package/lib/components/labels/StatusLabels.d.ts +2 -10
- package/lib/components/labels/StatusLabels.js +2 -26
- package/lib/components/principal/PrincipalAppearance.d.ts +21 -42
- package/lib/components/principal/PrincipalAppearance.js +232 -174
- package/lib/components/principal/PrincipalAvatar.d.ts +8 -1
- package/lib/components/principal/PrincipalAvatar.js +2 -2
- package/lib/components/principal/PrincipalDetailsOverlay.d.ts +8 -1
- package/lib/components/principal/PrincipalDetailsOverlay.js +2 -2
- package/lib/components/screencapture/Screencapture.js +46 -3
- package/lib/components/sharing/ShareAccessComponent.d.ts +45 -4
- package/lib/components/sharing/ShareAccessComponent.js +110 -78
- package/lib/components/sharing/SharingEditor.d.ts +5 -1
- package/lib/components/sharing/SharingEditor.js +9 -4
- package/lib/components/sharing/index.d.ts +1 -0
- package/lib/components/sharing/index.js +1 -0
- package/lib/components/sharing/sandboxSharing.d.ts +26 -0
- package/lib/components/sharing/sandboxSharing.js +47 -0
- package/lib/components/spaces/SpaceDetailsOverlay.d.ts +39 -0
- package/lib/components/spaces/SpaceDetailsOverlay.js +80 -0
- package/lib/components/spaces/SpaceDisplay.d.ts +38 -0
- package/lib/components/spaces/SpaceDisplay.js +92 -0
- package/lib/components/spaces/index.d.ts +4 -0
- package/lib/components/spaces/index.js +7 -0
- package/lib/components/spaces/spaceDisplayModel.d.ts +93 -0
- package/lib/components/spaces/spaceDisplayModel.js +170 -0
- package/lib/config/Configuration.d.ts +4 -4
- package/lib/config/Configuration.js +7 -7
- package/lib/config/index.d.ts +1 -0
- package/lib/config/index.js +1 -0
- package/lib/config/planes.d.ts +45 -0
- package/lib/config/planes.js +53 -0
- package/lib/hooks/cacheConverters.d.ts +0 -7
- package/lib/hooks/cacheConverters.js +0 -25
- package/lib/hooks/index.d.ts +1 -0
- package/lib/hooks/index.js +1 -0
- package/lib/hooks/useBackdrop.d.ts +4 -5
- package/lib/hooks/useBillingEntityStore.js +3 -0
- package/lib/hooks/useCache.d.ts +197 -44
- package/lib/hooks/useCache.js +1586 -494
- package/lib/hooks/useContents.d.ts +22 -1
- package/lib/hooks/useContents.js +90 -21
- package/lib/hooks/useMcp.d.ts +87 -1
- package/lib/hooks/useMcp.js +248 -13
- package/lib/hooks/useNavigate.d.ts +41 -3
- package/lib/hooks/useNavigate.js +116 -52
- package/lib/hooks/usePrincipalCacheStore.js +3 -0
- package/lib/hooks/usePrincipalStore.js +3 -0
- package/lib/hooks/useSpaceCacheStore.d.ts +50 -0
- package/lib/hooks/useSpaceCacheStore.js +85 -0
- package/lib/hooks/useWindowSize.js +1 -1
- package/lib/models/CreditsDTO.d.ts +6 -5
- package/lib/models/CreditsDTO.js +7 -5
- package/lib/models/IAMProvidersSpecs.d.ts +10 -5
- package/lib/models/IAMProvidersSpecs.js +0 -23
- package/lib/models/McpBinding.d.ts +6 -0
- package/lib/models/Organization.d.ts +69 -0
- package/lib/models/Organization.js +96 -0
- package/lib/models/Secret.d.ts +1 -2
- package/lib/models/Secret.js +2 -18
- package/lib/models/StartedBy.d.ts +34 -0
- package/lib/models/StartedBy.js +36 -0
- package/lib/models/Team.d.ts +2 -0
- package/lib/models/Team.js +2 -0
- package/lib/models/User.d.ts +4 -0
- package/lib/models/User.js +3 -0
- package/lib/models/UserSettings.d.ts +13 -0
- package/lib/models/UserSettings.js +9 -0
- package/lib/models/contents/__fixtures__/v1-contracts.json +158 -30
- package/lib/navigation/adapters/react-router.d.ts +1 -1
- package/lib/navigation/adapters/react-router.js +3 -0
- package/lib/navigation/components.js +2 -3
- package/lib/routes/publicPaths.js +4 -0
- package/lib/state/index.d.ts +1 -0
- package/lib/state/index.js +1 -0
- package/lib/state/sessionEnd.d.ts +55 -0
- package/lib/state/sessionEnd.js +162 -0
- package/lib/state/substates/CoreState.js +3 -16
- package/lib/state/substates/IAMState.d.ts +3 -3
- package/lib/state/substates/IAMState.js +7 -0
- package/lib/state/substates/LayoutState.js +12 -0
- package/lib/state/substates/NavigationState.d.ts +82 -0
- package/lib/state/substates/NavigationState.js +255 -0
- package/lib/state/substates/ProfileState.d.ts +1 -0
- package/lib/state/substates/ProfileState.js +1 -0
- package/lib/state/substates/index.d.ts +1 -0
- package/lib/state/substates/index.js +1 -0
- package/lib/utils/Lazy.d.ts +20 -0
- package/lib/utils/Lazy.js +32 -6
- package/lib/utils/Screencapture.d.ts +15 -0
- package/lib/utils/Screencapture.js +26 -20
- package/lib/utils/WithSuspense.d.ts +11 -1
- package/lib/utils/WithSuspense.js +11 -1
- package/lib/views/mcp/AdmittedClients.d.ts +44 -0
- package/lib/views/mcp/AdmittedClients.js +59 -0
- package/lib/views/mcp/ApprovalQueue.d.ts +66 -0
- package/lib/views/mcp/ApprovalQueue.js +95 -0
- package/lib/views/mcp/ConnectedAgents.js +38 -3
- package/lib/views/mcp/EnterpriseConsole.d.ts +1 -1
- package/lib/views/mcp/EnterpriseConsole.js +23 -1
- package/lib/views/mcp/IdentityProviders.d.ts +33 -0
- package/lib/views/mcp/IdentityProviders.js +299 -0
- package/lib/views/mcp/McpDashboard.js +90 -7
- package/lib/views/mcp/McpHome.js +1 -1
- package/lib/views/mcp/McpObservability.d.ts +30 -1
- package/lib/views/mcp/McpObservability.js +105 -19
- package/lib/views/mcp/NotebookRuns.d.ts +39 -0
- package/lib/views/mcp/NotebookRuns.js +44 -0
- package/lib/views/mcp/OrganizationPolicy.js +2 -1
- package/lib/views/mcp/PersonalPolicy.js +2 -1
- package/lib/views/mcp/PolicyForm.d.ts +17 -1
- package/lib/views/mcp/PolicyForm.js +25 -3
- package/lib/views/mcp/PolicyHistory.d.ts +13 -0
- package/lib/views/mcp/PolicyHistory.js +14 -1
- package/lib/views/mcp/RunDetail.d.ts +87 -0
- package/lib/views/mcp/RunDetail.js +186 -0
- package/lib/views/mcp/ScimProvisioning.d.ts +54 -0
- package/lib/views/mcp/ScimProvisioning.js +173 -0
- package/lib/views/mcp/TeamPolicies.js +66 -5
- package/lib/views/mcp/ToolAccess.d.ts +60 -0
- package/lib/views/mcp/ToolAccess.js +101 -0
- package/lib/views/mcp/TraceTimeline.d.ts +81 -0
- package/lib/views/mcp/TraceTimeline.js +132 -0
- package/lib/views/mcp/index.d.ts +5 -0
- package/lib/views/mcp/index.js +5 -0
- package/lib/views/otel/simpleAuthStore.js +3 -0
- package/lib/views/secrets/SecretEdit.js +3 -2
- package/lib/views/secrets/SecretNew.js +5 -1
- package/package.json +271 -264
- package/scripts/generate-mcp-types.py +16 -10
- package/scripts/generate-orchestration-types.py +920 -0
|
@@ -0,0 +1,402 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The evals API, one function per endpoint.
|
|
3
|
+
*
|
|
4
|
+
* Paths mirror `tests/test_evals_api_contract.py` in the AI Agents service
|
|
5
|
+
* and `evalsApiContract.spec.ts` in the landings UI. Downloads (export and
|
|
6
|
+
* report) are URLs rather than requests: the browser opens them and the
|
|
7
|
+
* service answers with an attachment.
|
|
8
|
+
*
|
|
9
|
+
* @module api/evals/client
|
|
10
|
+
*/
|
|
11
|
+
import { type EvalsClientOptions } from './request';
|
|
12
|
+
import type { CaseListResponse, CreateReportRequest, MoveReportRequest, RegenerateReportResponse, ReportListResponse, ReportResponse, ReportsQuery, DecisionListResponse, DecisionRequest, DecisionResponse, DecisionSubject, SharedWithMeResponse, ContinueReportImportResponse, ReportImportListResponse, ReportImportRequest, ReportImportResponse, EvalsPermissionsResponse, EvalsSharingResponse, EvalsSharingUpdate, SharedEvalsRecord, EvalEvaluatorRef, EvalsetVersionListResponse, EvalsetVersionResponse, InvestigationListResponse, InvestigationResponse, InvestigationScope, InvestigationStatus, LexicalReportResponse, ReportDocumentResponse, ResumeSandboxRequest, ResumeSandboxResponse, TaskInvestigationResponse, UpdateInvestigationRequest, EvalsetCategory, ImportEvalsetRequest, ImportEvalsetResponse, CaseResultListResponse, CaseResultResponse, RunNetworkResponse, RunNetworkMessageResponse, LaunchNetworkResponse, CreateLaunchRequest, LaunchCancelResponse, LaunchListResponse, ClaimTrialResponse, LaunchPlanResponse, LaunchResponse, LiveAlertListResponse, SubjectsResponse, ReviewCaseRequest, RunCancelResponse, CaseRequest, CaseResponse, CreateEvalsetRequest, CreateExperimentRequest, CreateLiveEventRequest, CreateRunRequest, EvalKind, EvalRunEnvironment, EvalsetDeleteResponse, EvalsetListResponse, EvalsetResponse, EvalsetPackagePreviewResponse, EvalsetComparisonResponse, SubjectUsageResponse, SandboxProvenanceResponse, BenchmarkComputeResponse, EvalsSearchResponse, EvalsetLeaderboardResponse, EvalsetPublicationResponse, EvalsetPublicationListResponse, ExperimentDeleteResponse, ExperimentListResponse, ExperimentResponse, LiveEventListResponse, LiveEventResponse, LiveTargetDeleteResponse, LiveTargetListResponse, PublicEvalsetDetailsResponse, RunListResponse, RunResponse, SuccessResponse, UpdateEvalsetRequest, UpdateExperimentRequest } from './types';
|
|
13
|
+
export type ListEvalsetsQuery = {
|
|
14
|
+
kind?: EvalKind;
|
|
15
|
+
run_environment?: EvalRunEnvironment;
|
|
16
|
+
category?: EvalsetCategory;
|
|
17
|
+
q?: string;
|
|
18
|
+
/** Only the benchmarks reading this dataset (B5-12, section 15.3). */
|
|
19
|
+
dataset?: string;
|
|
20
|
+
/** …and pinned to this revision of it: one revision is not evidence about another. */
|
|
21
|
+
revision?: string;
|
|
22
|
+
limit?: number;
|
|
23
|
+
offset?: number;
|
|
24
|
+
};
|
|
25
|
+
export declare const listEvalsets: (options: EvalsClientOptions, query?: ListEvalsetsQuery) => Promise<EvalsetListResponse>;
|
|
26
|
+
export declare const getEvalset: (options: EvalsClientOptions, evalsetId: string) => Promise<EvalsetResponse>;
|
|
27
|
+
export declare const createEvalset: (options: EvalsClientOptions, body: CreateEvalsetRequest) => Promise<EvalsetResponse>;
|
|
28
|
+
export declare const updateEvalset: (options: EvalsClientOptions, evalsetId: string, body: UpdateEvalsetRequest) => Promise<EvalsetResponse>;
|
|
29
|
+
export declare const setEvalsetPublic: (options: EvalsClientOptions, evalsetId: string, isPublic: boolean) => Promise<EvalsetResponse>;
|
|
30
|
+
/**
|
|
31
|
+
* Take a benchmark into the account the options name (B5-03): a private copy
|
|
32
|
+
* of its definition that says what it was taken from. A published benchmark
|
|
33
|
+
* is anybody's to take; an unpublished one, a Viewer's. `derivation` says why:
|
|
34
|
+
* `fork` to change it, `compare` to run it against one's own agent (B5-05).
|
|
35
|
+
*/
|
|
36
|
+
export declare const cloneEvalset: (options: EvalsClientOptions, evalsetId: string, body?: {
|
|
37
|
+
name?: string;
|
|
38
|
+
derivation?: "fork" | "compare";
|
|
39
|
+
}) => Promise<EvalsetResponse>;
|
|
40
|
+
/**
|
|
41
|
+
* Propose the next version of a benchmark's definition, from an investigation
|
|
42
|
+
* (B5-07): a task added or corrected, a corrected evaluator, a threshold
|
|
43
|
+
* moved. The version records the investigation that asked for it.
|
|
44
|
+
*/
|
|
45
|
+
export declare const reviseEvalset: (options: EvalsClientOptions, evalsetId: string, body: {
|
|
46
|
+
investigation_id: string;
|
|
47
|
+
note?: string;
|
|
48
|
+
case?: Record<string, unknown>;
|
|
49
|
+
evalset_evaluators?: EvalEvaluatorRef[];
|
|
50
|
+
report_evaluators?: EvalEvaluatorRef[];
|
|
51
|
+
}) => Promise<EvalsetResponse>;
|
|
52
|
+
/**
|
|
53
|
+
* What publishing this benchmark would put in the library (B5-02): the
|
|
54
|
+
* contents of the package, the enumeration the publication review shows, what
|
|
55
|
+
* is left out and why, and everything that would refuse it — before anybody
|
|
56
|
+
* presses the button, because a publication cannot be taken back from whoever
|
|
57
|
+
* already read it.
|
|
58
|
+
*/
|
|
59
|
+
export declare const previewEvalsetPublication: (options: EvalsClientOptions, evalsetId: string, query?: {
|
|
60
|
+
/** Launch ids, comma separated; the benchmark's own where empty. */
|
|
61
|
+
launches?: string;
|
|
62
|
+
report?: string;
|
|
63
|
+
/** Comment uids selected for publication; nothing is published unnamed. */
|
|
64
|
+
comments?: string;
|
|
65
|
+
decisions?: string;
|
|
66
|
+
version?: number;
|
|
67
|
+
}) => Promise<EvalsetPackagePreviewResponse>;
|
|
68
|
+
/**
|
|
69
|
+
* Publish the package: the immutable snapshot of the definition, the data, the
|
|
70
|
+
* subjects, the environment, the results, the report, the evidence and the
|
|
71
|
+
* evaluators (section 14.3). Refused with `detail.problems` where section 14.5
|
|
72
|
+
* says it cannot be published.
|
|
73
|
+
*/
|
|
74
|
+
export declare const publishEvalsetPackage: (options: EvalsClientOptions, evalsetId: string, body?: {
|
|
75
|
+
launch_ids?: string[];
|
|
76
|
+
report_id?: string;
|
|
77
|
+
comment_uids?: string[];
|
|
78
|
+
decision_uids?: string[];
|
|
79
|
+
note?: string;
|
|
80
|
+
version?: number;
|
|
81
|
+
}) => Promise<EvalsetPublicationResponse>;
|
|
82
|
+
/** Every package published of this benchmark, withdrawn ones included. */
|
|
83
|
+
export declare const listEvalsetPublications: (options: EvalsClientOptions, evalsetId: string, query?: {
|
|
84
|
+
limit?: number;
|
|
85
|
+
offset?: number;
|
|
86
|
+
}) => Promise<EvalsetPublicationListResponse>;
|
|
87
|
+
/**
|
|
88
|
+
* Take a package out of the library. What it holds stays as it was written:
|
|
89
|
+
* withdrawing a snapshot does not make it editable.
|
|
90
|
+
*/
|
|
91
|
+
export declare const withdrawEvalsetPublication: (options: EvalsClientOptions, evalsetId: string, publicationId: string) => Promise<EvalsetPublicationResponse>;
|
|
92
|
+
/** The package a published benchmark stands for, to anybody (B5-02). */
|
|
93
|
+
export declare const getPublicEvalsetPublication: (options: EvalsClientOptions, evalsetId: string) => Promise<EvalsetPublicationResponse>;
|
|
94
|
+
/**
|
|
95
|
+
* Launches of one benchmark side by side (B5-08): what each scored, what each
|
|
96
|
+
* subject scored in each, and which tasks went from passing to failing or
|
|
97
|
+
* back. Launch ids are compared in the order given, oldest first; the two
|
|
98
|
+
* newest launches where none is named.
|
|
99
|
+
*/
|
|
100
|
+
export declare const compareEvalsetLaunches: (options: EvalsClientOptions, evalsetId: string, query?: {
|
|
101
|
+
launches?: string;
|
|
102
|
+
/** The published report that sent somebody here, when one did (B5-14). */
|
|
103
|
+
report?: string;
|
|
104
|
+
}) => Promise<EvalsetComparisonResponse>;
|
|
105
|
+
/**
|
|
106
|
+
* The subjects that ran a published benchmark, best first. Only runs a package
|
|
107
|
+
* published are counted (B5-02), and each row names the package its best run
|
|
108
|
+
* came from.
|
|
109
|
+
*/
|
|
110
|
+
export declare const getPublicEvalsetLeaderboard: (options: EvalsClientOptions, evalsetId: string) => Promise<EvalsetLeaderboardResponse>;
|
|
111
|
+
/**
|
|
112
|
+
* What an agent or a model has actually run (B5-12): the benchmarks that ran
|
|
113
|
+
* it, what its runs scored, what they cost, how long they took, and the
|
|
114
|
+
* launches still going. Empty where it has never been run.
|
|
115
|
+
*
|
|
116
|
+
* The ref keeps its slashes — an agentspec id has one — so each segment is
|
|
117
|
+
* encoded rather than the whole string.
|
|
118
|
+
*/
|
|
119
|
+
export declare const getSubjectUsage: (options: EvalsClientOptions, subjectRef: string) => Promise<SubjectUsageResponse>;
|
|
120
|
+
/**
|
|
121
|
+
* Whether anything will run the account's benchmarks (B5-12, section 15.2):
|
|
122
|
+
* the executor, whether a worker polls the benchmark queue, and the account's
|
|
123
|
+
* own queued and running launches.
|
|
124
|
+
*/
|
|
125
|
+
export declare const getBenchmarkCompute: (options: EvalsClientOptions) => Promise<BenchmarkComputeResponse>;
|
|
126
|
+
/**
|
|
127
|
+
* What benchmark work a sandbox is doing (B5-12, section 15.4).
|
|
128
|
+
*
|
|
129
|
+
* The ref is a pool slot's name (`benchmark-<run>-slot-<n>`) or the snapshot
|
|
130
|
+
* a sandbox was restored from. A sandbox of nobody's benchmark answers
|
|
131
|
+
* `found: false`, so a page asks without knowing in advance.
|
|
132
|
+
*/
|
|
133
|
+
export declare const getSandboxProvenance: (options: EvalsClientOptions, sandboxRef: string) => Promise<SandboxProvenanceResponse>;
|
|
134
|
+
/**
|
|
135
|
+
* Find things across the caller's benchmarks (B5-11, section 19): benchmarks,
|
|
136
|
+
* launches, tasks, reports and investigations, each row keeping its kind and
|
|
137
|
+
* what it belongs to.
|
|
138
|
+
*
|
|
139
|
+
* Nobody types a field name, so the service reads the words first: a status
|
|
140
|
+
* word becomes a status, `run 128` becomes a launch number, and what is left
|
|
141
|
+
* is the text. An empty query finds nothing rather than everything.
|
|
142
|
+
*/
|
|
143
|
+
export declare const searchEvals: (options: EvalsClientOptions, query?: {
|
|
144
|
+
q?: string;
|
|
145
|
+
kinds?: string;
|
|
146
|
+
limit?: number;
|
|
147
|
+
}) => Promise<EvalsSearchResponse>;
|
|
148
|
+
export declare const renameEvalset: (options: EvalsClientOptions, evalsetId: string, name: string) => Promise<EvalsetResponse>;
|
|
149
|
+
export declare const deleteEvalset: (options: EvalsClientOptions, evalsetId: string) => Promise<EvalsetDeleteResponse>;
|
|
150
|
+
export declare const listCases: (options: EvalsClientOptions, evalsetId: string) => Promise<CaseListResponse>;
|
|
151
|
+
export declare const createCase: (options: EvalsClientOptions, evalsetId: string, body: CaseRequest) => Promise<CaseResponse>;
|
|
152
|
+
export declare const updateCase: (options: EvalsClientOptions, evalsetId: string, caseId: string, body: Partial<CaseRequest>) => Promise<CaseResponse>;
|
|
153
|
+
export declare const deleteCase: (options: EvalsClientOptions, evalsetId: string, caseId: string) => Promise<SuccessResponse>;
|
|
154
|
+
export type EvalsetExportFormat = 'json' | 'pydantic-evals';
|
|
155
|
+
export type EvalsetReportFormat = 'markdown' | 'csv';
|
|
156
|
+
/** The URL the browser opens to download an evalset definition. */
|
|
157
|
+
export declare const evalsetExportUrl: (options: EvalsClientOptions, evalsetId: string, format?: EvalsetExportFormat) => string;
|
|
158
|
+
/** The URL the browser opens to download a generated report. */
|
|
159
|
+
export declare const evalsetReportUrl: (options: EvalsClientOptions, evalsetId: string, format?: EvalsetReportFormat, runLimit?: number) => string;
|
|
160
|
+
export declare const getPublicEvalset: (options: EvalsClientOptions, evalsetId: string) => Promise<EvalsetResponse>;
|
|
161
|
+
export declare const getPublicEvalsetDetails: (options: EvalsClientOptions, evalsetId: string, runsLimit?: number) => Promise<PublicEvalsetDetailsResponse>;
|
|
162
|
+
export declare const listPublicRuns: (options: EvalsClientOptions, evalsetId: string, experimentId: string, query?: {
|
|
163
|
+
limit?: number;
|
|
164
|
+
offset?: number;
|
|
165
|
+
}) => Promise<RunListResponse>;
|
|
166
|
+
export type ListExperimentsQuery = {
|
|
167
|
+
evalset_id?: string;
|
|
168
|
+
status?: string;
|
|
169
|
+
limit?: number;
|
|
170
|
+
offset?: number;
|
|
171
|
+
};
|
|
172
|
+
export declare const createExperiment: (options: EvalsClientOptions, body: CreateExperimentRequest) => Promise<ExperimentResponse>;
|
|
173
|
+
export declare const listExperiments: (options: EvalsClientOptions, query?: ListExperimentsQuery) => Promise<ExperimentListResponse>;
|
|
174
|
+
export declare const getExperiment: (options: EvalsClientOptions, experimentId: string) => Promise<ExperimentResponse>;
|
|
175
|
+
export declare const updateExperiment: (options: EvalsClientOptions, experimentId: string, body: UpdateExperimentRequest) => Promise<ExperimentResponse>;
|
|
176
|
+
export declare const deleteExperiment: (options: EvalsClientOptions, experimentId: string) => Promise<ExperimentDeleteResponse>;
|
|
177
|
+
export declare const createRun: (options: EvalsClientOptions, experimentId: string, body: CreateRunRequest) => Promise<RunResponse>;
|
|
178
|
+
export declare const listRuns: (options: EvalsClientOptions, experimentId: string, query?: {
|
|
179
|
+
limit?: number;
|
|
180
|
+
offset?: number;
|
|
181
|
+
}) => Promise<RunListResponse>;
|
|
182
|
+
export declare const getRun: (options: EvalsClientOptions, runId: string) => Promise<RunResponse>;
|
|
183
|
+
export declare const deleteRun: (options: EvalsClientOptions, runId: string) => Promise<SuccessResponse>;
|
|
184
|
+
export declare const compareRuns: (options: EvalsClientOptions, runIds: string[]) => Promise<RunListResponse>;
|
|
185
|
+
export type ListLiveEventsQuery = {
|
|
186
|
+
target_id: string;
|
|
187
|
+
target_type?: string;
|
|
188
|
+
window?: string;
|
|
189
|
+
evaluator_name?: string;
|
|
190
|
+
limit?: number;
|
|
191
|
+
offset?: number;
|
|
192
|
+
};
|
|
193
|
+
export declare const createLiveEvent: (options: EvalsClientOptions, body: CreateLiveEventRequest) => Promise<LiveEventResponse>;
|
|
194
|
+
export declare const listLiveTargets: (options: EvalsClientOptions, query?: {
|
|
195
|
+
window?: string;
|
|
196
|
+
limit?: number;
|
|
197
|
+
}) => Promise<LiveTargetListResponse>;
|
|
198
|
+
export declare const listLiveEvents: (options: EvalsClientOptions, query: ListLiveEventsQuery) => Promise<LiveEventListResponse>;
|
|
199
|
+
export type ListLiveAlertsQuery = {
|
|
200
|
+
target_id?: string;
|
|
201
|
+
target_type?: string;
|
|
202
|
+
experiment_id?: string;
|
|
203
|
+
launch_id?: string;
|
|
204
|
+
window?: string;
|
|
205
|
+
limit?: number;
|
|
206
|
+
};
|
|
207
|
+
/** The alerts the rolling windows raised: failure spikes and drifts (B2-13). */
|
|
208
|
+
export declare const listLiveAlerts: (options: EvalsClientOptions, query?: ListLiveAlertsQuery) => Promise<LiveAlertListResponse>;
|
|
209
|
+
export declare const deleteLiveTarget: (options: EvalsClientOptions, targetId: string, targetType?: string) => Promise<LiveTargetDeleteResponse>;
|
|
210
|
+
export type ListRunsQuery = {
|
|
211
|
+
evalset_id?: string;
|
|
212
|
+
launch_id?: string;
|
|
213
|
+
experiment_id?: string;
|
|
214
|
+
status?: string;
|
|
215
|
+
limit?: number;
|
|
216
|
+
offset?: number;
|
|
217
|
+
};
|
|
218
|
+
export declare const listRunsAcross: (options: EvalsClientOptions, query?: ListRunsQuery) => Promise<RunListResponse>;
|
|
219
|
+
export declare const cancelRun: (options: EvalsClientOptions, runId: string) => Promise<RunCancelResponse>;
|
|
220
|
+
export type ListCaseResultsQuery = {
|
|
221
|
+
status?: string;
|
|
222
|
+
category?: string;
|
|
223
|
+
difficulty?: string;
|
|
224
|
+
failure_mode?: string;
|
|
225
|
+
min_score?: number;
|
|
226
|
+
max_score?: number;
|
|
227
|
+
q?: string;
|
|
228
|
+
sort?: string;
|
|
229
|
+
limit?: number;
|
|
230
|
+
offset?: number;
|
|
231
|
+
};
|
|
232
|
+
/** The task grid of a run, filtered and paged by the service. */
|
|
233
|
+
export declare const listCaseResults: (options: EvalsClientOptions, runId: string, query?: ListCaseResultsQuery) => Promise<CaseResultListResponse>;
|
|
234
|
+
export declare const getCaseResult: (options: EvalsClientOptions, runId: string, caseId: string) => Promise<CaseResultResponse>;
|
|
235
|
+
/**
|
|
236
|
+
* The run as the network of agents that worked on it. `budget` is how many
|
|
237
|
+
* tasks' messages to answer: every task that did not pass, then passed ones
|
|
238
|
+
* while there is room.
|
|
239
|
+
*/
|
|
240
|
+
export declare const getRunNetwork: (options: EvalsClientOptions, runId: string, query?: {
|
|
241
|
+
budget?: number;
|
|
242
|
+
}) => Promise<RunNetworkResponse>;
|
|
243
|
+
/**
|
|
244
|
+
* One message of a run's network, in full. The id is the service's own and
|
|
245
|
+
* holds colons, which the route reads as a path: each segment is encoded, the
|
|
246
|
+
* separators are kept.
|
|
247
|
+
*/
|
|
248
|
+
export declare const getRunNetworkMessage: (options: EvalsClientOptions, runId: string, messageId: string) => Promise<RunNetworkMessageResponse>;
|
|
249
|
+
/** A launch's runs, each as one line of its own network. */
|
|
250
|
+
export declare const getLaunchNetwork: (options: EvalsClientOptions, launchId: string) => Promise<LaunchNetworkResponse>;
|
|
251
|
+
export declare const reviewCaseResult: (options: EvalsClientOptions, runId: string, caseId: string, body: ReviewCaseRequest) => Promise<CaseResultResponse>;
|
|
252
|
+
/**
|
|
253
|
+
* Give an anonymous trial's work to the person who just signed in (B2-14):
|
|
254
|
+
* the benchmark, the launch, its runs and everything under them, moved
|
|
255
|
+
* rather than copied so a link the visitor kept still opens.
|
|
256
|
+
*
|
|
257
|
+
* Takes two credentials: the person's, in the client's options, and the
|
|
258
|
+
* trial's own key, which travels in its header.
|
|
259
|
+
*/
|
|
260
|
+
export declare const claimTrial: (options: EvalsClientOptions, trialUid: string, trialToken: string) => Promise<ClaimTrialResponse>;
|
|
261
|
+
/** The subjects an experiment can have, and the models offered (B2-11). */
|
|
262
|
+
export declare const listSubjects: (options: EvalsClientOptions) => Promise<SubjectsResponse>;
|
|
263
|
+
/**
|
|
264
|
+
* What a page may report about what somebody did (B2-25, B3-12).
|
|
265
|
+
*
|
|
266
|
+
* The six lines of the funnel no record answers: nothing is written when
|
|
267
|
+
* somebody opens the Run Benchmark wizard, reads a task, asks the result agent
|
|
268
|
+
* a question, runs a follow-up cell, or pins evidence into a report. A closed
|
|
269
|
+
* set, because a free-form label from a browser is a cardinality bomb with an
|
|
270
|
+
* authenticated route in front of it.
|
|
271
|
+
*/
|
|
272
|
+
export type ProductEvent = 'wizard.started' | 'wizard.completed' | 'task.opened' | 'question.asked' | 'cell.executed' | 'evidence.added' | 'network.viewed' | 'network.node_selected' | 'network.message_opened' | 'network.investigate_clicked';
|
|
273
|
+
/**
|
|
274
|
+
* Count one of those.
|
|
275
|
+
*
|
|
276
|
+
* Nothing is stored, and the answer says only that it was counted. Never worth
|
|
277
|
+
* failing a page for: a measure that breaks what somebody was doing is worse
|
|
278
|
+
* than a measure nobody has.
|
|
279
|
+
*/
|
|
280
|
+
export declare const recordProductEvent: (options: EvalsClientOptions, event: ProductEvent) => Promise<{
|
|
281
|
+
success: boolean;
|
|
282
|
+
}>;
|
|
283
|
+
/**
|
|
284
|
+
* The plan of a launch before it is made (B2-07): estimated duration and
|
|
285
|
+
* cost, the compute, and the problems that would stop it. Same body as the
|
|
286
|
+
* launch; nothing is created.
|
|
287
|
+
*/
|
|
288
|
+
export declare const validateLaunch: (options: EvalsClientOptions, evalsetId: string, body: CreateLaunchRequest) => Promise<LaunchPlanResponse>;
|
|
289
|
+
/** One submission of a benchmark across its experiments. */
|
|
290
|
+
export declare const createLaunch: (options: EvalsClientOptions, evalsetId: string, body: CreateLaunchRequest) => Promise<LaunchResponse>;
|
|
291
|
+
export type ListLaunchesQuery = {
|
|
292
|
+
evalset_id?: string;
|
|
293
|
+
status?: string;
|
|
294
|
+
include_archived?: boolean;
|
|
295
|
+
limit?: number;
|
|
296
|
+
offset?: number;
|
|
297
|
+
};
|
|
298
|
+
export declare const listLaunches: (options: EvalsClientOptions, query?: ListLaunchesQuery) => Promise<LaunchListResponse>;
|
|
299
|
+
export declare const getLaunch: (options: EvalsClientOptions, launchId: string) => Promise<LaunchResponse>;
|
|
300
|
+
export declare const cancelLaunch: (options: EvalsClientOptions, launchId: string) => Promise<LaunchCancelResponse>;
|
|
301
|
+
export declare const archiveLaunch: (options: EvalsClientOptions, launchId: string) => Promise<SuccessResponse>;
|
|
302
|
+
/** An evalset from a spec file, the same body the CLI and the action send. */
|
|
303
|
+
export declare const importEvalset: (options: EvalsClientOptions, body: ImportEvalsetRequest) => Promise<ImportEvalsetResponse>;
|
|
304
|
+
export declare const listEvalsetVersions: (options: EvalsClientOptions, evalsetId: string, query?: {
|
|
305
|
+
limit?: number;
|
|
306
|
+
offset?: number;
|
|
307
|
+
}) => Promise<EvalsetVersionListResponse>;
|
|
308
|
+
export declare const getEvalsetVersion: (options: EvalsClientOptions, evalsetId: string, version: number) => Promise<EvalsetVersionResponse>;
|
|
309
|
+
/** The benchmark's whole report (every launch) as a serialized Lexical editor state. */
|
|
310
|
+
export declare const getEvalsetReportDocument: (options: EvalsClientOptions, evalsetId: string, query?: {
|
|
311
|
+
run_limit?: number;
|
|
312
|
+
}) => Promise<LexicalReportResponse>;
|
|
313
|
+
/** The launch's report as a serialized Lexical editor state. */
|
|
314
|
+
export declare const getLaunchReportDocument: (options: EvalsClientOptions, launchId: string, query?: {
|
|
315
|
+
run_limit?: number;
|
|
316
|
+
}) => Promise<LexicalReportResponse>;
|
|
317
|
+
/** Where a launch's Markdown or CSV report is downloaded from. */
|
|
318
|
+
export declare const launchReportUrl: (options: EvalsClientOptions, launchId: string, format: "markdown" | "csv") => string;
|
|
319
|
+
/** One run's report as a serialized Lexical editor state, with its task documents. */
|
|
320
|
+
export declare const getRunReportDocument: (options: EvalsClientOptions, runId: string) => Promise<LexicalReportResponse>;
|
|
321
|
+
export declare const runReportUrl: (options: EvalsClientOptions, runId: string, format: "markdown" | "csv") => string;
|
|
322
|
+
/** Write the launch's report into the account's benchmarks space. */
|
|
323
|
+
export declare const writeLaunchReportDocument: (options: EvalsClientOptions, launchId: string, body?: {
|
|
324
|
+
name?: string;
|
|
325
|
+
}) => Promise<ReportDocumentResponse>;
|
|
326
|
+
/** Open the investigation of one task, or get the one already open. */
|
|
327
|
+
export declare const openTaskInvestigation: (options: EvalsClientOptions, runId: string, caseId: string) => Promise<TaskInvestigationResponse>;
|
|
328
|
+
export declare const openLaunchInvestigation: (options: EvalsClientOptions, launchId: string) => Promise<InvestigationResponse>;
|
|
329
|
+
export type InvestigationsQuery = {
|
|
330
|
+
evalset_id?: string;
|
|
331
|
+
launch_id?: string;
|
|
332
|
+
run_id?: string;
|
|
333
|
+
scope?: InvestigationScope;
|
|
334
|
+
status?: InvestigationStatus;
|
|
335
|
+
limit?: number;
|
|
336
|
+
offset?: number;
|
|
337
|
+
};
|
|
338
|
+
export declare const listInvestigations: (options: EvalsClientOptions, query?: InvestigationsQuery) => Promise<InvestigationListResponse>;
|
|
339
|
+
export declare const getInvestigation: (options: EvalsClientOptions, investigationId: string) => Promise<InvestigationResponse>;
|
|
340
|
+
export declare const updateInvestigation: (options: EvalsClientOptions, investigationId: string, body: UpdateInvestigationRequest) => Promise<InvestigationResponse>;
|
|
341
|
+
/**
|
|
342
|
+
* Publish an investigation to the library, or take it back: its owner's alone.
|
|
343
|
+
* The document and the notebooks it is written in go public and private with
|
|
344
|
+
* it.
|
|
345
|
+
*/
|
|
346
|
+
export declare const setInvestigationPublic: (options: EvalsClientOptions, investigationId: string, isPublic: boolean) => Promise<InvestigationResponse>;
|
|
347
|
+
/**
|
|
348
|
+
* Keep the investigation's page as it is shown, under a name everybody on
|
|
349
|
+
* the investigation sees; a view of that name is replaced (B4-11).
|
|
350
|
+
*/
|
|
351
|
+
export declare const saveInvestigationView: (options: EvalsClientOptions, investigationId: string, body: {
|
|
352
|
+
name: string;
|
|
353
|
+
query: string;
|
|
354
|
+
}) => Promise<InvestigationResponse>;
|
|
355
|
+
/** Forget a saved view of the investigation's page. */
|
|
356
|
+
export declare const forgetInvestigationView: (options: EvalsClientOptions, investigationId: string, name: string) => Promise<InvestigationResponse>;
|
|
357
|
+
/** Bring the task's sandbox back from its snapshot, bound to its investigation. */
|
|
358
|
+
export declare const resumeTaskSandbox: (options: EvalsClientOptions, runId: string, caseId: string, body?: ResumeSandboxRequest) => Promise<ResumeSandboxResponse>;
|
|
359
|
+
/** Who a benchmark, one of its runs or an investigation is shared with; its owner's to read. */
|
|
360
|
+
export declare const getEvalsSharing: (options: EvalsClientOptions, record: SharedEvalsRecord, uid: string) => Promise<EvalsSharingResponse>;
|
|
361
|
+
/** Replace the grants at the levels named; the others are kept. Its owner's to do. */
|
|
362
|
+
export declare const updateEvalsSharing: (options: EvalsClientOptions, record: SharedEvalsRecord, uid: string, access: EvalsSharingUpdate) => Promise<EvalsSharingResponse>;
|
|
363
|
+
/** The caller's role on the record, and what it allows. */
|
|
364
|
+
export declare const getEvalsPermissions: (options: EvalsClientOptions, record: SharedEvalsRecord, uid: string) => Promise<EvalsPermissionsResponse>;
|
|
365
|
+
/** A report over runs of a benchmark, written as a draft. */
|
|
366
|
+
export declare const createReport: (options: EvalsClientOptions, body: CreateReportRequest) => Promise<ReportResponse>;
|
|
367
|
+
export declare const listReports: (options: EvalsClientOptions, query?: ReportsQuery) => Promise<ReportListResponse>;
|
|
368
|
+
export declare const getReport: (options: EvalsClientOptions, reportId: string) => Promise<ReportResponse>;
|
|
369
|
+
/** Send a report for review, send it back, approve it or publish it. */
|
|
370
|
+
export declare const moveReport: (options: EvalsClientOptions, reportId: string, body: MoveReportRequest) => Promise<ReportResponse>;
|
|
371
|
+
/** A new version over newer runs, or the same ones; the old report is superseded by it. */
|
|
372
|
+
export declare const regenerateReport: (options: EvalsClientOptions, reportId: string, launchIds?: string[]) => Promise<RegenerateReportResponse>;
|
|
373
|
+
/** What a reviewer decided about part of a result, recorded on a report or in an investigation. */
|
|
374
|
+
export declare const recordDecision: (options: EvalsClientOptions, subject: DecisionSubject, uid: string, body: DecisionRequest) => Promise<DecisionResponse>;
|
|
375
|
+
/** The decisions made on a report or in an investigation, in the order they were made. */
|
|
376
|
+
export declare const listDecisions: (options: EvalsClientOptions, subject: DecisionSubject, uid: string) => Promise<DecisionListResponse>;
|
|
377
|
+
/**
|
|
378
|
+
* What was decided about a benchmark, or about one of its launches (B6-04).
|
|
379
|
+
*
|
|
380
|
+
* `listDecisions` answers what was decided *in* one report or investigation;
|
|
381
|
+
* this answers what was decided about the benchmark, which is what something
|
|
382
|
+
* outside the product has to ask — CI knows the benchmark it ran and nothing
|
|
383
|
+
* else. Naming neither answers nothing rather than everything.
|
|
384
|
+
*/
|
|
385
|
+
export declare const listDecisionsAbout: (options: EvalsClientOptions, query?: {
|
|
386
|
+
evalset_id?: string;
|
|
387
|
+
launch_id?: string;
|
|
388
|
+
limit?: number;
|
|
389
|
+
}) => Promise<DecisionListResponse>;
|
|
390
|
+
export type ReportExportFormat = 'markdown' | 'csv';
|
|
391
|
+
/** Where a report is downloaded as it was written, with its decisions (B4-08). */
|
|
392
|
+
export declare const reportExportUrl: (options: EvalsClientOptions, reportId: string, format?: ReportExportFormat) => string;
|
|
393
|
+
/** The reports and investigations other people shared with the caller. */
|
|
394
|
+
export declare const listSharedWithMe: (options: EvalsClientOptions) => Promise<SharedWithMeResponse>;
|
|
395
|
+
/** Keep a report file CI produced on its benchmark, as its text. */
|
|
396
|
+
export declare const importReport: (options: EvalsClientOptions, evalsetId: string, body: ReportImportRequest) => Promise<ReportImportResponse>;
|
|
397
|
+
/** The reports imported on a benchmark, newest first, without their text. */
|
|
398
|
+
export declare const listReportImports: (options: EvalsClientOptions, evalsetId: string) => Promise<ReportImportListResponse>;
|
|
399
|
+
/** One imported report, with its text as it was. */
|
|
400
|
+
export declare const getReportImport: (options: EvalsClientOptions, importId: string) => Promise<ReportImportResponse>;
|
|
401
|
+
/** Continue investigation: a live report and an investigation over its runs. */
|
|
402
|
+
export declare const continueReportImport: (options: EvalsClientOptions, importId: string) => Promise<ContinueReportImportResponse>;
|