@mcpjam/inspector 3.7.2 → 3.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/client/assets/dead-clicks-autocapture-BTvoPsHr.js +2 -0
- package/dist/client/assets/exception-autocapture-CbZNLkrO.js +3 -0
- package/dist/client/assets/highlighted-body-OFNGDK62-waZXuFZ0.js +2 -0
- package/dist/client/assets/{index-Ci1cyoqB.js → index-BBgBtcD7.js} +2570 -8
- package/dist/client/assets/index-DysN-8ru.css +1 -0
- package/dist/client/assets/jszip.min-Bo9LSZ25.js +3 -0
- package/dist/client/assets/posthog-recorder-DBHV7V7y.js +109 -0
- package/dist/client/assets/surveys-DMDr7cqK.js +2 -0
- package/dist/client/assets/web-vitals-Cc_UIaaC.js +2 -0
- package/dist/client/index.html +2 -2
- package/dist/server/{chunk-XWZ5Q2R3.js → chunk-ZXGRD3M7.js} +2 -2
- package/dist/server/harness-install-cli.js +1 -1
- package/dist/server/index.js +1768 -1143
- package/package.json +3 -2
- package/dist/client/assets/dead-clicks-autocapture-JWMxkfKA.js +0 -2
- package/dist/client/assets/exception-autocapture-DJenlTnU.js +0 -2
- package/dist/client/assets/highlighted-body-OFNGDK62-COYEfo_x.js +0 -2
- package/dist/client/assets/index-B_UEMZg7.css +0 -1
- package/dist/client/assets/jszip.min-CLkKbqc6.js +0 -2
- package/dist/client/assets/posthog-recorder-jK8kgzdR.js +0 -2
- package/dist/client/assets/surveys-D_p8JkXq.js +0 -2
- package/dist/client/assets/web-vitals-DtvDk_lF.js +0 -2
package/dist/server/index.js
CHANGED
|
@@ -35,7 +35,7 @@ import {
|
|
|
35
35
|
supportsOwnershipProof,
|
|
36
36
|
terminateOwnedProcess,
|
|
37
37
|
terminateOwnedProcessGroup
|
|
38
|
-
} from "./chunk-
|
|
38
|
+
} from "./chunk-ZXGRD3M7.js";
|
|
39
39
|
import {
|
|
40
40
|
ALLOWED_HOSTS,
|
|
41
41
|
BROWSER_CONSENT_HEADER,
|
|
@@ -3683,11 +3683,11 @@ function messagesOf(trace) {
|
|
|
3683
3683
|
}
|
|
3684
3684
|
function countUserTurns(messages) {
|
|
3685
3685
|
if (!Array.isArray(messages)) return void 0;
|
|
3686
|
-
let
|
|
3686
|
+
let count5 = 0;
|
|
3687
3687
|
for (const message of messages) {
|
|
3688
|
-
if (isRecord3(message) && message.role === "user")
|
|
3688
|
+
if (isRecord3(message) && message.role === "user") count5 += 1;
|
|
3689
3689
|
}
|
|
3690
|
-
return
|
|
3690
|
+
return count5;
|
|
3691
3691
|
}
|
|
3692
3692
|
function buildTurnTranscript(input) {
|
|
3693
3693
|
return {
|
|
@@ -21155,8 +21155,8 @@ function errorCheck(detail) {
|
|
|
21155
21155
|
function skippedCheck(detail) {
|
|
21156
21156
|
return { status: "skipped", detail };
|
|
21157
21157
|
}
|
|
21158
|
-
function describeCount(
|
|
21159
|
-
return `${
|
|
21158
|
+
function describeCount(count32, label) {
|
|
21159
|
+
return `${count32} ${label}${count32 === 1 ? "" : "s"} discovered.`;
|
|
21160
21160
|
}
|
|
21161
21161
|
function isServerDoctorError(error) {
|
|
21162
21162
|
return !!error && typeof error === "object" && typeof error.code === "string" && typeof error.message === "string";
|
|
@@ -24145,6 +24145,374 @@ z2.discriminatedUnion("status", [
|
|
|
24145
24145
|
...suspectedConditionProvenance
|
|
24146
24146
|
}).strict()
|
|
24147
24147
|
]);
|
|
24148
|
+
var SWARM_SESSION_VERDICT_CONTRACT_VERSION = 1;
|
|
24149
|
+
var SWARM_ATTEMPT_STATUSES = [
|
|
24150
|
+
"pending",
|
|
24151
|
+
"running",
|
|
24152
|
+
"succeeded",
|
|
24153
|
+
"failed",
|
|
24154
|
+
"rate_limited"
|
|
24155
|
+
];
|
|
24156
|
+
var SWARM_SESSION_LIFECYCLES = [
|
|
24157
|
+
"pending",
|
|
24158
|
+
"running",
|
|
24159
|
+
"ran",
|
|
24160
|
+
"broke",
|
|
24161
|
+
"limited",
|
|
24162
|
+
"withdrawn"
|
|
24163
|
+
];
|
|
24164
|
+
var SWARM_SESSION_VERDICTS = [
|
|
24165
|
+
"passed",
|
|
24166
|
+
"failed",
|
|
24167
|
+
"inconclusive",
|
|
24168
|
+
"notEstablished"
|
|
24169
|
+
];
|
|
24170
|
+
var SWARM_GRADING_STATES = [
|
|
24171
|
+
"notRequested",
|
|
24172
|
+
"queued",
|
|
24173
|
+
"running",
|
|
24174
|
+
"settled",
|
|
24175
|
+
"unavailable"
|
|
24176
|
+
];
|
|
24177
|
+
var SWARM_SESSION_VERDICT_OF_REASON = {
|
|
24178
|
+
attemptPending: "notEstablished",
|
|
24179
|
+
attemptRunning: "notEstablished",
|
|
24180
|
+
withdrawn: "notEstablished",
|
|
24181
|
+
spendCapReached: "notEstablished",
|
|
24182
|
+
notRun: "notEstablished",
|
|
24183
|
+
executionFailed: "notEstablished",
|
|
24184
|
+
ungraded: "notEstablished",
|
|
24185
|
+
gradingNotClaimed: "notEstablished",
|
|
24186
|
+
criteriaPending: "notEstablished",
|
|
24187
|
+
judgePending: "notEstablished",
|
|
24188
|
+
gatingCriterionFailed: "failed",
|
|
24189
|
+
judgeFailed: "failed",
|
|
24190
|
+
criteriaGradingErrored: "inconclusive",
|
|
24191
|
+
gatingCriterionUnmeasured: "inconclusive",
|
|
24192
|
+
judgeErrored: "inconclusive",
|
|
24193
|
+
gradingUnavailable: "inconclusive",
|
|
24194
|
+
allGatingCriteriaPassed: "passed",
|
|
24195
|
+
judgePassed: "passed",
|
|
24196
|
+
allGradersPassed: "passed"
|
|
24197
|
+
};
|
|
24198
|
+
var SWARM_SESSION_VERDICT_REASONS = Object.keys(
|
|
24199
|
+
SWARM_SESSION_VERDICT_OF_REASON
|
|
24200
|
+
);
|
|
24201
|
+
var swarmSessionLifecycleSchema = z2.enum(SWARM_SESSION_LIFECYCLES);
|
|
24202
|
+
var swarmSessionVerdictValueSchema = z2.enum(SWARM_SESSION_VERDICTS);
|
|
24203
|
+
var swarmGraderRoleSchema = z2.enum(["advisory", "required"]);
|
|
24204
|
+
var swarmSessionAttemptInputSchema = z2.object({
|
|
24205
|
+
status: z2.enum(SWARM_ATTEMPT_STATUSES),
|
|
24206
|
+
errorCode: z2.string().nullable().optional()
|
|
24207
|
+
}).strict();
|
|
24208
|
+
var swarmGradingReadinessSchema = z2.object({
|
|
24209
|
+
/** Durable readiness for decisive grading only, not background observations. */
|
|
24210
|
+
state: z2.enum(SWARM_GRADING_STATES),
|
|
24211
|
+
reasonCode: z2.string().min(1).optional()
|
|
24212
|
+
}).strict();
|
|
24213
|
+
var criterionResultSchema = z2.object({
|
|
24214
|
+
criterionId: z2.string().min(1),
|
|
24215
|
+
passed: z2.boolean(),
|
|
24216
|
+
status: z2.enum(["scored", "error"]).optional()
|
|
24217
|
+
}).strict();
|
|
24218
|
+
z2.object({
|
|
24219
|
+
attempt: swarmSessionAttemptInputSchema.nullable(),
|
|
24220
|
+
hasTranscript: z2.boolean(),
|
|
24221
|
+
rubric: z2.array(
|
|
24222
|
+
z2.object({
|
|
24223
|
+
id: z2.string().min(1),
|
|
24224
|
+
role: swarmGraderRoleSchema,
|
|
24225
|
+
predicateType: z2.string().min(1).optional()
|
|
24226
|
+
}).strict()
|
|
24227
|
+
),
|
|
24228
|
+
criteria: z2.object({
|
|
24229
|
+
status: z2.enum(["pending", "completed", "failed"]),
|
|
24230
|
+
criterionIds: z2.array(z2.string().min(1)).optional(),
|
|
24231
|
+
results: z2.array(criterionResultSchema).optional()
|
|
24232
|
+
}).strict().nullable(),
|
|
24233
|
+
goalScore: z2.discriminatedUnion("status", [
|
|
24234
|
+
z2.object({ status: z2.literal("completed"), passed: z2.boolean() }).strict(),
|
|
24235
|
+
z2.object({ status: z2.literal("running") }).strict(),
|
|
24236
|
+
z2.object({ status: z2.literal("failed") }).strict()
|
|
24237
|
+
]).nullable(),
|
|
24238
|
+
judge: z2.object({
|
|
24239
|
+
automatic: z2.boolean(),
|
|
24240
|
+
role: swarmGraderRoleSchema,
|
|
24241
|
+
/** Explicit on-demand intent; automatic:false alone does not mean a judge was requested. */
|
|
24242
|
+
requested: z2.boolean().optional()
|
|
24243
|
+
}).strict(),
|
|
24244
|
+
grading: swarmGradingReadinessSchema
|
|
24245
|
+
}).strict().superRefine((value, ctx) => {
|
|
24246
|
+
const definitions = new Set(value.rubric.map((entry22) => entry22.id));
|
|
24247
|
+
if (definitions.size !== value.rubric.length) {
|
|
24248
|
+
ctx.addIssue({
|
|
24249
|
+
code: "custom",
|
|
24250
|
+
path: ["rubric"],
|
|
24251
|
+
message: "Duplicate criterion definition"
|
|
24252
|
+
});
|
|
24253
|
+
}
|
|
24254
|
+
for (const [field2, ids] of [
|
|
24255
|
+
["criterionIds", value.criteria?.criterionIds ?? []],
|
|
24256
|
+
["results", value.criteria?.results?.map((row2) => row2.criterionId) ?? []]
|
|
24257
|
+
]) {
|
|
24258
|
+
if (new Set(ids).size !== ids.length || ids.some((id3) => !definitions.has(id3))) {
|
|
24259
|
+
ctx.addIssue({
|
|
24260
|
+
code: "custom",
|
|
24261
|
+
path: ["criteria", field2],
|
|
24262
|
+
message: "Criterion IDs must be unique and defined in the snapshot"
|
|
24263
|
+
});
|
|
24264
|
+
}
|
|
24265
|
+
}
|
|
24266
|
+
const claimed = value.criteria?.criterionIds;
|
|
24267
|
+
if (claimed && value.criteria?.results?.some((row2) => !claimed.includes(row2.criterionId))) {
|
|
24268
|
+
ctx.addIssue({
|
|
24269
|
+
code: "custom",
|
|
24270
|
+
path: ["criteria", "results"],
|
|
24271
|
+
message: "Result outside claimed scope"
|
|
24272
|
+
});
|
|
24273
|
+
}
|
|
24274
|
+
});
|
|
24275
|
+
var SWARM_SESSION_TRIAL_STATUSES = [
|
|
24276
|
+
"pending",
|
|
24277
|
+
"running",
|
|
24278
|
+
"completed",
|
|
24279
|
+
"failed",
|
|
24280
|
+
"setup_failed",
|
|
24281
|
+
"cancelled"
|
|
24282
|
+
];
|
|
24283
|
+
var swarmSessionTrialSchema = z2.object({
|
|
24284
|
+
status: z2.enum(SWARM_SESSION_TRIAL_STATUSES),
|
|
24285
|
+
taskVerdict: z2.enum(["passed", "failed"]).optional(),
|
|
24286
|
+
evaluatorError: z2.literal(true).optional()
|
|
24287
|
+
}).strict().superRefine((value, ctx) => {
|
|
24288
|
+
const verdict = value.taskVerdict !== void 0;
|
|
24289
|
+
const error = value.evaluatorError === true;
|
|
24290
|
+
if (value.status === "completed" ? verdict === error : verdict || error) {
|
|
24291
|
+
ctx.addIssue({
|
|
24292
|
+
code: "custom",
|
|
24293
|
+
message: "Only completed trials carry exactly one verdict or evaluator error"
|
|
24294
|
+
});
|
|
24295
|
+
}
|
|
24296
|
+
});
|
|
24297
|
+
var swarmSessionGraderCountsSchema = z2.object({
|
|
24298
|
+
gating: z2.number().int().nonnegative(),
|
|
24299
|
+
gatingPassed: z2.number().int().nonnegative(),
|
|
24300
|
+
gatingFailed: z2.number().int().nonnegative(),
|
|
24301
|
+
gatingUnmeasured: z2.number().int().nonnegative(),
|
|
24302
|
+
advisoryFailed: z2.number().int().nonnegative()
|
|
24303
|
+
}).strict();
|
|
24304
|
+
var swarmSessionVerdictSchema = z2.object({
|
|
24305
|
+
contractVersion: z2.literal(SWARM_SESSION_VERDICT_CONTRACT_VERSION),
|
|
24306
|
+
lifecycle: swarmSessionLifecycleSchema,
|
|
24307
|
+
verdict: swarmSessionVerdictValueSchema,
|
|
24308
|
+
reason: z2.enum(SWARM_SESSION_VERDICT_REASONS),
|
|
24309
|
+
verdictSource: z2.enum([
|
|
24310
|
+
"goalJudge",
|
|
24311
|
+
"requiredAssertions",
|
|
24312
|
+
"combined",
|
|
24313
|
+
"none"
|
|
24314
|
+
]),
|
|
24315
|
+
grading: swarmGradingReadinessSchema,
|
|
24316
|
+
graders: z2.object({
|
|
24317
|
+
criteria: z2.enum([
|
|
24318
|
+
"notConfigured",
|
|
24319
|
+
"notClaimed",
|
|
24320
|
+
"pending",
|
|
24321
|
+
"scored",
|
|
24322
|
+
"errored"
|
|
24323
|
+
]),
|
|
24324
|
+
judge: z2.enum([
|
|
24325
|
+
"notConfigured",
|
|
24326
|
+
"silent",
|
|
24327
|
+
"pending",
|
|
24328
|
+
"scored",
|
|
24329
|
+
"errored"
|
|
24330
|
+
])
|
|
24331
|
+
}).strict(),
|
|
24332
|
+
counts: swarmSessionGraderCountsSchema,
|
|
24333
|
+
trial: swarmSessionTrialSchema.nullable()
|
|
24334
|
+
}).strict().superRefine((value, ctx) => {
|
|
24335
|
+
const fail3 = (message) => ctx.addIssue({ code: "custom", message });
|
|
24336
|
+
if (SWARM_SESSION_VERDICT_OF_REASON[value.reason] !== value.verdict)
|
|
24337
|
+
fail3("Reason contradicts verdict");
|
|
24338
|
+
const c = value.counts;
|
|
24339
|
+
if (c.gating !== c.gatingPassed + c.gatingFailed + c.gatingUnmeasured)
|
|
24340
|
+
fail3("Required measurement counts do not add up");
|
|
24341
|
+
const trial = value.trial;
|
|
24342
|
+
if (trial?.status === "completed") {
|
|
24343
|
+
if (value.lifecycle !== "ran")
|
|
24344
|
+
fail3("Only a completed execution can be a completed trial");
|
|
24345
|
+
if (trial.taskVerdict !== void 0 && trial.taskVerdict !== value.verdict)
|
|
24346
|
+
fail3("Trial contradicts goal result");
|
|
24347
|
+
if (trial.evaluatorError && value.verdict !== "inconclusive")
|
|
24348
|
+
fail3("Evaluator error requires inconclusive grading");
|
|
24349
|
+
}
|
|
24350
|
+
const expected = {
|
|
24351
|
+
pending: "pending",
|
|
24352
|
+
running: "running",
|
|
24353
|
+
broke: "failed",
|
|
24354
|
+
limited: "setup_failed",
|
|
24355
|
+
withdrawn: "cancelled"
|
|
24356
|
+
};
|
|
24357
|
+
if (value.lifecycle !== "ran" && trial?.status !== expected[value.lifecycle])
|
|
24358
|
+
fail3("Trial must preserve execution lifecycle");
|
|
24359
|
+
if (value.lifecycle === "ran" && trial !== null && trial.status !== "completed")
|
|
24360
|
+
fail3("Completed execution must not become skipped or running");
|
|
24361
|
+
if ((value.verdict === "passed" || value.verdict === "failed") && value.verdictSource === "none")
|
|
24362
|
+
fail3("A measured goal verdict needs a source");
|
|
24363
|
+
if (value.lifecycle === "ran" && value.verdict !== "notEstablished" && trial === null)
|
|
24364
|
+
fail3("Settled grading on completed execution needs an observation");
|
|
24365
|
+
if (value.verdict === "notEstablished" && value.verdictSource !== "none")
|
|
24366
|
+
fail3("An undecided goal has no verdict source");
|
|
24367
|
+
});
|
|
24368
|
+
var SWARM_FINDING_CONTRACT_VERSION = 1;
|
|
24369
|
+
var SWARM_FINDING_DISPOSITIONS = [
|
|
24370
|
+
"notRun",
|
|
24371
|
+
"blockedConnecting",
|
|
24372
|
+
"lostFindingTool",
|
|
24373
|
+
"blockedCallingTool",
|
|
24374
|
+
"blockedByResponse",
|
|
24375
|
+
"goalMissed",
|
|
24376
|
+
"goalMetWithFriction",
|
|
24377
|
+
"goalMet",
|
|
24378
|
+
"notMeasured"
|
|
24379
|
+
];
|
|
24380
|
+
var SWARM_FINDING_TONES = ["fail", "warn", "ok", "muted"];
|
|
24381
|
+
var SWARM_FINDING_SUMMARY_KINDS = [
|
|
24382
|
+
"notLaunched",
|
|
24383
|
+
"broken",
|
|
24384
|
+
"friction",
|
|
24385
|
+
"landed",
|
|
24386
|
+
"ungraded",
|
|
24387
|
+
"unread"
|
|
24388
|
+
];
|
|
24389
|
+
var SWARM_FINDING_COVERAGE_NOTES = [
|
|
24390
|
+
"sessionScanCapped",
|
|
24391
|
+
"budgetExhausted",
|
|
24392
|
+
"transcriptMissing",
|
|
24393
|
+
"contextTooLarge",
|
|
24394
|
+
"extractionRejected",
|
|
24395
|
+
"chainUnmeasured",
|
|
24396
|
+
"judgeNotRun",
|
|
24397
|
+
"sessionsWithdrawn",
|
|
24398
|
+
"sessionsRateLimited",
|
|
24399
|
+
"partialRead",
|
|
24400
|
+
"toolCatalogMissing"
|
|
24401
|
+
];
|
|
24402
|
+
var SWARM_FINDING_SCOPE_LEVELS = [
|
|
24403
|
+
"session",
|
|
24404
|
+
"goal",
|
|
24405
|
+
"persona",
|
|
24406
|
+
"target",
|
|
24407
|
+
"wave"
|
|
24408
|
+
];
|
|
24409
|
+
var SWARM_FINDING_BASES = [
|
|
24410
|
+
"verifiedMechanism",
|
|
24411
|
+
"sessionReport",
|
|
24412
|
+
"populationFact"
|
|
24413
|
+
];
|
|
24414
|
+
var SWARM_FINDING_CHAIN_STAGE_BASES = [
|
|
24415
|
+
"derived",
|
|
24416
|
+
"reported",
|
|
24417
|
+
"unmeasured"
|
|
24418
|
+
];
|
|
24419
|
+
var SWARM_FINDING_TONE_OF_DISPOSITION = Object.freeze({
|
|
24420
|
+
notRun: "muted",
|
|
24421
|
+
blockedConnecting: "fail",
|
|
24422
|
+
lostFindingTool: "fail",
|
|
24423
|
+
blockedCallingTool: "fail",
|
|
24424
|
+
blockedByResponse: "fail",
|
|
24425
|
+
goalMissed: "fail",
|
|
24426
|
+
goalMetWithFriction: "warn",
|
|
24427
|
+
goalMet: "ok",
|
|
24428
|
+
notMeasured: "muted"
|
|
24429
|
+
});
|
|
24430
|
+
var vocabulary = (members) => members.map((value) => "`" + value + "`").join(", ");
|
|
24431
|
+
var count = z2.number().int().nonnegative();
|
|
24432
|
+
var persona = z2.object({ personaRefId: z2.string().nullable(), name: z2.string() }).strict();
|
|
24433
|
+
var disposition = z2.enum(SWARM_FINDING_DISPOSITIONS).describe(vocabulary(SWARM_FINDING_DISPOSITIONS));
|
|
24434
|
+
var coverageNotes = z2.array(
|
|
24435
|
+
z2.enum(SWARM_FINDING_COVERAGE_NOTES).describe(vocabulary(SWARM_FINDING_COVERAGE_NOTES))
|
|
24436
|
+
);
|
|
24437
|
+
var citations = z2.array(z2.string().regex(/^[^/]+\/.+$/)).max(30);
|
|
24438
|
+
var toneMatchesDisposition = (row2) => row2.tone === SWARM_FINDING_TONE_OF_DISPOSITION[row2.disposition];
|
|
24439
|
+
var toneMismatch = {
|
|
24440
|
+
message: "tone must equal SWARM_FINDING_TONE_OF_DISPOSITION[disposition]",
|
|
24441
|
+
path: ["tone"]
|
|
24442
|
+
};
|
|
24443
|
+
var swarmJourneyFindingSchema = z2.object({
|
|
24444
|
+
id: z2.string().min(1),
|
|
24445
|
+
basis: z2.enum(SWARM_FINDING_BASES).describe(vocabulary(SWARM_FINDING_BASES)),
|
|
24446
|
+
scopeLevel: z2.enum(SWARM_FINDING_SCOPE_LEVELS).describe(vocabulary(SWARM_FINDING_SCOPE_LEVELS)),
|
|
24447
|
+
persona,
|
|
24448
|
+
goal: z2.object({
|
|
24449
|
+
runId: z2.string(),
|
|
24450
|
+
journeyRefId: z2.string(),
|
|
24451
|
+
title: z2.string()
|
|
24452
|
+
}).strict(),
|
|
24453
|
+
target: z2.object({
|
|
24454
|
+
kind: z2.enum(["environment", "host"]),
|
|
24455
|
+
id: z2.string(),
|
|
24456
|
+
label: z2.string(),
|
|
24457
|
+
modelId: z2.string().nullable()
|
|
24458
|
+
}).strict(),
|
|
24459
|
+
population: z2.object({ count, total: count, unit: z2.literal("sessions") }).strict(),
|
|
24460
|
+
sessionIds: z2.array(z2.string()).max(1e3),
|
|
24461
|
+
citations,
|
|
24462
|
+
verdictSeen: swarmSessionVerdictValueSchema,
|
|
24463
|
+
chainStage: userValueStageSchema.nullable().describe(vocabulary(USER_VALUE_STAGES)),
|
|
24464
|
+
chainStageState: stageStateSchema.nullable().describe(vocabulary(STAGE_STATES)),
|
|
24465
|
+
chainStageBasis: z2.enum(SWARM_FINDING_CHAIN_STAGE_BASES),
|
|
24466
|
+
disposition,
|
|
24467
|
+
tone: z2.enum(SWARM_FINDING_TONES),
|
|
24468
|
+
coverageNotes,
|
|
24469
|
+
outcomePhrase: z2.string().max(100).regex(/^[^0-9]*$/).refine(
|
|
24470
|
+
(value) => value.trim().split(/\s+/).length <= 8,
|
|
24471
|
+
"At most eight words"
|
|
24472
|
+
).nullable(),
|
|
24473
|
+
mechanismPhrase: z2.string().nullable(),
|
|
24474
|
+
fixPhrase: z2.string().nullable(),
|
|
24475
|
+
reportExcerpt: z2.object({ actual: z2.string().max(1800), citations }).strict().nullable(),
|
|
24476
|
+
mechanismId: z2.string().nullable()
|
|
24477
|
+
}).strict().refine(toneMatchesDisposition, toneMismatch);
|
|
24478
|
+
z2.object({
|
|
24479
|
+
contractVersion: z2.literal(SWARM_FINDING_CONTRACT_VERSION),
|
|
24480
|
+
generatedAt: count,
|
|
24481
|
+
sourceRevision: z2.string(),
|
|
24482
|
+
pipelineVersion: count,
|
|
24483
|
+
extractionVersion: count,
|
|
24484
|
+
extractionModel: z2.string(),
|
|
24485
|
+
reasoningModel: z2.string(),
|
|
24486
|
+
summaryKind: z2.enum(SWARM_FINDING_SUMMARY_KINDS).describe(vocabulary(SWARM_FINDING_SUMMARY_KINDS)),
|
|
24487
|
+
population: z2.object({
|
|
24488
|
+
configured: count,
|
|
24489
|
+
started: count,
|
|
24490
|
+
read: count,
|
|
24491
|
+
unread: count,
|
|
24492
|
+
withdrawn: count,
|
|
24493
|
+
limited: count,
|
|
24494
|
+
graded: count
|
|
24495
|
+
}).strict(),
|
|
24496
|
+
coverageNotes,
|
|
24497
|
+
disclosure: z2.object({
|
|
24498
|
+
rail: z2.enum(["gateway", "openrouter"]),
|
|
24499
|
+
evidenceSent: z2.array(z2.string())
|
|
24500
|
+
}).strict(),
|
|
24501
|
+
personas: z2.array(
|
|
24502
|
+
z2.object({
|
|
24503
|
+
persona,
|
|
24504
|
+
disposition,
|
|
24505
|
+
tone: z2.enum(SWARM_FINDING_TONES),
|
|
24506
|
+
goalRunIds: z2.array(z2.string())
|
|
24507
|
+
}).strict().refine(toneMatchesDisposition, toneMismatch)
|
|
24508
|
+
),
|
|
24509
|
+
findings: z2.array(swarmJourneyFindingSchema).max(200)
|
|
24510
|
+
}).strict();
|
|
24511
|
+
z2.object({
|
|
24512
|
+
status: z2.enum(["pending", "completed", "failed", "skipped"]),
|
|
24513
|
+
errorCode: z2.string().optional(),
|
|
24514
|
+
updatedAt: count
|
|
24515
|
+
}).strict();
|
|
24148
24516
|
var USER_VALUE_STAGE_LABELS = Object.freeze({
|
|
24149
24517
|
connection: "Connection",
|
|
24150
24518
|
discovery: "Discovery",
|
|
@@ -25683,248 +26051,28 @@ var evalBacktestContinuationSchema = z2.object({
|
|
|
25683
26051
|
evalBacktestDraftSchema.extend({
|
|
25684
26052
|
continuation: evalBacktestContinuationSchema.optional()
|
|
25685
26053
|
});
|
|
25686
|
-
var SWARM_SESSION_VERDICT_CONTRACT_VERSION = 1;
|
|
25687
|
-
var SWARM_ATTEMPT_STATUSES = [
|
|
25688
|
-
"pending",
|
|
25689
|
-
"running",
|
|
25690
|
-
"succeeded",
|
|
25691
|
-
"failed",
|
|
25692
|
-
"rate_limited"
|
|
25693
|
-
];
|
|
25694
|
-
var SWARM_SESSION_LIFECYCLES = [
|
|
25695
|
-
"pending",
|
|
25696
|
-
"running",
|
|
25697
|
-
"ran",
|
|
25698
|
-
"broke",
|
|
25699
|
-
"limited",
|
|
25700
|
-
"withdrawn"
|
|
25701
|
-
];
|
|
25702
|
-
var SWARM_SESSION_VERDICTS = [
|
|
25703
|
-
"passed",
|
|
25704
|
-
"failed",
|
|
25705
|
-
"inconclusive",
|
|
25706
|
-
"notEstablished"
|
|
25707
|
-
];
|
|
25708
|
-
var SWARM_GRADING_STATES = [
|
|
25709
|
-
"notRequested",
|
|
25710
|
-
"queued",
|
|
25711
|
-
"running",
|
|
25712
|
-
"settled",
|
|
25713
|
-
"unavailable"
|
|
25714
|
-
];
|
|
25715
|
-
var SWARM_SESSION_VERDICT_OF_REASON = {
|
|
25716
|
-
attemptPending: "notEstablished",
|
|
25717
|
-
attemptRunning: "notEstablished",
|
|
25718
|
-
withdrawn: "notEstablished",
|
|
25719
|
-
spendCapReached: "notEstablished",
|
|
25720
|
-
notRun: "notEstablished",
|
|
25721
|
-
executionFailed: "notEstablished",
|
|
25722
|
-
ungraded: "notEstablished",
|
|
25723
|
-
gradingNotClaimed: "notEstablished",
|
|
25724
|
-
criteriaPending: "notEstablished",
|
|
25725
|
-
judgePending: "notEstablished",
|
|
25726
|
-
gatingCriterionFailed: "failed",
|
|
25727
|
-
judgeFailed: "failed",
|
|
25728
|
-
criteriaGradingErrored: "inconclusive",
|
|
25729
|
-
gatingCriterionUnmeasured: "inconclusive",
|
|
25730
|
-
judgeErrored: "inconclusive",
|
|
25731
|
-
gradingUnavailable: "inconclusive",
|
|
25732
|
-
allGatingCriteriaPassed: "passed",
|
|
25733
|
-
judgePassed: "passed",
|
|
25734
|
-
allGradersPassed: "passed"
|
|
25735
|
-
};
|
|
25736
|
-
var SWARM_SESSION_VERDICT_REASONS = Object.keys(
|
|
25737
|
-
SWARM_SESSION_VERDICT_OF_REASON
|
|
25738
|
-
);
|
|
25739
|
-
var swarmSessionLifecycleSchema = z2.enum(SWARM_SESSION_LIFECYCLES);
|
|
25740
|
-
var swarmSessionVerdictValueSchema = z2.enum(SWARM_SESSION_VERDICTS);
|
|
25741
|
-
var swarmGraderRoleSchema = z2.enum(["advisory", "required"]);
|
|
25742
|
-
var swarmSessionAttemptInputSchema = z2.object({
|
|
25743
|
-
status: z2.enum(SWARM_ATTEMPT_STATUSES),
|
|
25744
|
-
errorCode: z2.string().nullable().optional()
|
|
25745
|
-
}).strict();
|
|
25746
|
-
var swarmGradingReadinessSchema = z2.object({
|
|
25747
|
-
/** Durable readiness for decisive grading only, not background observations. */
|
|
25748
|
-
state: z2.enum(SWARM_GRADING_STATES),
|
|
25749
|
-
reasonCode: z2.string().min(1).optional()
|
|
25750
|
-
}).strict();
|
|
25751
|
-
var criterionResultSchema = z2.object({
|
|
25752
|
-
criterionId: z2.string().min(1),
|
|
25753
|
-
passed: z2.boolean(),
|
|
25754
|
-
status: z2.enum(["scored", "error"]).optional()
|
|
25755
|
-
}).strict();
|
|
25756
|
-
z2.object({
|
|
25757
|
-
attempt: swarmSessionAttemptInputSchema.nullable(),
|
|
25758
|
-
hasTranscript: z2.boolean(),
|
|
25759
|
-
rubric: z2.array(
|
|
25760
|
-
z2.object({
|
|
25761
|
-
id: z2.string().min(1),
|
|
25762
|
-
role: swarmGraderRoleSchema,
|
|
25763
|
-
predicateType: z2.string().min(1).optional()
|
|
25764
|
-
}).strict()
|
|
25765
|
-
),
|
|
25766
|
-
criteria: z2.object({
|
|
25767
|
-
status: z2.enum(["pending", "completed", "failed"]),
|
|
25768
|
-
criterionIds: z2.array(z2.string().min(1)).optional(),
|
|
25769
|
-
results: z2.array(criterionResultSchema).optional()
|
|
25770
|
-
}).strict().nullable(),
|
|
25771
|
-
goalScore: z2.discriminatedUnion("status", [
|
|
25772
|
-
z2.object({ status: z2.literal("completed"), passed: z2.boolean() }).strict(),
|
|
25773
|
-
z2.object({ status: z2.literal("running") }).strict(),
|
|
25774
|
-
z2.object({ status: z2.literal("failed") }).strict()
|
|
25775
|
-
]).nullable(),
|
|
25776
|
-
judge: z2.object({
|
|
25777
|
-
automatic: z2.boolean(),
|
|
25778
|
-
role: swarmGraderRoleSchema,
|
|
25779
|
-
/** Explicit on-demand intent; automatic:false alone does not mean a judge was requested. */
|
|
25780
|
-
requested: z2.boolean().optional()
|
|
25781
|
-
}).strict(),
|
|
25782
|
-
grading: swarmGradingReadinessSchema
|
|
25783
|
-
}).strict().superRefine((value, ctx) => {
|
|
25784
|
-
const definitions = new Set(value.rubric.map((entry22) => entry22.id));
|
|
25785
|
-
if (definitions.size !== value.rubric.length) {
|
|
25786
|
-
ctx.addIssue({
|
|
25787
|
-
code: "custom",
|
|
25788
|
-
path: ["rubric"],
|
|
25789
|
-
message: "Duplicate criterion definition"
|
|
25790
|
-
});
|
|
25791
|
-
}
|
|
25792
|
-
for (const [field2, ids] of [
|
|
25793
|
-
["criterionIds", value.criteria?.criterionIds ?? []],
|
|
25794
|
-
["results", value.criteria?.results?.map((row2) => row2.criterionId) ?? []]
|
|
25795
|
-
]) {
|
|
25796
|
-
if (new Set(ids).size !== ids.length || ids.some((id3) => !definitions.has(id3))) {
|
|
25797
|
-
ctx.addIssue({
|
|
25798
|
-
code: "custom",
|
|
25799
|
-
path: ["criteria", field2],
|
|
25800
|
-
message: "Criterion IDs must be unique and defined in the snapshot"
|
|
25801
|
-
});
|
|
25802
|
-
}
|
|
25803
|
-
}
|
|
25804
|
-
const claimed = value.criteria?.criterionIds;
|
|
25805
|
-
if (claimed && value.criteria?.results?.some((row2) => !claimed.includes(row2.criterionId))) {
|
|
25806
|
-
ctx.addIssue({
|
|
25807
|
-
code: "custom",
|
|
25808
|
-
path: ["criteria", "results"],
|
|
25809
|
-
message: "Result outside claimed scope"
|
|
25810
|
-
});
|
|
25811
|
-
}
|
|
25812
|
-
});
|
|
25813
|
-
var SWARM_SESSION_TRIAL_STATUSES = [
|
|
25814
|
-
"pending",
|
|
25815
|
-
"running",
|
|
25816
|
-
"completed",
|
|
25817
|
-
"failed",
|
|
25818
|
-
"setup_failed",
|
|
25819
|
-
"cancelled"
|
|
25820
|
-
];
|
|
25821
|
-
var swarmSessionTrialSchema = z2.object({
|
|
25822
|
-
status: z2.enum(SWARM_SESSION_TRIAL_STATUSES),
|
|
25823
|
-
taskVerdict: z2.enum(["passed", "failed"]).optional(),
|
|
25824
|
-
evaluatorError: z2.literal(true).optional()
|
|
25825
|
-
}).strict().superRefine((value, ctx) => {
|
|
25826
|
-
const verdict = value.taskVerdict !== void 0;
|
|
25827
|
-
const error = value.evaluatorError === true;
|
|
25828
|
-
if (value.status === "completed" ? verdict === error : verdict || error) {
|
|
25829
|
-
ctx.addIssue({
|
|
25830
|
-
code: "custom",
|
|
25831
|
-
message: "Only completed trials carry exactly one verdict or evaluator error"
|
|
25832
|
-
});
|
|
25833
|
-
}
|
|
25834
|
-
});
|
|
25835
|
-
var swarmSessionGraderCountsSchema = z2.object({
|
|
25836
|
-
gating: z2.number().int().nonnegative(),
|
|
25837
|
-
gatingPassed: z2.number().int().nonnegative(),
|
|
25838
|
-
gatingFailed: z2.number().int().nonnegative(),
|
|
25839
|
-
gatingUnmeasured: z2.number().int().nonnegative(),
|
|
25840
|
-
advisoryFailed: z2.number().int().nonnegative()
|
|
25841
|
-
}).strict();
|
|
25842
|
-
var swarmSessionVerdictSchema = z2.object({
|
|
25843
|
-
contractVersion: z2.literal(SWARM_SESSION_VERDICT_CONTRACT_VERSION),
|
|
25844
|
-
lifecycle: swarmSessionLifecycleSchema,
|
|
25845
|
-
verdict: swarmSessionVerdictValueSchema,
|
|
25846
|
-
reason: z2.enum(SWARM_SESSION_VERDICT_REASONS),
|
|
25847
|
-
verdictSource: z2.enum([
|
|
25848
|
-
"goalJudge",
|
|
25849
|
-
"requiredAssertions",
|
|
25850
|
-
"combined",
|
|
25851
|
-
"none"
|
|
25852
|
-
]),
|
|
25853
|
-
grading: swarmGradingReadinessSchema,
|
|
25854
|
-
graders: z2.object({
|
|
25855
|
-
criteria: z2.enum([
|
|
25856
|
-
"notConfigured",
|
|
25857
|
-
"notClaimed",
|
|
25858
|
-
"pending",
|
|
25859
|
-
"scored",
|
|
25860
|
-
"errored"
|
|
25861
|
-
]),
|
|
25862
|
-
judge: z2.enum([
|
|
25863
|
-
"notConfigured",
|
|
25864
|
-
"silent",
|
|
25865
|
-
"pending",
|
|
25866
|
-
"scored",
|
|
25867
|
-
"errored"
|
|
25868
|
-
])
|
|
25869
|
-
}).strict(),
|
|
25870
|
-
counts: swarmSessionGraderCountsSchema,
|
|
25871
|
-
trial: swarmSessionTrialSchema.nullable()
|
|
25872
|
-
}).strict().superRefine((value, ctx) => {
|
|
25873
|
-
const fail3 = (message) => ctx.addIssue({ code: "custom", message });
|
|
25874
|
-
if (SWARM_SESSION_VERDICT_OF_REASON[value.reason] !== value.verdict)
|
|
25875
|
-
fail3("Reason contradicts verdict");
|
|
25876
|
-
const c = value.counts;
|
|
25877
|
-
if (c.gating !== c.gatingPassed + c.gatingFailed + c.gatingUnmeasured)
|
|
25878
|
-
fail3("Required measurement counts do not add up");
|
|
25879
|
-
const trial = value.trial;
|
|
25880
|
-
if (trial?.status === "completed") {
|
|
25881
|
-
if (value.lifecycle !== "ran")
|
|
25882
|
-
fail3("Only a completed execution can be a completed trial");
|
|
25883
|
-
if (trial.taskVerdict !== void 0 && trial.taskVerdict !== value.verdict)
|
|
25884
|
-
fail3("Trial contradicts goal result");
|
|
25885
|
-
if (trial.evaluatorError && value.verdict !== "inconclusive")
|
|
25886
|
-
fail3("Evaluator error requires inconclusive grading");
|
|
25887
|
-
}
|
|
25888
|
-
const expected = {
|
|
25889
|
-
pending: "pending",
|
|
25890
|
-
running: "running",
|
|
25891
|
-
broke: "failed",
|
|
25892
|
-
limited: "setup_failed",
|
|
25893
|
-
withdrawn: "cancelled"
|
|
25894
|
-
};
|
|
25895
|
-
if (value.lifecycle !== "ran" && trial?.status !== expected[value.lifecycle])
|
|
25896
|
-
fail3("Trial must preserve execution lifecycle");
|
|
25897
|
-
if (value.lifecycle === "ran" && trial !== null && trial.status !== "completed")
|
|
25898
|
-
fail3("Completed execution must not become skipped or running");
|
|
25899
|
-
if ((value.verdict === "passed" || value.verdict === "failed") && value.verdictSource === "none")
|
|
25900
|
-
fail3("A measured goal verdict needs a source");
|
|
25901
|
-
if (value.lifecycle === "ran" && value.verdict !== "notEstablished" && trial === null)
|
|
25902
|
-
fail3("Settled grading on completed execution needs an observation");
|
|
25903
|
-
if (value.verdict === "notEstablished" && value.verdictSource !== "none")
|
|
25904
|
-
fail3("An undecided goal has no verdict source");
|
|
25905
|
-
});
|
|
25906
26054
|
var SWARM_REPORT_CONTRACT_VERSION = 1;
|
|
25907
|
-
var
|
|
26055
|
+
var count2 = z2.number().int().nonnegative();
|
|
25908
26056
|
var journeyRunVerdictSummarySchema = z2.discriminatedUnion("status", [
|
|
25909
26057
|
z2.object({
|
|
25910
26058
|
status: z2.literal("pending"),
|
|
25911
|
-
pendingSessions:
|
|
25912
|
-
updatedAt:
|
|
26059
|
+
pendingSessions: count2,
|
|
26060
|
+
updatedAt: count2
|
|
25913
26061
|
}).strict(),
|
|
25914
26062
|
z2.object({
|
|
25915
26063
|
status: z2.literal("decided"),
|
|
25916
26064
|
decision: evalVerdictDecisionSchema,
|
|
25917
|
-
updatedAt:
|
|
26065
|
+
updatedAt: count2
|
|
25918
26066
|
}).strict(),
|
|
25919
26067
|
z2.object({
|
|
25920
26068
|
status: z2.literal("notEstablished"),
|
|
25921
26069
|
reason: z2.literal("gradingNotConfigured"),
|
|
25922
|
-
updatedAt:
|
|
26070
|
+
updatedAt: count2
|
|
25923
26071
|
}).strict(),
|
|
25924
26072
|
z2.object({
|
|
25925
26073
|
status: z2.literal("integrityFailed"),
|
|
25926
26074
|
reason: z2.string().min(1),
|
|
25927
|
-
updatedAt:
|
|
26075
|
+
updatedAt: count2
|
|
25928
26076
|
}).strict()
|
|
25929
26077
|
]);
|
|
25930
26078
|
var swarmObservationInputSchema = z2.object({
|
|
@@ -25962,7 +26110,7 @@ var swarmReportSessionSchema = z2.object({
|
|
|
25962
26110
|
z2.object({
|
|
25963
26111
|
runId: z2.string().min(1),
|
|
25964
26112
|
executionComplete: z2.boolean(),
|
|
25965
|
-
configuredSessions:
|
|
26113
|
+
configuredSessions: count2,
|
|
25966
26114
|
verdictSummary: journeyRunVerdictSummarySchema.nullable(),
|
|
25967
26115
|
/** Frozen run-wide definitions: absent measurement rows still count as unavailable. */
|
|
25968
26116
|
evaluatorDefinitions: z2.array(
|
|
@@ -26017,11 +26165,11 @@ var swarmObservationCoverageSchema = z2.object({
|
|
|
26017
26165
|
"The stage this observation was read from, and the only stage it is evidence ABOUT. `connection` \u2014 the server was reachable and the session initialized. `discovery` \u2014 its tools and resources were listed and readable. `selection` \u2014 the model chose the right tool for the request. `call` \u2014 the call was made with usable arguments. `response` \u2014 the server returned data the model could use. `userValue` \u2014 the user's actual request was satisfied."
|
|
26018
26166
|
),
|
|
26019
26167
|
unit: z2.literal("sessions"),
|
|
26020
|
-
total:
|
|
26021
|
-
passed:
|
|
26022
|
-
failed:
|
|
26023
|
-
pending:
|
|
26024
|
-
unavailable:
|
|
26168
|
+
total: count2,
|
|
26169
|
+
passed: count2,
|
|
26170
|
+
failed: count2,
|
|
26171
|
+
pending: count2,
|
|
26172
|
+
unavailable: count2
|
|
26025
26173
|
}).strict().superRefine((value, ctx) => {
|
|
26026
26174
|
if (value.total !== value.passed + value.failed + value.pending + value.unavailable) {
|
|
26027
26175
|
ctx.addIssue({
|
|
@@ -26051,26 +26199,26 @@ z2.object({
|
|
|
26051
26199
|
]).optional(),
|
|
26052
26200
|
execution: z2.object({
|
|
26053
26201
|
unit: z2.literal("sessions"),
|
|
26054
|
-
configured:
|
|
26055
|
-
reported:
|
|
26056
|
-
started:
|
|
26057
|
-
notStarted:
|
|
26058
|
-
unknown:
|
|
26059
|
-
completed:
|
|
26060
|
-
interrupted:
|
|
26202
|
+
configured: count2,
|
|
26203
|
+
reported: count2,
|
|
26204
|
+
started: count2,
|
|
26205
|
+
notStarted: count2,
|
|
26206
|
+
unknown: count2,
|
|
26207
|
+
completed: count2,
|
|
26208
|
+
interrupted: count2,
|
|
26061
26209
|
/** Only explicit complete execution evidence can establish this. */
|
|
26062
26210
|
neverLaunched: z2.boolean()
|
|
26063
26211
|
}).strict(),
|
|
26064
26212
|
goalGrading: z2.object({
|
|
26065
26213
|
unit: z2.literal("sessions"),
|
|
26066
|
-
reported:
|
|
26067
|
-
passed:
|
|
26068
|
-
failed:
|
|
26069
|
-
pending:
|
|
26070
|
-
unavailable:
|
|
26071
|
-
notRequested:
|
|
26214
|
+
reported: count2,
|
|
26215
|
+
passed: count2,
|
|
26216
|
+
failed: count2,
|
|
26217
|
+
pending: count2,
|
|
26218
|
+
unavailable: count2,
|
|
26219
|
+
notRequested: count2,
|
|
26072
26220
|
/** Readiness is separate: a proven failure can exist while other required grading is pending. */
|
|
26073
|
-
waitingForDecisiveGrading:
|
|
26221
|
+
waitingForDecisiveGrading: count2
|
|
26074
26222
|
}).strict(),
|
|
26075
26223
|
observations: z2.array(swarmObservationCoverageSchema)
|
|
26076
26224
|
}).strict().superRefine((report, ctx) => {
|
|
@@ -26176,7 +26324,7 @@ var noopPosthog = {
|
|
|
26176
26324
|
var posthog = isTelemetryDisabled ? noopPosthog : new PostHog("phc_dTOPniyUNU2kD8Jx8yHMXSqiZHM8I91uWopTMX6EBE9", {
|
|
26177
26325
|
host: "https://us.i.posthog.com"
|
|
26178
26326
|
});
|
|
26179
|
-
var SDK_VERSION = "8.
|
|
26327
|
+
var SDK_VERSION = "8.11.0";
|
|
26180
26328
|
var SDK_RELEASE = `@mcpjam/sdk@${SDK_VERSION}`;
|
|
26181
26329
|
var CAPTURED_ERROR_SYMBOL = Symbol.for("@mcpjam/sdk/captured-eval-error");
|
|
26182
26330
|
init_internal();
|
|
@@ -26187,7 +26335,7 @@ init_eval_tool_execution();
|
|
|
26187
26335
|
init_internal();
|
|
26188
26336
|
init_internal();
|
|
26189
26337
|
function readSdkVersion() {
|
|
26190
|
-
return "8.
|
|
26338
|
+
return "8.11.0";
|
|
26191
26339
|
}
|
|
26192
26340
|
var DEFAULT_PLATFORM_USER_AGENT = `mcpjam-sdk/${readSdkVersion()}`;
|
|
26193
26341
|
init_HostRunner();
|
|
@@ -41192,7 +41340,7 @@ async function runTransportChecks(ctx, selectedCheckIds) {
|
|
|
41192
41340
|
const readErrors = sseResponses.map(
|
|
41193
41341
|
({ response }) => response.bodyError
|
|
41194
41342
|
);
|
|
41195
|
-
const allStreamsReadable = requestErrors.every((error) => error === void 0) && !responseFailures && readErrors.every((error) => error === void 0) && eventCounts.every((
|
|
41343
|
+
const allStreamsReadable = requestErrors.every((error) => error === void 0) && !responseFailures && readErrors.every((error) => error === void 0) && eventCounts.every((count32) => count32 > 0);
|
|
41196
41344
|
results.push(
|
|
41197
41345
|
allStreamsReadable ? passedResult2(
|
|
41198
41346
|
TRANSPORT_CHECK_METADATA["server-sse-streams-functional"],
|
|
@@ -73552,17 +73700,8 @@ function resolveSlug(error) {
|
|
|
73552
73700
|
if (/missing\s+(?:or\s+invalid\s+)?bearer/i.test(message) || /bearer\s+token\s+(?:is\s+)?required/i.test(message)) {
|
|
73553
73701
|
return { slug: "auth/missing_bearer" };
|
|
73554
73702
|
}
|
|
73555
|
-
const
|
|
73556
|
-
|
|
73557
|
-
);
|
|
73558
|
-
if (limitPeriod) {
|
|
73559
|
-
return {
|
|
73560
|
-
slug: limitPeriod[1].toLowerCase() === "monthly" ? "provider/mcpjam_limit_monthly" : "provider/mcpjam_limit_daily"
|
|
73561
|
-
};
|
|
73562
|
-
}
|
|
73563
|
-
if (/mcpjam[\w\s-]{0,40}model limit/i.test(message)) {
|
|
73564
|
-
return { slug: "provider/mcpjam_limit" };
|
|
73565
|
-
}
|
|
73703
|
+
const limitSlug = mcpjamLimitSlugForMessage(message);
|
|
73704
|
+
if (limitSlug) return { slug: limitSlug };
|
|
73566
73705
|
const httpStatus2 = getHttpStatus(error);
|
|
73567
73706
|
if (httpStatus2 !== void 0) {
|
|
73568
73707
|
const slug = classifyHttpStatus(httpStatus2);
|
|
@@ -73743,6 +73882,16 @@ function crashFallback(error, emptyPlaceholder) {
|
|
|
73743
73882
|
rawMessage
|
|
73744
73883
|
};
|
|
73745
73884
|
}
|
|
73885
|
+
function mcpjamLimitSlugForMessage(message) {
|
|
73886
|
+
const limitPeriod = /\b(daily|monthly)\s+mcpjam[\w\s-]{0,40}model limit/i.exec(message);
|
|
73887
|
+
if (limitPeriod) {
|
|
73888
|
+
return limitPeriod[1].toLowerCase() === "monthly" ? "provider/mcpjam_limit_monthly" : "provider/mcpjam_limit_daily";
|
|
73889
|
+
}
|
|
73890
|
+
if (/mcpjam[\w\s-]{0,40}model limit/i.test(message)) {
|
|
73891
|
+
return "provider/mcpjam_limit";
|
|
73892
|
+
}
|
|
73893
|
+
return void 0;
|
|
73894
|
+
}
|
|
73746
73895
|
var MAX_FIELD_CHARS2 = 120;
|
|
73747
73896
|
function clip(value) {
|
|
73748
73897
|
value = String(redactForTelemetry(value));
|
|
@@ -82916,7 +83065,7 @@ var NEGATIVE_TEST_MODE_DETAILS = {
|
|
|
82916
83065
|
}
|
|
82917
83066
|
};
|
|
82918
83067
|
function readSdkVersion2() {
|
|
82919
|
-
return "8.
|
|
83068
|
+
return "8.11.0";
|
|
82920
83069
|
}
|
|
82921
83070
|
var CONFORMANCE_CHECKER_VERSION2 = readSdkVersion2();
|
|
82922
83071
|
var MUST_POINTS2 = 95;
|
|
@@ -86966,6 +87115,9 @@ function backoffDelayMs(attempt, policy) {
|
|
|
86966
87115
|
);
|
|
86967
87116
|
return Math.max(0, Math.round((policy.jitter ?? defaultJitter)(exponential)));
|
|
86968
87117
|
}
|
|
87118
|
+
function clampDelay(value, low, high) {
|
|
87119
|
+
return Math.min(high, Math.max(low, value));
|
|
87120
|
+
}
|
|
86969
87121
|
function abortableSleep(ms, signal) {
|
|
86970
87122
|
if (signal?.aborted || ms <= 0) return Promise.resolve();
|
|
86971
87123
|
return new Promise((resolve11) => {
|
|
@@ -89159,8 +89311,8 @@ var CommandQueue = class {
|
|
|
89159
89311
|
}
|
|
89160
89312
|
/** Whether every per-tab FIFO is drained. Used for profile snapshots. */
|
|
89161
89313
|
isIdle() {
|
|
89162
|
-
for (const
|
|
89163
|
-
if (
|
|
89314
|
+
for (const count5 of this.depth.values()) {
|
|
89315
|
+
if (count5 > 0) return false;
|
|
89164
89316
|
}
|
|
89165
89317
|
return true;
|
|
89166
89318
|
}
|
|
@@ -94170,14 +94322,14 @@ function assignRefs(root) {
|
|
|
94170
94322
|
const entries = /* @__PURE__ */ new Map();
|
|
94171
94323
|
if (!root) return entries;
|
|
94172
94324
|
const seen = /* @__PURE__ */ new Map();
|
|
94173
|
-
const
|
|
94325
|
+
const count5 = (node) => {
|
|
94174
94326
|
if (isRefWorthy(node)) {
|
|
94175
94327
|
const key = `${node.role}:${node.name ?? ""}`;
|
|
94176
94328
|
seen.set(key, (seen.get(key) ?? 0) + 1);
|
|
94177
94329
|
}
|
|
94178
|
-
for (const child of node.children ?? [])
|
|
94330
|
+
for (const child of node.children ?? []) count5(child);
|
|
94179
94331
|
};
|
|
94180
|
-
|
|
94332
|
+
count5(root);
|
|
94181
94333
|
const position = /* @__PURE__ */ new Map();
|
|
94182
94334
|
let next = 1;
|
|
94183
94335
|
const visit2 = (node) => {
|
|
@@ -94455,7 +94607,7 @@ var WebMcpBridge = class {
|
|
|
94455
94607
|
}
|
|
94456
94608
|
return false;
|
|
94457
94609
|
}
|
|
94458
|
-
allowChanges(
|
|
94610
|
+
allowChanges(count5) {
|
|
94459
94611
|
if (!this.localSecurity) return true;
|
|
94460
94612
|
if (this.discoveryBlocked) return false;
|
|
94461
94613
|
const now = Date.now();
|
|
@@ -94465,8 +94617,8 @@ var WebMcpBridge = class {
|
|
|
94465
94617
|
rate.tokens + Math.max(0, now - rate.at) * 1.024
|
|
94466
94618
|
);
|
|
94467
94619
|
rate.at = now;
|
|
94468
|
-
if (
|
|
94469
|
-
rate.tokens -=
|
|
94620
|
+
if (count5 > rate.tokens) return this.blockDiscovery();
|
|
94621
|
+
rate.tokens -= count5;
|
|
94470
94622
|
this.changeTokens = rate.tokens;
|
|
94471
94623
|
this.changeAt = now;
|
|
94472
94624
|
return true;
|
|
@@ -102463,8 +102615,8 @@ function createFrameRelayStats(options) {
|
|
|
102463
102615
|
countDrop(n = 1) {
|
|
102464
102616
|
dropped += Math.max(0, Math.round(n));
|
|
102465
102617
|
},
|
|
102466
|
-
setSubscribers(
|
|
102467
|
-
subscribers3 = Math.max(0, Math.round(
|
|
102618
|
+
setSubscribers(count5) {
|
|
102619
|
+
subscribers3 = Math.max(0, Math.round(count5));
|
|
102468
102620
|
},
|
|
102469
102621
|
mergeDaemon(counters) {
|
|
102470
102622
|
daemon = counters;
|
|
@@ -125646,7 +125798,7 @@ function cloneTraceValue(value) {
|
|
|
125646
125798
|
}
|
|
125647
125799
|
function getPromptIndex(messageHistory) {
|
|
125648
125800
|
const userCount = messageHistory.reduce(
|
|
125649
|
-
(
|
|
125801
|
+
(count5, message) => count5 + (message?.role === "user" ? 1 : 0),
|
|
125650
125802
|
0
|
|
125651
125803
|
);
|
|
125652
125804
|
return Math.max(0, userCount - 1);
|
|
@@ -130521,6 +130673,375 @@ function projectSuspectedConditionVerdict(metadata) {
|
|
|
130521
130673
|
if (parsed.success) return { suspectedConditionVerdict: parsed.data };
|
|
130522
130674
|
return { suspectedConditionUnverified: true };
|
|
130523
130675
|
}
|
|
130676
|
+
var SWARM_SESSION_VERDICT_CONTRACT_VERSION2 = 1;
|
|
130677
|
+
var SWARM_STAGE_EVIDENCE_VERSION = 2;
|
|
130678
|
+
var SWARM_ATTEMPT_STATUSES2 = [
|
|
130679
|
+
"pending",
|
|
130680
|
+
"running",
|
|
130681
|
+
"succeeded",
|
|
130682
|
+
"failed",
|
|
130683
|
+
"rate_limited"
|
|
130684
|
+
];
|
|
130685
|
+
var SWARM_SESSION_LIFECYCLES2 = [
|
|
130686
|
+
"pending",
|
|
130687
|
+
"running",
|
|
130688
|
+
"ran",
|
|
130689
|
+
"broke",
|
|
130690
|
+
"limited",
|
|
130691
|
+
"withdrawn"
|
|
130692
|
+
];
|
|
130693
|
+
var SWARM_SESSION_VERDICTS2 = [
|
|
130694
|
+
"passed",
|
|
130695
|
+
"failed",
|
|
130696
|
+
"inconclusive",
|
|
130697
|
+
"notEstablished"
|
|
130698
|
+
];
|
|
130699
|
+
var SWARM_GRADING_STATES2 = [
|
|
130700
|
+
"notRequested",
|
|
130701
|
+
"queued",
|
|
130702
|
+
"running",
|
|
130703
|
+
"settled",
|
|
130704
|
+
"unavailable"
|
|
130705
|
+
];
|
|
130706
|
+
var SWARM_SESSION_VERDICT_OF_REASON2 = {
|
|
130707
|
+
attemptPending: "notEstablished",
|
|
130708
|
+
attemptRunning: "notEstablished",
|
|
130709
|
+
withdrawn: "notEstablished",
|
|
130710
|
+
spendCapReached: "notEstablished",
|
|
130711
|
+
notRun: "notEstablished",
|
|
130712
|
+
executionFailed: "notEstablished",
|
|
130713
|
+
ungraded: "notEstablished",
|
|
130714
|
+
gradingNotClaimed: "notEstablished",
|
|
130715
|
+
criteriaPending: "notEstablished",
|
|
130716
|
+
judgePending: "notEstablished",
|
|
130717
|
+
gatingCriterionFailed: "failed",
|
|
130718
|
+
judgeFailed: "failed",
|
|
130719
|
+
criteriaGradingErrored: "inconclusive",
|
|
130720
|
+
gatingCriterionUnmeasured: "inconclusive",
|
|
130721
|
+
judgeErrored: "inconclusive",
|
|
130722
|
+
gradingUnavailable: "inconclusive",
|
|
130723
|
+
allGatingCriteriaPassed: "passed",
|
|
130724
|
+
judgePassed: "passed",
|
|
130725
|
+
allGradersPassed: "passed"
|
|
130726
|
+
};
|
|
130727
|
+
var SWARM_SESSION_VERDICT_REASONS2 = Object.keys(
|
|
130728
|
+
SWARM_SESSION_VERDICT_OF_REASON2
|
|
130729
|
+
);
|
|
130730
|
+
var swarmSessionLifecycleSchema2 = z14.enum(SWARM_SESSION_LIFECYCLES2);
|
|
130731
|
+
var swarmSessionVerdictValueSchema2 = z14.enum(SWARM_SESSION_VERDICTS2);
|
|
130732
|
+
var swarmGraderRoleSchema2 = z14.enum(["advisory", "required"]);
|
|
130733
|
+
var swarmSessionAttemptInputSchema2 = z14.object({
|
|
130734
|
+
status: z14.enum(SWARM_ATTEMPT_STATUSES2),
|
|
130735
|
+
errorCode: z14.string().nullable().optional()
|
|
130736
|
+
}).strict();
|
|
130737
|
+
var swarmGradingReadinessSchema2 = z14.object({
|
|
130738
|
+
/** Durable readiness for decisive grading only, not background observations. */
|
|
130739
|
+
state: z14.enum(SWARM_GRADING_STATES2),
|
|
130740
|
+
reasonCode: z14.string().min(1).optional()
|
|
130741
|
+
}).strict();
|
|
130742
|
+
var criterionResultSchema2 = z14.object({
|
|
130743
|
+
criterionId: z14.string().min(1),
|
|
130744
|
+
passed: z14.boolean(),
|
|
130745
|
+
status: z14.enum(["scored", "error"]).optional()
|
|
130746
|
+
}).strict();
|
|
130747
|
+
var swarmSessionVerdictInputSchema = z14.object({
|
|
130748
|
+
attempt: swarmSessionAttemptInputSchema2.nullable(),
|
|
130749
|
+
hasTranscript: z14.boolean(),
|
|
130750
|
+
rubric: z14.array(
|
|
130751
|
+
z14.object({
|
|
130752
|
+
id: z14.string().min(1),
|
|
130753
|
+
role: swarmGraderRoleSchema2,
|
|
130754
|
+
predicateType: z14.string().min(1).optional()
|
|
130755
|
+
}).strict()
|
|
130756
|
+
),
|
|
130757
|
+
criteria: z14.object({
|
|
130758
|
+
status: z14.enum(["pending", "completed", "failed"]),
|
|
130759
|
+
criterionIds: z14.array(z14.string().min(1)).optional(),
|
|
130760
|
+
results: z14.array(criterionResultSchema2).optional()
|
|
130761
|
+
}).strict().nullable(),
|
|
130762
|
+
goalScore: z14.discriminatedUnion("status", [
|
|
130763
|
+
z14.object({ status: z14.literal("completed"), passed: z14.boolean() }).strict(),
|
|
130764
|
+
z14.object({ status: z14.literal("running") }).strict(),
|
|
130765
|
+
z14.object({ status: z14.literal("failed") }).strict()
|
|
130766
|
+
]).nullable(),
|
|
130767
|
+
judge: z14.object({
|
|
130768
|
+
automatic: z14.boolean(),
|
|
130769
|
+
role: swarmGraderRoleSchema2,
|
|
130770
|
+
/** Explicit on-demand intent; automatic:false alone does not mean a judge was requested. */
|
|
130771
|
+
requested: z14.boolean().optional()
|
|
130772
|
+
}).strict(),
|
|
130773
|
+
grading: swarmGradingReadinessSchema2
|
|
130774
|
+
}).strict().superRefine((value, ctx) => {
|
|
130775
|
+
const definitions = new Set(value.rubric.map((entry3) => entry3.id));
|
|
130776
|
+
if (definitions.size !== value.rubric.length) {
|
|
130777
|
+
ctx.addIssue({
|
|
130778
|
+
code: "custom",
|
|
130779
|
+
path: ["rubric"],
|
|
130780
|
+
message: "Duplicate criterion definition"
|
|
130781
|
+
});
|
|
130782
|
+
}
|
|
130783
|
+
for (const [field, ids] of [
|
|
130784
|
+
["criterionIds", value.criteria?.criterionIds ?? []],
|
|
130785
|
+
["results", value.criteria?.results?.map((row2) => row2.criterionId) ?? []]
|
|
130786
|
+
]) {
|
|
130787
|
+
if (new Set(ids).size !== ids.length || ids.some((id3) => !definitions.has(id3))) {
|
|
130788
|
+
ctx.addIssue({
|
|
130789
|
+
code: "custom",
|
|
130790
|
+
path: ["criteria", field],
|
|
130791
|
+
message: "Criterion IDs must be unique and defined in the snapshot"
|
|
130792
|
+
});
|
|
130793
|
+
}
|
|
130794
|
+
}
|
|
130795
|
+
const claimed = value.criteria?.criterionIds;
|
|
130796
|
+
if (claimed && value.criteria?.results?.some((row2) => !claimed.includes(row2.criterionId))) {
|
|
130797
|
+
ctx.addIssue({
|
|
130798
|
+
code: "custom",
|
|
130799
|
+
path: ["criteria", "results"],
|
|
130800
|
+
message: "Result outside claimed scope"
|
|
130801
|
+
});
|
|
130802
|
+
}
|
|
130803
|
+
});
|
|
130804
|
+
var SWARM_SESSION_TRIAL_STATUSES2 = [
|
|
130805
|
+
"pending",
|
|
130806
|
+
"running",
|
|
130807
|
+
"completed",
|
|
130808
|
+
"failed",
|
|
130809
|
+
"setup_failed",
|
|
130810
|
+
"cancelled"
|
|
130811
|
+
];
|
|
130812
|
+
var swarmSessionTrialSchema2 = z14.object({
|
|
130813
|
+
status: z14.enum(SWARM_SESSION_TRIAL_STATUSES2),
|
|
130814
|
+
taskVerdict: z14.enum(["passed", "failed"]).optional(),
|
|
130815
|
+
evaluatorError: z14.literal(true).optional()
|
|
130816
|
+
}).strict().superRefine((value, ctx) => {
|
|
130817
|
+
const verdict = value.taskVerdict !== void 0;
|
|
130818
|
+
const error = value.evaluatorError === true;
|
|
130819
|
+
if (value.status === "completed" ? verdict === error : verdict || error) {
|
|
130820
|
+
ctx.addIssue({
|
|
130821
|
+
code: "custom",
|
|
130822
|
+
message: "Only completed trials carry exactly one verdict or evaluator error"
|
|
130823
|
+
});
|
|
130824
|
+
}
|
|
130825
|
+
});
|
|
130826
|
+
var swarmSessionGraderCountsSchema2 = z14.object({
|
|
130827
|
+
gating: z14.number().int().nonnegative(),
|
|
130828
|
+
gatingPassed: z14.number().int().nonnegative(),
|
|
130829
|
+
gatingFailed: z14.number().int().nonnegative(),
|
|
130830
|
+
gatingUnmeasured: z14.number().int().nonnegative(),
|
|
130831
|
+
advisoryFailed: z14.number().int().nonnegative()
|
|
130832
|
+
}).strict();
|
|
130833
|
+
var swarmSessionVerdictSchema2 = z14.object({
|
|
130834
|
+
contractVersion: z14.literal(SWARM_SESSION_VERDICT_CONTRACT_VERSION2),
|
|
130835
|
+
lifecycle: swarmSessionLifecycleSchema2,
|
|
130836
|
+
verdict: swarmSessionVerdictValueSchema2,
|
|
130837
|
+
reason: z14.enum(SWARM_SESSION_VERDICT_REASONS2),
|
|
130838
|
+
verdictSource: z14.enum([
|
|
130839
|
+
"goalJudge",
|
|
130840
|
+
"requiredAssertions",
|
|
130841
|
+
"combined",
|
|
130842
|
+
"none"
|
|
130843
|
+
]),
|
|
130844
|
+
grading: swarmGradingReadinessSchema2,
|
|
130845
|
+
graders: z14.object({
|
|
130846
|
+
criteria: z14.enum([
|
|
130847
|
+
"notConfigured",
|
|
130848
|
+
"notClaimed",
|
|
130849
|
+
"pending",
|
|
130850
|
+
"scored",
|
|
130851
|
+
"errored"
|
|
130852
|
+
]),
|
|
130853
|
+
judge: z14.enum([
|
|
130854
|
+
"notConfigured",
|
|
130855
|
+
"silent",
|
|
130856
|
+
"pending",
|
|
130857
|
+
"scored",
|
|
130858
|
+
"errored"
|
|
130859
|
+
])
|
|
130860
|
+
}).strict(),
|
|
130861
|
+
counts: swarmSessionGraderCountsSchema2,
|
|
130862
|
+
trial: swarmSessionTrialSchema2.nullable()
|
|
130863
|
+
}).strict().superRefine((value, ctx) => {
|
|
130864
|
+
const fail3 = (message) => ctx.addIssue({ code: "custom", message });
|
|
130865
|
+
if (SWARM_SESSION_VERDICT_OF_REASON2[value.reason] !== value.verdict)
|
|
130866
|
+
fail3("Reason contradicts verdict");
|
|
130867
|
+
const c = value.counts;
|
|
130868
|
+
if (c.gating !== c.gatingPassed + c.gatingFailed + c.gatingUnmeasured)
|
|
130869
|
+
fail3("Required measurement counts do not add up");
|
|
130870
|
+
const trial = value.trial;
|
|
130871
|
+
if (trial?.status === "completed") {
|
|
130872
|
+
if (value.lifecycle !== "ran")
|
|
130873
|
+
fail3("Only a completed execution can be a completed trial");
|
|
130874
|
+
if (trial.taskVerdict !== void 0 && trial.taskVerdict !== value.verdict)
|
|
130875
|
+
fail3("Trial contradicts goal result");
|
|
130876
|
+
if (trial.evaluatorError && value.verdict !== "inconclusive")
|
|
130877
|
+
fail3("Evaluator error requires inconclusive grading");
|
|
130878
|
+
}
|
|
130879
|
+
const expected = {
|
|
130880
|
+
pending: "pending",
|
|
130881
|
+
running: "running",
|
|
130882
|
+
broke: "failed",
|
|
130883
|
+
limited: "setup_failed",
|
|
130884
|
+
withdrawn: "cancelled"
|
|
130885
|
+
};
|
|
130886
|
+
if (value.lifecycle !== "ran" && trial?.status !== expected[value.lifecycle])
|
|
130887
|
+
fail3("Trial must preserve execution lifecycle");
|
|
130888
|
+
if (value.lifecycle === "ran" && trial !== null && trial.status !== "completed")
|
|
130889
|
+
fail3("Completed execution must not become skipped or running");
|
|
130890
|
+
if ((value.verdict === "passed" || value.verdict === "failed") && value.verdictSource === "none")
|
|
130891
|
+
fail3("A measured goal verdict needs a source");
|
|
130892
|
+
if (value.lifecycle === "ran" && value.verdict !== "notEstablished" && trial === null)
|
|
130893
|
+
fail3("Settled grading on completed execution needs an observation");
|
|
130894
|
+
if (value.verdict === "notEstablished" && value.verdictSource !== "none")
|
|
130895
|
+
fail3("An undecided goal has no verdict source");
|
|
130896
|
+
});
|
|
130897
|
+
var SWARM_FINDING_CONTRACT_VERSION2 = 1;
|
|
130898
|
+
var SWARM_FINDING_DISPOSITIONS2 = [
|
|
130899
|
+
"notRun",
|
|
130900
|
+
"blockedConnecting",
|
|
130901
|
+
"lostFindingTool",
|
|
130902
|
+
"blockedCallingTool",
|
|
130903
|
+
"blockedByResponse",
|
|
130904
|
+
"goalMissed",
|
|
130905
|
+
"goalMetWithFriction",
|
|
130906
|
+
"goalMet",
|
|
130907
|
+
"notMeasured"
|
|
130908
|
+
];
|
|
130909
|
+
var SWARM_FINDING_TONES2 = ["fail", "warn", "ok", "muted"];
|
|
130910
|
+
var SWARM_FINDING_SUMMARY_KINDS2 = [
|
|
130911
|
+
"notLaunched",
|
|
130912
|
+
"broken",
|
|
130913
|
+
"friction",
|
|
130914
|
+
"landed",
|
|
130915
|
+
"ungraded",
|
|
130916
|
+
"unread"
|
|
130917
|
+
];
|
|
130918
|
+
var SWARM_FINDING_COVERAGE_NOTES2 = [
|
|
130919
|
+
"sessionScanCapped",
|
|
130920
|
+
"budgetExhausted",
|
|
130921
|
+
"transcriptMissing",
|
|
130922
|
+
"contextTooLarge",
|
|
130923
|
+
"extractionRejected",
|
|
130924
|
+
"chainUnmeasured",
|
|
130925
|
+
"judgeNotRun",
|
|
130926
|
+
"sessionsWithdrawn",
|
|
130927
|
+
"sessionsRateLimited",
|
|
130928
|
+
"partialRead",
|
|
130929
|
+
"toolCatalogMissing"
|
|
130930
|
+
];
|
|
130931
|
+
var SWARM_FINDING_SCOPE_LEVELS2 = [
|
|
130932
|
+
"session",
|
|
130933
|
+
"goal",
|
|
130934
|
+
"persona",
|
|
130935
|
+
"target",
|
|
130936
|
+
"wave"
|
|
130937
|
+
];
|
|
130938
|
+
var SWARM_FINDING_BASES2 = [
|
|
130939
|
+
"verifiedMechanism",
|
|
130940
|
+
"sessionReport",
|
|
130941
|
+
"populationFact"
|
|
130942
|
+
];
|
|
130943
|
+
var SWARM_FINDING_CHAIN_STAGE_BASES2 = [
|
|
130944
|
+
"derived",
|
|
130945
|
+
"reported",
|
|
130946
|
+
"unmeasured"
|
|
130947
|
+
];
|
|
130948
|
+
var SWARM_FINDING_TONE_OF_DISPOSITION2 = Object.freeze({
|
|
130949
|
+
notRun: "muted",
|
|
130950
|
+
blockedConnecting: "fail",
|
|
130951
|
+
lostFindingTool: "fail",
|
|
130952
|
+
blockedCallingTool: "fail",
|
|
130953
|
+
blockedByResponse: "fail",
|
|
130954
|
+
goalMissed: "fail",
|
|
130955
|
+
goalMetWithFriction: "warn",
|
|
130956
|
+
goalMet: "ok",
|
|
130957
|
+
notMeasured: "muted"
|
|
130958
|
+
});
|
|
130959
|
+
var vocabulary2 = (members) => members.map((value) => "`" + value + "`").join(", ");
|
|
130960
|
+
var count3 = z14.number().int().nonnegative();
|
|
130961
|
+
var persona2 = z14.object({ personaRefId: z14.string().nullable(), name: z14.string() }).strict();
|
|
130962
|
+
var disposition2 = z14.enum(SWARM_FINDING_DISPOSITIONS2).describe(vocabulary2(SWARM_FINDING_DISPOSITIONS2));
|
|
130963
|
+
var coverageNotes2 = z14.array(
|
|
130964
|
+
z14.enum(SWARM_FINDING_COVERAGE_NOTES2).describe(vocabulary2(SWARM_FINDING_COVERAGE_NOTES2))
|
|
130965
|
+
);
|
|
130966
|
+
var citations2 = z14.array(z14.string().regex(/^[^/]+\/.+$/)).max(30);
|
|
130967
|
+
var toneMatchesDisposition2 = (row2) => row2.tone === SWARM_FINDING_TONE_OF_DISPOSITION2[row2.disposition];
|
|
130968
|
+
var toneMismatch2 = {
|
|
130969
|
+
message: "tone must equal SWARM_FINDING_TONE_OF_DISPOSITION[disposition]",
|
|
130970
|
+
path: ["tone"]
|
|
130971
|
+
};
|
|
130972
|
+
var swarmJourneyFindingSchema2 = z14.object({
|
|
130973
|
+
id: z14.string().min(1),
|
|
130974
|
+
basis: z14.enum(SWARM_FINDING_BASES2).describe(vocabulary2(SWARM_FINDING_BASES2)),
|
|
130975
|
+
scopeLevel: z14.enum(SWARM_FINDING_SCOPE_LEVELS2).describe(vocabulary2(SWARM_FINDING_SCOPE_LEVELS2)),
|
|
130976
|
+
persona: persona2,
|
|
130977
|
+
goal: z14.object({
|
|
130978
|
+
runId: z14.string(),
|
|
130979
|
+
journeyRefId: z14.string(),
|
|
130980
|
+
title: z14.string()
|
|
130981
|
+
}).strict(),
|
|
130982
|
+
target: z14.object({
|
|
130983
|
+
kind: z14.enum(["environment", "host"]),
|
|
130984
|
+
id: z14.string(),
|
|
130985
|
+
label: z14.string(),
|
|
130986
|
+
modelId: z14.string().nullable()
|
|
130987
|
+
}).strict(),
|
|
130988
|
+
population: z14.object({ count: count3, total: count3, unit: z14.literal("sessions") }).strict(),
|
|
130989
|
+
sessionIds: z14.array(z14.string()).max(1e3),
|
|
130990
|
+
citations: citations2,
|
|
130991
|
+
verdictSeen: swarmSessionVerdictValueSchema2,
|
|
130992
|
+
chainStage: userValueStageSchema2.nullable().describe(vocabulary2(USER_VALUE_STAGES2)),
|
|
130993
|
+
chainStageState: stageStateSchema2.nullable().describe(vocabulary2(STAGE_STATES2)),
|
|
130994
|
+
chainStageBasis: z14.enum(SWARM_FINDING_CHAIN_STAGE_BASES2),
|
|
130995
|
+
disposition: disposition2,
|
|
130996
|
+
tone: z14.enum(SWARM_FINDING_TONES2),
|
|
130997
|
+
coverageNotes: coverageNotes2,
|
|
130998
|
+
outcomePhrase: z14.string().max(100).regex(/^[^0-9]*$/).refine(
|
|
130999
|
+
(value) => value.trim().split(/\s+/).length <= 8,
|
|
131000
|
+
"At most eight words"
|
|
131001
|
+
).nullable(),
|
|
131002
|
+
mechanismPhrase: z14.string().nullable(),
|
|
131003
|
+
fixPhrase: z14.string().nullable(),
|
|
131004
|
+
reportExcerpt: z14.object({ actual: z14.string().max(1800), citations: citations2 }).strict().nullable(),
|
|
131005
|
+
mechanismId: z14.string().nullable()
|
|
131006
|
+
}).strict().refine(toneMatchesDisposition2, toneMismatch2);
|
|
131007
|
+
var swarmJourneyFindingsSchema = z14.object({
|
|
131008
|
+
contractVersion: z14.literal(SWARM_FINDING_CONTRACT_VERSION2),
|
|
131009
|
+
generatedAt: count3,
|
|
131010
|
+
sourceRevision: z14.string(),
|
|
131011
|
+
pipelineVersion: count3,
|
|
131012
|
+
extractionVersion: count3,
|
|
131013
|
+
extractionModel: z14.string(),
|
|
131014
|
+
reasoningModel: z14.string(),
|
|
131015
|
+
summaryKind: z14.enum(SWARM_FINDING_SUMMARY_KINDS2).describe(vocabulary2(SWARM_FINDING_SUMMARY_KINDS2)),
|
|
131016
|
+
population: z14.object({
|
|
131017
|
+
configured: count3,
|
|
131018
|
+
started: count3,
|
|
131019
|
+
read: count3,
|
|
131020
|
+
unread: count3,
|
|
131021
|
+
withdrawn: count3,
|
|
131022
|
+
limited: count3,
|
|
131023
|
+
graded: count3
|
|
131024
|
+
}).strict(),
|
|
131025
|
+
coverageNotes: coverageNotes2,
|
|
131026
|
+
disclosure: z14.object({
|
|
131027
|
+
rail: z14.enum(["gateway", "openrouter"]),
|
|
131028
|
+
evidenceSent: z14.array(z14.string())
|
|
131029
|
+
}).strict(),
|
|
131030
|
+
personas: z14.array(
|
|
131031
|
+
z14.object({
|
|
131032
|
+
persona: persona2,
|
|
131033
|
+
disposition: disposition2,
|
|
131034
|
+
tone: z14.enum(SWARM_FINDING_TONES2),
|
|
131035
|
+
goalRunIds: z14.array(z14.string())
|
|
131036
|
+
}).strict().refine(toneMatchesDisposition2, toneMismatch2)
|
|
131037
|
+
),
|
|
131038
|
+
findings: z14.array(swarmJourneyFindingSchema2).max(200)
|
|
131039
|
+
}).strict();
|
|
131040
|
+
var swarmJourneyFindingsJobSchema = z14.object({
|
|
131041
|
+
status: z14.enum(["pending", "completed", "failed", "skipped"]),
|
|
131042
|
+
errorCode: z14.string().optional(),
|
|
131043
|
+
updatedAt: count3
|
|
131044
|
+
}).strict();
|
|
130524
131045
|
var USER_VALUE_STAGE_LABELS2 = Object.freeze({
|
|
130525
131046
|
connection: "Connection",
|
|
130526
131047
|
discovery: "Discovery",
|
|
@@ -130719,7 +131240,51 @@ var SUSPECTED_CONDITION_CONFIDENCE_LABELS = Object.freeze({
|
|
|
130719
131240
|
medium: "medium confidence",
|
|
130720
131241
|
high: "high confidence"
|
|
130721
131242
|
});
|
|
131243
|
+
var SWARM_FINDING_DISPOSITION_LABELS = Object.freeze({
|
|
131244
|
+
notRun: "Not run",
|
|
131245
|
+
blockedConnecting: "Stuck",
|
|
131246
|
+
lostFindingTool: "Lost",
|
|
131247
|
+
blockedCallingTool: "Annoyed",
|
|
131248
|
+
blockedByResponse: "Frustrated",
|
|
131249
|
+
goalMissed: "Stalled",
|
|
131250
|
+
goalMetWithFriction: "Uneasy",
|
|
131251
|
+
goalMet: "Relieved",
|
|
131252
|
+
notMeasured: "Unscored"
|
|
131253
|
+
});
|
|
131254
|
+
var SWARM_FINDING_COVERAGE_NOTE_LABELS = Object.freeze({
|
|
131255
|
+
sessionScanCapped: "Session scan limit reached",
|
|
131256
|
+
budgetExhausted: "Analysis budget exhausted",
|
|
131257
|
+
transcriptMissing: "Transcript unavailable",
|
|
131258
|
+
contextTooLarge: "Transcript exceeds analysis limits",
|
|
131259
|
+
extractionRejected: "Analysis could not be verified",
|
|
131260
|
+
chainUnmeasured: "Journey stages not measured",
|
|
131261
|
+
judgeNotRun: "Judge did not run",
|
|
131262
|
+
sessionsWithdrawn: "Some sessions were withdrawn",
|
|
131263
|
+
sessionsRateLimited: "Some sessions were rate limited",
|
|
131264
|
+
partialRead: "Only part of this wave was read",
|
|
131265
|
+
toolCatalogMissing: "Tool catalog unavailable"
|
|
131266
|
+
});
|
|
131267
|
+
var SWARM_FINDING_SUMMARY_KIND_LABELS = Object.freeze({
|
|
131268
|
+
notLaunched: "Not launched",
|
|
131269
|
+
broken: "Goals blocked",
|
|
131270
|
+
friction: "Goals met with friction",
|
|
131271
|
+
landed: "Goals met",
|
|
131272
|
+
ungraded: "Not graded",
|
|
131273
|
+
unread: "Not fully read"
|
|
131274
|
+
});
|
|
131275
|
+
var SWARM_FINDING_BASIS_LABELS = Object.freeze({
|
|
131276
|
+
verifiedMechanism: "Verified explanation",
|
|
131277
|
+
sessionReport: "Session report",
|
|
131278
|
+
populationFact: "Population fact"
|
|
131279
|
+
});
|
|
130722
131280
|
var DECISION_LABEL_VOCABULARIES = Object.freeze({
|
|
131281
|
+
swarmFindingDispositions: SWARM_FINDING_DISPOSITIONS2,
|
|
131282
|
+
swarmFindingCoverageNotes: SWARM_FINDING_COVERAGE_NOTES2,
|
|
131283
|
+
swarmFindingSummaryKinds: SWARM_FINDING_SUMMARY_KINDS2,
|
|
131284
|
+
swarmFindingBases: SWARM_FINDING_BASES2,
|
|
131285
|
+
swarmFindingScopeLevels: SWARM_FINDING_SCOPE_LEVELS2,
|
|
131286
|
+
// Tones are presentation vocabulary shared with unrelated UI schemas.
|
|
131287
|
+
// Their totality is tested directly against the disposition-to-tone map.
|
|
130723
131288
|
stages: USER_VALUE_STAGES2,
|
|
130724
131289
|
stageStates: STAGE_STATES2,
|
|
130725
131290
|
failureCategories: FAILURE_CATEGORIES2,
|
|
@@ -131140,10 +131705,10 @@ function assembleEvalRunDecisionSummary(input) {
|
|
|
131140
131705
|
}
|
|
131141
131706
|
function legacyTrialCounts(summary) {
|
|
131142
131707
|
if (!summary || typeof summary !== "object") return void 0;
|
|
131143
|
-
const
|
|
131144
|
-
const total =
|
|
131145
|
-
const passed2 =
|
|
131146
|
-
const failed3 =
|
|
131708
|
+
const count32 = (value) => typeof value === "number" && Number.isInteger(value) && value >= 0 ? value : void 0;
|
|
131709
|
+
const total = count32(summary.total);
|
|
131710
|
+
const passed2 = count32(summary.passed);
|
|
131711
|
+
const failed3 = count32(summary.failed);
|
|
131147
131712
|
if (total === void 0 && passed2 === void 0 && failed3 === void 0) {
|
|
131148
131713
|
return void 0;
|
|
131149
131714
|
}
|
|
@@ -132557,249 +133122,28 @@ var evalBacktestContinuationSchema2 = z14.object({
|
|
|
132557
133122
|
var evalBacktestRequestSchema = evalBacktestDraftSchema2.extend({
|
|
132558
133123
|
continuation: evalBacktestContinuationSchema2.optional()
|
|
132559
133124
|
});
|
|
132560
|
-
var SWARM_SESSION_VERDICT_CONTRACT_VERSION2 = 1;
|
|
132561
|
-
var SWARM_STAGE_EVIDENCE_VERSION = 2;
|
|
132562
|
-
var SWARM_ATTEMPT_STATUSES2 = [
|
|
132563
|
-
"pending",
|
|
132564
|
-
"running",
|
|
132565
|
-
"succeeded",
|
|
132566
|
-
"failed",
|
|
132567
|
-
"rate_limited"
|
|
132568
|
-
];
|
|
132569
|
-
var SWARM_SESSION_LIFECYCLES2 = [
|
|
132570
|
-
"pending",
|
|
132571
|
-
"running",
|
|
132572
|
-
"ran",
|
|
132573
|
-
"broke",
|
|
132574
|
-
"limited",
|
|
132575
|
-
"withdrawn"
|
|
132576
|
-
];
|
|
132577
|
-
var SWARM_SESSION_VERDICTS2 = [
|
|
132578
|
-
"passed",
|
|
132579
|
-
"failed",
|
|
132580
|
-
"inconclusive",
|
|
132581
|
-
"notEstablished"
|
|
132582
|
-
];
|
|
132583
|
-
var SWARM_GRADING_STATES2 = [
|
|
132584
|
-
"notRequested",
|
|
132585
|
-
"queued",
|
|
132586
|
-
"running",
|
|
132587
|
-
"settled",
|
|
132588
|
-
"unavailable"
|
|
132589
|
-
];
|
|
132590
|
-
var SWARM_SESSION_VERDICT_OF_REASON2 = {
|
|
132591
|
-
attemptPending: "notEstablished",
|
|
132592
|
-
attemptRunning: "notEstablished",
|
|
132593
|
-
withdrawn: "notEstablished",
|
|
132594
|
-
spendCapReached: "notEstablished",
|
|
132595
|
-
notRun: "notEstablished",
|
|
132596
|
-
executionFailed: "notEstablished",
|
|
132597
|
-
ungraded: "notEstablished",
|
|
132598
|
-
gradingNotClaimed: "notEstablished",
|
|
132599
|
-
criteriaPending: "notEstablished",
|
|
132600
|
-
judgePending: "notEstablished",
|
|
132601
|
-
gatingCriterionFailed: "failed",
|
|
132602
|
-
judgeFailed: "failed",
|
|
132603
|
-
criteriaGradingErrored: "inconclusive",
|
|
132604
|
-
gatingCriterionUnmeasured: "inconclusive",
|
|
132605
|
-
judgeErrored: "inconclusive",
|
|
132606
|
-
gradingUnavailable: "inconclusive",
|
|
132607
|
-
allGatingCriteriaPassed: "passed",
|
|
132608
|
-
judgePassed: "passed",
|
|
132609
|
-
allGradersPassed: "passed"
|
|
132610
|
-
};
|
|
132611
|
-
var SWARM_SESSION_VERDICT_REASONS2 = Object.keys(
|
|
132612
|
-
SWARM_SESSION_VERDICT_OF_REASON2
|
|
132613
|
-
);
|
|
132614
|
-
var swarmSessionLifecycleSchema2 = z14.enum(SWARM_SESSION_LIFECYCLES2);
|
|
132615
|
-
var swarmSessionVerdictValueSchema2 = z14.enum(SWARM_SESSION_VERDICTS2);
|
|
132616
|
-
var swarmGraderRoleSchema2 = z14.enum(["advisory", "required"]);
|
|
132617
|
-
var swarmSessionAttemptInputSchema2 = z14.object({
|
|
132618
|
-
status: z14.enum(SWARM_ATTEMPT_STATUSES2),
|
|
132619
|
-
errorCode: z14.string().nullable().optional()
|
|
132620
|
-
}).strict();
|
|
132621
|
-
var swarmGradingReadinessSchema2 = z14.object({
|
|
132622
|
-
/** Durable readiness for decisive grading only, not background observations. */
|
|
132623
|
-
state: z14.enum(SWARM_GRADING_STATES2),
|
|
132624
|
-
reasonCode: z14.string().min(1).optional()
|
|
132625
|
-
}).strict();
|
|
132626
|
-
var criterionResultSchema2 = z14.object({
|
|
132627
|
-
criterionId: z14.string().min(1),
|
|
132628
|
-
passed: z14.boolean(),
|
|
132629
|
-
status: z14.enum(["scored", "error"]).optional()
|
|
132630
|
-
}).strict();
|
|
132631
|
-
var swarmSessionVerdictInputSchema = z14.object({
|
|
132632
|
-
attempt: swarmSessionAttemptInputSchema2.nullable(),
|
|
132633
|
-
hasTranscript: z14.boolean(),
|
|
132634
|
-
rubric: z14.array(
|
|
132635
|
-
z14.object({
|
|
132636
|
-
id: z14.string().min(1),
|
|
132637
|
-
role: swarmGraderRoleSchema2,
|
|
132638
|
-
predicateType: z14.string().min(1).optional()
|
|
132639
|
-
}).strict()
|
|
132640
|
-
),
|
|
132641
|
-
criteria: z14.object({
|
|
132642
|
-
status: z14.enum(["pending", "completed", "failed"]),
|
|
132643
|
-
criterionIds: z14.array(z14.string().min(1)).optional(),
|
|
132644
|
-
results: z14.array(criterionResultSchema2).optional()
|
|
132645
|
-
}).strict().nullable(),
|
|
132646
|
-
goalScore: z14.discriminatedUnion("status", [
|
|
132647
|
-
z14.object({ status: z14.literal("completed"), passed: z14.boolean() }).strict(),
|
|
132648
|
-
z14.object({ status: z14.literal("running") }).strict(),
|
|
132649
|
-
z14.object({ status: z14.literal("failed") }).strict()
|
|
132650
|
-
]).nullable(),
|
|
132651
|
-
judge: z14.object({
|
|
132652
|
-
automatic: z14.boolean(),
|
|
132653
|
-
role: swarmGraderRoleSchema2,
|
|
132654
|
-
/** Explicit on-demand intent; automatic:false alone does not mean a judge was requested. */
|
|
132655
|
-
requested: z14.boolean().optional()
|
|
132656
|
-
}).strict(),
|
|
132657
|
-
grading: swarmGradingReadinessSchema2
|
|
132658
|
-
}).strict().superRefine((value, ctx) => {
|
|
132659
|
-
const definitions = new Set(value.rubric.map((entry3) => entry3.id));
|
|
132660
|
-
if (definitions.size !== value.rubric.length) {
|
|
132661
|
-
ctx.addIssue({
|
|
132662
|
-
code: "custom",
|
|
132663
|
-
path: ["rubric"],
|
|
132664
|
-
message: "Duplicate criterion definition"
|
|
132665
|
-
});
|
|
132666
|
-
}
|
|
132667
|
-
for (const [field, ids] of [
|
|
132668
|
-
["criterionIds", value.criteria?.criterionIds ?? []],
|
|
132669
|
-
["results", value.criteria?.results?.map((row2) => row2.criterionId) ?? []]
|
|
132670
|
-
]) {
|
|
132671
|
-
if (new Set(ids).size !== ids.length || ids.some((id3) => !definitions.has(id3))) {
|
|
132672
|
-
ctx.addIssue({
|
|
132673
|
-
code: "custom",
|
|
132674
|
-
path: ["criteria", field],
|
|
132675
|
-
message: "Criterion IDs must be unique and defined in the snapshot"
|
|
132676
|
-
});
|
|
132677
|
-
}
|
|
132678
|
-
}
|
|
132679
|
-
const claimed = value.criteria?.criterionIds;
|
|
132680
|
-
if (claimed && value.criteria?.results?.some((row2) => !claimed.includes(row2.criterionId))) {
|
|
132681
|
-
ctx.addIssue({
|
|
132682
|
-
code: "custom",
|
|
132683
|
-
path: ["criteria", "results"],
|
|
132684
|
-
message: "Result outside claimed scope"
|
|
132685
|
-
});
|
|
132686
|
-
}
|
|
132687
|
-
});
|
|
132688
|
-
var SWARM_SESSION_TRIAL_STATUSES2 = [
|
|
132689
|
-
"pending",
|
|
132690
|
-
"running",
|
|
132691
|
-
"completed",
|
|
132692
|
-
"failed",
|
|
132693
|
-
"setup_failed",
|
|
132694
|
-
"cancelled"
|
|
132695
|
-
];
|
|
132696
|
-
var swarmSessionTrialSchema2 = z14.object({
|
|
132697
|
-
status: z14.enum(SWARM_SESSION_TRIAL_STATUSES2),
|
|
132698
|
-
taskVerdict: z14.enum(["passed", "failed"]).optional(),
|
|
132699
|
-
evaluatorError: z14.literal(true).optional()
|
|
132700
|
-
}).strict().superRefine((value, ctx) => {
|
|
132701
|
-
const verdict = value.taskVerdict !== void 0;
|
|
132702
|
-
const error = value.evaluatorError === true;
|
|
132703
|
-
if (value.status === "completed" ? verdict === error : verdict || error) {
|
|
132704
|
-
ctx.addIssue({
|
|
132705
|
-
code: "custom",
|
|
132706
|
-
message: "Only completed trials carry exactly one verdict or evaluator error"
|
|
132707
|
-
});
|
|
132708
|
-
}
|
|
132709
|
-
});
|
|
132710
|
-
var swarmSessionGraderCountsSchema2 = z14.object({
|
|
132711
|
-
gating: z14.number().int().nonnegative(),
|
|
132712
|
-
gatingPassed: z14.number().int().nonnegative(),
|
|
132713
|
-
gatingFailed: z14.number().int().nonnegative(),
|
|
132714
|
-
gatingUnmeasured: z14.number().int().nonnegative(),
|
|
132715
|
-
advisoryFailed: z14.number().int().nonnegative()
|
|
132716
|
-
}).strict();
|
|
132717
|
-
var swarmSessionVerdictSchema2 = z14.object({
|
|
132718
|
-
contractVersion: z14.literal(SWARM_SESSION_VERDICT_CONTRACT_VERSION2),
|
|
132719
|
-
lifecycle: swarmSessionLifecycleSchema2,
|
|
132720
|
-
verdict: swarmSessionVerdictValueSchema2,
|
|
132721
|
-
reason: z14.enum(SWARM_SESSION_VERDICT_REASONS2),
|
|
132722
|
-
verdictSource: z14.enum([
|
|
132723
|
-
"goalJudge",
|
|
132724
|
-
"requiredAssertions",
|
|
132725
|
-
"combined",
|
|
132726
|
-
"none"
|
|
132727
|
-
]),
|
|
132728
|
-
grading: swarmGradingReadinessSchema2,
|
|
132729
|
-
graders: z14.object({
|
|
132730
|
-
criteria: z14.enum([
|
|
132731
|
-
"notConfigured",
|
|
132732
|
-
"notClaimed",
|
|
132733
|
-
"pending",
|
|
132734
|
-
"scored",
|
|
132735
|
-
"errored"
|
|
132736
|
-
]),
|
|
132737
|
-
judge: z14.enum([
|
|
132738
|
-
"notConfigured",
|
|
132739
|
-
"silent",
|
|
132740
|
-
"pending",
|
|
132741
|
-
"scored",
|
|
132742
|
-
"errored"
|
|
132743
|
-
])
|
|
132744
|
-
}).strict(),
|
|
132745
|
-
counts: swarmSessionGraderCountsSchema2,
|
|
132746
|
-
trial: swarmSessionTrialSchema2.nullable()
|
|
132747
|
-
}).strict().superRefine((value, ctx) => {
|
|
132748
|
-
const fail3 = (message) => ctx.addIssue({ code: "custom", message });
|
|
132749
|
-
if (SWARM_SESSION_VERDICT_OF_REASON2[value.reason] !== value.verdict)
|
|
132750
|
-
fail3("Reason contradicts verdict");
|
|
132751
|
-
const c = value.counts;
|
|
132752
|
-
if (c.gating !== c.gatingPassed + c.gatingFailed + c.gatingUnmeasured)
|
|
132753
|
-
fail3("Required measurement counts do not add up");
|
|
132754
|
-
const trial = value.trial;
|
|
132755
|
-
if (trial?.status === "completed") {
|
|
132756
|
-
if (value.lifecycle !== "ran")
|
|
132757
|
-
fail3("Only a completed execution can be a completed trial");
|
|
132758
|
-
if (trial.taskVerdict !== void 0 && trial.taskVerdict !== value.verdict)
|
|
132759
|
-
fail3("Trial contradicts goal result");
|
|
132760
|
-
if (trial.evaluatorError && value.verdict !== "inconclusive")
|
|
132761
|
-
fail3("Evaluator error requires inconclusive grading");
|
|
132762
|
-
}
|
|
132763
|
-
const expected = {
|
|
132764
|
-
pending: "pending",
|
|
132765
|
-
running: "running",
|
|
132766
|
-
broke: "failed",
|
|
132767
|
-
limited: "setup_failed",
|
|
132768
|
-
withdrawn: "cancelled"
|
|
132769
|
-
};
|
|
132770
|
-
if (value.lifecycle !== "ran" && trial?.status !== expected[value.lifecycle])
|
|
132771
|
-
fail3("Trial must preserve execution lifecycle");
|
|
132772
|
-
if (value.lifecycle === "ran" && trial !== null && trial.status !== "completed")
|
|
132773
|
-
fail3("Completed execution must not become skipped or running");
|
|
132774
|
-
if ((value.verdict === "passed" || value.verdict === "failed") && value.verdictSource === "none")
|
|
132775
|
-
fail3("A measured goal verdict needs a source");
|
|
132776
|
-
if (value.lifecycle === "ran" && value.verdict !== "notEstablished" && trial === null)
|
|
132777
|
-
fail3("Settled grading on completed execution needs an observation");
|
|
132778
|
-
if (value.verdict === "notEstablished" && value.verdictSource !== "none")
|
|
132779
|
-
fail3("An undecided goal has no verdict source");
|
|
132780
|
-
});
|
|
132781
133125
|
var SWARM_REPORT_CONTRACT_VERSION2 = 1;
|
|
132782
|
-
var
|
|
133126
|
+
var count22 = z14.number().int().nonnegative();
|
|
132783
133127
|
var journeyRunVerdictSummarySchema2 = z14.discriminatedUnion("status", [
|
|
132784
133128
|
z14.object({
|
|
132785
133129
|
status: z14.literal("pending"),
|
|
132786
|
-
pendingSessions:
|
|
132787
|
-
updatedAt:
|
|
133130
|
+
pendingSessions: count22,
|
|
133131
|
+
updatedAt: count22
|
|
132788
133132
|
}).strict(),
|
|
132789
133133
|
z14.object({
|
|
132790
133134
|
status: z14.literal("decided"),
|
|
132791
133135
|
decision: evalVerdictDecisionSchema2,
|
|
132792
|
-
updatedAt:
|
|
133136
|
+
updatedAt: count22
|
|
132793
133137
|
}).strict(),
|
|
132794
133138
|
z14.object({
|
|
132795
133139
|
status: z14.literal("notEstablished"),
|
|
132796
133140
|
reason: z14.literal("gradingNotConfigured"),
|
|
132797
|
-
updatedAt:
|
|
133141
|
+
updatedAt: count22
|
|
132798
133142
|
}).strict(),
|
|
132799
133143
|
z14.object({
|
|
132800
133144
|
status: z14.literal("integrityFailed"),
|
|
132801
133145
|
reason: z14.string().min(1),
|
|
132802
|
-
updatedAt:
|
|
133146
|
+
updatedAt: count22
|
|
132803
133147
|
}).strict()
|
|
132804
133148
|
]);
|
|
132805
133149
|
var swarmObservationInputSchema2 = z14.object({
|
|
@@ -132837,7 +133181,7 @@ var swarmReportSessionSchema2 = z14.object({
|
|
|
132837
133181
|
var swarmReportInputSchema = z14.object({
|
|
132838
133182
|
runId: z14.string().min(1),
|
|
132839
133183
|
executionComplete: z14.boolean(),
|
|
132840
|
-
configuredSessions:
|
|
133184
|
+
configuredSessions: count22,
|
|
132841
133185
|
verdictSummary: journeyRunVerdictSummarySchema2.nullable(),
|
|
132842
133186
|
/** Frozen run-wide definitions: absent measurement rows still count as unavailable. */
|
|
132843
133187
|
evaluatorDefinitions: z14.array(
|
|
@@ -132892,11 +133236,11 @@ var swarmObservationCoverageSchema2 = z14.object({
|
|
|
132892
133236
|
"The stage this observation was read from, and the only stage it is evidence ABOUT. `connection` \u2014 the server was reachable and the session initialized. `discovery` \u2014 its tools and resources were listed and readable. `selection` \u2014 the model chose the right tool for the request. `call` \u2014 the call was made with usable arguments. `response` \u2014 the server returned data the model could use. `userValue` \u2014 the user's actual request was satisfied."
|
|
132893
133237
|
),
|
|
132894
133238
|
unit: z14.literal("sessions"),
|
|
132895
|
-
total:
|
|
132896
|
-
passed:
|
|
132897
|
-
failed:
|
|
132898
|
-
pending:
|
|
132899
|
-
unavailable:
|
|
133239
|
+
total: count22,
|
|
133240
|
+
passed: count22,
|
|
133241
|
+
failed: count22,
|
|
133242
|
+
pending: count22,
|
|
133243
|
+
unavailable: count22
|
|
132900
133244
|
}).strict().superRefine((value, ctx) => {
|
|
132901
133245
|
if (value.total !== value.passed + value.failed + value.pending + value.unavailable) {
|
|
132902
133246
|
ctx.addIssue({
|
|
@@ -132926,26 +133270,26 @@ var swarmReportSchema = z14.object({
|
|
|
132926
133270
|
]).optional(),
|
|
132927
133271
|
execution: z14.object({
|
|
132928
133272
|
unit: z14.literal("sessions"),
|
|
132929
|
-
configured:
|
|
132930
|
-
reported:
|
|
132931
|
-
started:
|
|
132932
|
-
notStarted:
|
|
132933
|
-
unknown:
|
|
132934
|
-
completed:
|
|
132935
|
-
interrupted:
|
|
133273
|
+
configured: count22,
|
|
133274
|
+
reported: count22,
|
|
133275
|
+
started: count22,
|
|
133276
|
+
notStarted: count22,
|
|
133277
|
+
unknown: count22,
|
|
133278
|
+
completed: count22,
|
|
133279
|
+
interrupted: count22,
|
|
132936
133280
|
/** Only explicit complete execution evidence can establish this. */
|
|
132937
133281
|
neverLaunched: z14.boolean()
|
|
132938
133282
|
}).strict(),
|
|
132939
133283
|
goalGrading: z14.object({
|
|
132940
133284
|
unit: z14.literal("sessions"),
|
|
132941
|
-
reported:
|
|
132942
|
-
passed:
|
|
132943
|
-
failed:
|
|
132944
|
-
pending:
|
|
132945
|
-
unavailable:
|
|
132946
|
-
notRequested:
|
|
133285
|
+
reported: count22,
|
|
133286
|
+
passed: count22,
|
|
133287
|
+
failed: count22,
|
|
133288
|
+
pending: count22,
|
|
133289
|
+
unavailable: count22,
|
|
133290
|
+
notRequested: count22,
|
|
132947
133291
|
/** Readiness is separate: a proven failure can exist while other required grading is pending. */
|
|
132948
|
-
waitingForDecisiveGrading:
|
|
133292
|
+
waitingForDecisiveGrading: count22
|
|
132949
133293
|
}).strict(),
|
|
132950
133294
|
observations: z14.array(swarmObservationCoverageSchema2)
|
|
132951
133295
|
}).strict().superRefine((report, ctx) => {
|
|
@@ -147013,22 +147357,23 @@ function attachedFailureCode(error) {
|
|
|
147013
147357
|
return typeof code === "string" ? code : void 0;
|
|
147014
147358
|
}
|
|
147015
147359
|
function parseEngineErrorBody(status, bodyText) {
|
|
147016
|
-
let code;
|
|
147017
147360
|
try {
|
|
147018
147361
|
const body = JSON.parse(bodyText);
|
|
147019
|
-
if (body
|
|
147362
|
+
if (body && typeof body === "object") {
|
|
147020
147363
|
return {
|
|
147021
|
-
message: body.details ? `${body.error} ${body.details}` : body.error
|
|
147364
|
+
message: body.error ? body.details ? `${body.error} ${body.details}` : body.error : `Backend stream error: ${status} ${bodyText}`,
|
|
147022
147365
|
...body.code ? { code: body.code } : {},
|
|
147023
|
-
...body.details ? { details: body.details } : {}
|
|
147366
|
+
...body.details ? { details: body.details } : {},
|
|
147367
|
+
...typeof body.retryAfter === "number" && Number.isFinite(body.retryAfter) ? { retryAfterMs: body.retryAfter } : {},
|
|
147368
|
+
...typeof body.isRetryable === "boolean" ? { isRetryable: body.isRetryable } : {},
|
|
147369
|
+
...typeof body.refusalReason === "string" ? { refusalReason: body.refusalReason } : {},
|
|
147370
|
+
...typeof body.outstandingHolds === "number" ? { outstandingHolds: body.outstandingHolds } : {}
|
|
147024
147371
|
};
|
|
147025
147372
|
}
|
|
147026
|
-
code = typeof body?.code === "string" ? body.code : void 0;
|
|
147027
147373
|
} catch {
|
|
147028
147374
|
}
|
|
147029
147375
|
return {
|
|
147030
|
-
message: status !== void 0 ? `Backend stream error: ${status} ${bodyText}` : bodyText
|
|
147031
|
-
...code ? { code } : {}
|
|
147376
|
+
message: status !== void 0 ? `Backend stream error: ${status} ${bodyText}` : bodyText
|
|
147032
147377
|
};
|
|
147033
147378
|
}
|
|
147034
147379
|
function safelyEmitEngineError(onEngineError, event) {
|
|
@@ -147802,7 +148147,7 @@ async function processOneStep(ctx) {
|
|
|
147802
148147
|
});
|
|
147803
148148
|
}
|
|
147804
148149
|
safelyEmitEngineError(onEngineError, {
|
|
147805
|
-
|
|
148150
|
+
...parsed,
|
|
147806
148151
|
...parsed.code ? { code: parsed.code } : {},
|
|
147807
148152
|
...parsed.details ? { details: parsed.details } : {},
|
|
147808
148153
|
httpStatus: res.status,
|
|
@@ -153446,7 +153791,7 @@ function platformRefusalHint(refusal) {
|
|
|
153446
153791
|
return refusal.canTopUp === false ? `${when} This is a usage limit: topping up credits does not lift it.` : when;
|
|
153447
153792
|
}
|
|
153448
153793
|
function readSdkVersion3() {
|
|
153449
|
-
return "8.
|
|
153794
|
+
return "8.11.0";
|
|
153450
153795
|
}
|
|
153451
153796
|
var DEFAULT_PLATFORM_API_BASE_URL = "https://app.mcpjam.com/api/v1";
|
|
153452
153797
|
function resolvePlatformRequestUrl(spec) {
|
|
@@ -153612,11 +153957,11 @@ var PlatformApiClient = class _PlatformApiClient {
|
|
|
153612
153957
|
* header more. The original client is untouched; the CLI keeps it for the
|
|
153613
153958
|
* operations that still speak vocabulary 1.
|
|
153614
153959
|
*/
|
|
153615
|
-
withEvalVocabulary(
|
|
153616
|
-
if (
|
|
153960
|
+
withEvalVocabulary(vocabulary22) {
|
|
153961
|
+
if (vocabulary22 === this.evalVocabulary) return this;
|
|
153617
153962
|
return new _PlatformApiClient({
|
|
153618
153963
|
...this.constructorOptions,
|
|
153619
|
-
evalVocabulary:
|
|
153964
|
+
evalVocabulary: vocabulary22
|
|
153620
153965
|
});
|
|
153621
153966
|
}
|
|
153622
153967
|
/** Coding-agent browser entry point; command outcomes are returned in-band. */
|
|
@@ -159992,6 +160337,374 @@ z24.discriminatedUnion("status", [
|
|
|
159992
160337
|
...suspectedConditionProvenance3
|
|
159993
160338
|
}).strict()
|
|
159994
160339
|
]);
|
|
160340
|
+
var SWARM_SESSION_VERDICT_CONTRACT_VERSION3 = 1;
|
|
160341
|
+
var SWARM_ATTEMPT_STATUSES3 = [
|
|
160342
|
+
"pending",
|
|
160343
|
+
"running",
|
|
160344
|
+
"succeeded",
|
|
160345
|
+
"failed",
|
|
160346
|
+
"rate_limited"
|
|
160347
|
+
];
|
|
160348
|
+
var SWARM_SESSION_LIFECYCLES3 = [
|
|
160349
|
+
"pending",
|
|
160350
|
+
"running",
|
|
160351
|
+
"ran",
|
|
160352
|
+
"broke",
|
|
160353
|
+
"limited",
|
|
160354
|
+
"withdrawn"
|
|
160355
|
+
];
|
|
160356
|
+
var SWARM_SESSION_VERDICTS3 = [
|
|
160357
|
+
"passed",
|
|
160358
|
+
"failed",
|
|
160359
|
+
"inconclusive",
|
|
160360
|
+
"notEstablished"
|
|
160361
|
+
];
|
|
160362
|
+
var SWARM_GRADING_STATES3 = [
|
|
160363
|
+
"notRequested",
|
|
160364
|
+
"queued",
|
|
160365
|
+
"running",
|
|
160366
|
+
"settled",
|
|
160367
|
+
"unavailable"
|
|
160368
|
+
];
|
|
160369
|
+
var SWARM_SESSION_VERDICT_OF_REASON3 = {
|
|
160370
|
+
attemptPending: "notEstablished",
|
|
160371
|
+
attemptRunning: "notEstablished",
|
|
160372
|
+
withdrawn: "notEstablished",
|
|
160373
|
+
spendCapReached: "notEstablished",
|
|
160374
|
+
notRun: "notEstablished",
|
|
160375
|
+
executionFailed: "notEstablished",
|
|
160376
|
+
ungraded: "notEstablished",
|
|
160377
|
+
gradingNotClaimed: "notEstablished",
|
|
160378
|
+
criteriaPending: "notEstablished",
|
|
160379
|
+
judgePending: "notEstablished",
|
|
160380
|
+
gatingCriterionFailed: "failed",
|
|
160381
|
+
judgeFailed: "failed",
|
|
160382
|
+
criteriaGradingErrored: "inconclusive",
|
|
160383
|
+
gatingCriterionUnmeasured: "inconclusive",
|
|
160384
|
+
judgeErrored: "inconclusive",
|
|
160385
|
+
gradingUnavailable: "inconclusive",
|
|
160386
|
+
allGatingCriteriaPassed: "passed",
|
|
160387
|
+
judgePassed: "passed",
|
|
160388
|
+
allGradersPassed: "passed"
|
|
160389
|
+
};
|
|
160390
|
+
var SWARM_SESSION_VERDICT_REASONS3 = Object.keys(
|
|
160391
|
+
SWARM_SESSION_VERDICT_OF_REASON3
|
|
160392
|
+
);
|
|
160393
|
+
var swarmSessionLifecycleSchema3 = z24.enum(SWARM_SESSION_LIFECYCLES3);
|
|
160394
|
+
var swarmSessionVerdictValueSchema3 = z24.enum(SWARM_SESSION_VERDICTS3);
|
|
160395
|
+
var swarmGraderRoleSchema3 = z24.enum(["advisory", "required"]);
|
|
160396
|
+
var swarmSessionAttemptInputSchema3 = z24.object({
|
|
160397
|
+
status: z24.enum(SWARM_ATTEMPT_STATUSES3),
|
|
160398
|
+
errorCode: z24.string().nullable().optional()
|
|
160399
|
+
}).strict();
|
|
160400
|
+
var swarmGradingReadinessSchema3 = z24.object({
|
|
160401
|
+
/** Durable readiness for decisive grading only, not background observations. */
|
|
160402
|
+
state: z24.enum(SWARM_GRADING_STATES3),
|
|
160403
|
+
reasonCode: z24.string().min(1).optional()
|
|
160404
|
+
}).strict();
|
|
160405
|
+
var criterionResultSchema3 = z24.object({
|
|
160406
|
+
criterionId: z24.string().min(1),
|
|
160407
|
+
passed: z24.boolean(),
|
|
160408
|
+
status: z24.enum(["scored", "error"]).optional()
|
|
160409
|
+
}).strict();
|
|
160410
|
+
z24.object({
|
|
160411
|
+
attempt: swarmSessionAttemptInputSchema3.nullable(),
|
|
160412
|
+
hasTranscript: z24.boolean(),
|
|
160413
|
+
rubric: z24.array(
|
|
160414
|
+
z24.object({
|
|
160415
|
+
id: z24.string().min(1),
|
|
160416
|
+
role: swarmGraderRoleSchema3,
|
|
160417
|
+
predicateType: z24.string().min(1).optional()
|
|
160418
|
+
}).strict()
|
|
160419
|
+
),
|
|
160420
|
+
criteria: z24.object({
|
|
160421
|
+
status: z24.enum(["pending", "completed", "failed"]),
|
|
160422
|
+
criterionIds: z24.array(z24.string().min(1)).optional(),
|
|
160423
|
+
results: z24.array(criterionResultSchema3).optional()
|
|
160424
|
+
}).strict().nullable(),
|
|
160425
|
+
goalScore: z24.discriminatedUnion("status", [
|
|
160426
|
+
z24.object({ status: z24.literal("completed"), passed: z24.boolean() }).strict(),
|
|
160427
|
+
z24.object({ status: z24.literal("running") }).strict(),
|
|
160428
|
+
z24.object({ status: z24.literal("failed") }).strict()
|
|
160429
|
+
]).nullable(),
|
|
160430
|
+
judge: z24.object({
|
|
160431
|
+
automatic: z24.boolean(),
|
|
160432
|
+
role: swarmGraderRoleSchema3,
|
|
160433
|
+
/** Explicit on-demand intent; automatic:false alone does not mean a judge was requested. */
|
|
160434
|
+
requested: z24.boolean().optional()
|
|
160435
|
+
}).strict(),
|
|
160436
|
+
grading: swarmGradingReadinessSchema3
|
|
160437
|
+
}).strict().superRefine((value, ctx) => {
|
|
160438
|
+
const definitions = new Set(value.rubric.map((entry3) => entry3.id));
|
|
160439
|
+
if (definitions.size !== value.rubric.length) {
|
|
160440
|
+
ctx.addIssue({
|
|
160441
|
+
code: "custom",
|
|
160442
|
+
path: ["rubric"],
|
|
160443
|
+
message: "Duplicate criterion definition"
|
|
160444
|
+
});
|
|
160445
|
+
}
|
|
160446
|
+
for (const [field, ids] of [
|
|
160447
|
+
["criterionIds", value.criteria?.criterionIds ?? []],
|
|
160448
|
+
["results", value.criteria?.results?.map((row2) => row2.criterionId) ?? []]
|
|
160449
|
+
]) {
|
|
160450
|
+
if (new Set(ids).size !== ids.length || ids.some((id3) => !definitions.has(id3))) {
|
|
160451
|
+
ctx.addIssue({
|
|
160452
|
+
code: "custom",
|
|
160453
|
+
path: ["criteria", field],
|
|
160454
|
+
message: "Criterion IDs must be unique and defined in the snapshot"
|
|
160455
|
+
});
|
|
160456
|
+
}
|
|
160457
|
+
}
|
|
160458
|
+
const claimed = value.criteria?.criterionIds;
|
|
160459
|
+
if (claimed && value.criteria?.results?.some((row2) => !claimed.includes(row2.criterionId))) {
|
|
160460
|
+
ctx.addIssue({
|
|
160461
|
+
code: "custom",
|
|
160462
|
+
path: ["criteria", "results"],
|
|
160463
|
+
message: "Result outside claimed scope"
|
|
160464
|
+
});
|
|
160465
|
+
}
|
|
160466
|
+
});
|
|
160467
|
+
var SWARM_SESSION_TRIAL_STATUSES3 = [
|
|
160468
|
+
"pending",
|
|
160469
|
+
"running",
|
|
160470
|
+
"completed",
|
|
160471
|
+
"failed",
|
|
160472
|
+
"setup_failed",
|
|
160473
|
+
"cancelled"
|
|
160474
|
+
];
|
|
160475
|
+
var swarmSessionTrialSchema3 = z24.object({
|
|
160476
|
+
status: z24.enum(SWARM_SESSION_TRIAL_STATUSES3),
|
|
160477
|
+
taskVerdict: z24.enum(["passed", "failed"]).optional(),
|
|
160478
|
+
evaluatorError: z24.literal(true).optional()
|
|
160479
|
+
}).strict().superRefine((value, ctx) => {
|
|
160480
|
+
const verdict = value.taskVerdict !== void 0;
|
|
160481
|
+
const error = value.evaluatorError === true;
|
|
160482
|
+
if (value.status === "completed" ? verdict === error : verdict || error) {
|
|
160483
|
+
ctx.addIssue({
|
|
160484
|
+
code: "custom",
|
|
160485
|
+
message: "Only completed trials carry exactly one verdict or evaluator error"
|
|
160486
|
+
});
|
|
160487
|
+
}
|
|
160488
|
+
});
|
|
160489
|
+
var swarmSessionGraderCountsSchema3 = z24.object({
|
|
160490
|
+
gating: z24.number().int().nonnegative(),
|
|
160491
|
+
gatingPassed: z24.number().int().nonnegative(),
|
|
160492
|
+
gatingFailed: z24.number().int().nonnegative(),
|
|
160493
|
+
gatingUnmeasured: z24.number().int().nonnegative(),
|
|
160494
|
+
advisoryFailed: z24.number().int().nonnegative()
|
|
160495
|
+
}).strict();
|
|
160496
|
+
var swarmSessionVerdictSchema3 = z24.object({
|
|
160497
|
+
contractVersion: z24.literal(SWARM_SESSION_VERDICT_CONTRACT_VERSION3),
|
|
160498
|
+
lifecycle: swarmSessionLifecycleSchema3,
|
|
160499
|
+
verdict: swarmSessionVerdictValueSchema3,
|
|
160500
|
+
reason: z24.enum(SWARM_SESSION_VERDICT_REASONS3),
|
|
160501
|
+
verdictSource: z24.enum([
|
|
160502
|
+
"goalJudge",
|
|
160503
|
+
"requiredAssertions",
|
|
160504
|
+
"combined",
|
|
160505
|
+
"none"
|
|
160506
|
+
]),
|
|
160507
|
+
grading: swarmGradingReadinessSchema3,
|
|
160508
|
+
graders: z24.object({
|
|
160509
|
+
criteria: z24.enum([
|
|
160510
|
+
"notConfigured",
|
|
160511
|
+
"notClaimed",
|
|
160512
|
+
"pending",
|
|
160513
|
+
"scored",
|
|
160514
|
+
"errored"
|
|
160515
|
+
]),
|
|
160516
|
+
judge: z24.enum([
|
|
160517
|
+
"notConfigured",
|
|
160518
|
+
"silent",
|
|
160519
|
+
"pending",
|
|
160520
|
+
"scored",
|
|
160521
|
+
"errored"
|
|
160522
|
+
])
|
|
160523
|
+
}).strict(),
|
|
160524
|
+
counts: swarmSessionGraderCountsSchema3,
|
|
160525
|
+
trial: swarmSessionTrialSchema3.nullable()
|
|
160526
|
+
}).strict().superRefine((value, ctx) => {
|
|
160527
|
+
const fail3 = (message) => ctx.addIssue({ code: "custom", message });
|
|
160528
|
+
if (SWARM_SESSION_VERDICT_OF_REASON3[value.reason] !== value.verdict)
|
|
160529
|
+
fail3("Reason contradicts verdict");
|
|
160530
|
+
const c = value.counts;
|
|
160531
|
+
if (c.gating !== c.gatingPassed + c.gatingFailed + c.gatingUnmeasured)
|
|
160532
|
+
fail3("Required measurement counts do not add up");
|
|
160533
|
+
const trial = value.trial;
|
|
160534
|
+
if (trial?.status === "completed") {
|
|
160535
|
+
if (value.lifecycle !== "ran")
|
|
160536
|
+
fail3("Only a completed execution can be a completed trial");
|
|
160537
|
+
if (trial.taskVerdict !== void 0 && trial.taskVerdict !== value.verdict)
|
|
160538
|
+
fail3("Trial contradicts goal result");
|
|
160539
|
+
if (trial.evaluatorError && value.verdict !== "inconclusive")
|
|
160540
|
+
fail3("Evaluator error requires inconclusive grading");
|
|
160541
|
+
}
|
|
160542
|
+
const expected = {
|
|
160543
|
+
pending: "pending",
|
|
160544
|
+
running: "running",
|
|
160545
|
+
broke: "failed",
|
|
160546
|
+
limited: "setup_failed",
|
|
160547
|
+
withdrawn: "cancelled"
|
|
160548
|
+
};
|
|
160549
|
+
if (value.lifecycle !== "ran" && trial?.status !== expected[value.lifecycle])
|
|
160550
|
+
fail3("Trial must preserve execution lifecycle");
|
|
160551
|
+
if (value.lifecycle === "ran" && trial !== null && trial.status !== "completed")
|
|
160552
|
+
fail3("Completed execution must not become skipped or running");
|
|
160553
|
+
if ((value.verdict === "passed" || value.verdict === "failed") && value.verdictSource === "none")
|
|
160554
|
+
fail3("A measured goal verdict needs a source");
|
|
160555
|
+
if (value.lifecycle === "ran" && value.verdict !== "notEstablished" && trial === null)
|
|
160556
|
+
fail3("Settled grading on completed execution needs an observation");
|
|
160557
|
+
if (value.verdict === "notEstablished" && value.verdictSource !== "none")
|
|
160558
|
+
fail3("An undecided goal has no verdict source");
|
|
160559
|
+
});
|
|
160560
|
+
var SWARM_FINDING_CONTRACT_VERSION3 = 1;
|
|
160561
|
+
var SWARM_FINDING_DISPOSITIONS3 = [
|
|
160562
|
+
"notRun",
|
|
160563
|
+
"blockedConnecting",
|
|
160564
|
+
"lostFindingTool",
|
|
160565
|
+
"blockedCallingTool",
|
|
160566
|
+
"blockedByResponse",
|
|
160567
|
+
"goalMissed",
|
|
160568
|
+
"goalMetWithFriction",
|
|
160569
|
+
"goalMet",
|
|
160570
|
+
"notMeasured"
|
|
160571
|
+
];
|
|
160572
|
+
var SWARM_FINDING_TONES3 = ["fail", "warn", "ok", "muted"];
|
|
160573
|
+
var SWARM_FINDING_SUMMARY_KINDS3 = [
|
|
160574
|
+
"notLaunched",
|
|
160575
|
+
"broken",
|
|
160576
|
+
"friction",
|
|
160577
|
+
"landed",
|
|
160578
|
+
"ungraded",
|
|
160579
|
+
"unread"
|
|
160580
|
+
];
|
|
160581
|
+
var SWARM_FINDING_COVERAGE_NOTES3 = [
|
|
160582
|
+
"sessionScanCapped",
|
|
160583
|
+
"budgetExhausted",
|
|
160584
|
+
"transcriptMissing",
|
|
160585
|
+
"contextTooLarge",
|
|
160586
|
+
"extractionRejected",
|
|
160587
|
+
"chainUnmeasured",
|
|
160588
|
+
"judgeNotRun",
|
|
160589
|
+
"sessionsWithdrawn",
|
|
160590
|
+
"sessionsRateLimited",
|
|
160591
|
+
"partialRead",
|
|
160592
|
+
"toolCatalogMissing"
|
|
160593
|
+
];
|
|
160594
|
+
var SWARM_FINDING_SCOPE_LEVELS3 = [
|
|
160595
|
+
"session",
|
|
160596
|
+
"goal",
|
|
160597
|
+
"persona",
|
|
160598
|
+
"target",
|
|
160599
|
+
"wave"
|
|
160600
|
+
];
|
|
160601
|
+
var SWARM_FINDING_BASES3 = [
|
|
160602
|
+
"verifiedMechanism",
|
|
160603
|
+
"sessionReport",
|
|
160604
|
+
"populationFact"
|
|
160605
|
+
];
|
|
160606
|
+
var SWARM_FINDING_CHAIN_STAGE_BASES3 = [
|
|
160607
|
+
"derived",
|
|
160608
|
+
"reported",
|
|
160609
|
+
"unmeasured"
|
|
160610
|
+
];
|
|
160611
|
+
var SWARM_FINDING_TONE_OF_DISPOSITION3 = Object.freeze({
|
|
160612
|
+
notRun: "muted",
|
|
160613
|
+
blockedConnecting: "fail",
|
|
160614
|
+
lostFindingTool: "fail",
|
|
160615
|
+
blockedCallingTool: "fail",
|
|
160616
|
+
blockedByResponse: "fail",
|
|
160617
|
+
goalMissed: "fail",
|
|
160618
|
+
goalMetWithFriction: "warn",
|
|
160619
|
+
goalMet: "ok",
|
|
160620
|
+
notMeasured: "muted"
|
|
160621
|
+
});
|
|
160622
|
+
var vocabulary3 = (members) => members.map((value) => "`" + value + "`").join(", ");
|
|
160623
|
+
var count4 = z24.number().int().nonnegative();
|
|
160624
|
+
var persona3 = z24.object({ personaRefId: z24.string().nullable(), name: z24.string() }).strict();
|
|
160625
|
+
var disposition3 = z24.enum(SWARM_FINDING_DISPOSITIONS3).describe(vocabulary3(SWARM_FINDING_DISPOSITIONS3));
|
|
160626
|
+
var coverageNotes3 = z24.array(
|
|
160627
|
+
z24.enum(SWARM_FINDING_COVERAGE_NOTES3).describe(vocabulary3(SWARM_FINDING_COVERAGE_NOTES3))
|
|
160628
|
+
);
|
|
160629
|
+
var citations3 = z24.array(z24.string().regex(/^[^/]+\/.+$/)).max(30);
|
|
160630
|
+
var toneMatchesDisposition3 = (row2) => row2.tone === SWARM_FINDING_TONE_OF_DISPOSITION3[row2.disposition];
|
|
160631
|
+
var toneMismatch3 = {
|
|
160632
|
+
message: "tone must equal SWARM_FINDING_TONE_OF_DISPOSITION[disposition]",
|
|
160633
|
+
path: ["tone"]
|
|
160634
|
+
};
|
|
160635
|
+
var swarmJourneyFindingSchema3 = z24.object({
|
|
160636
|
+
id: z24.string().min(1),
|
|
160637
|
+
basis: z24.enum(SWARM_FINDING_BASES3).describe(vocabulary3(SWARM_FINDING_BASES3)),
|
|
160638
|
+
scopeLevel: z24.enum(SWARM_FINDING_SCOPE_LEVELS3).describe(vocabulary3(SWARM_FINDING_SCOPE_LEVELS3)),
|
|
160639
|
+
persona: persona3,
|
|
160640
|
+
goal: z24.object({
|
|
160641
|
+
runId: z24.string(),
|
|
160642
|
+
journeyRefId: z24.string(),
|
|
160643
|
+
title: z24.string()
|
|
160644
|
+
}).strict(),
|
|
160645
|
+
target: z24.object({
|
|
160646
|
+
kind: z24.enum(["environment", "host"]),
|
|
160647
|
+
id: z24.string(),
|
|
160648
|
+
label: z24.string(),
|
|
160649
|
+
modelId: z24.string().nullable()
|
|
160650
|
+
}).strict(),
|
|
160651
|
+
population: z24.object({ count: count4, total: count4, unit: z24.literal("sessions") }).strict(),
|
|
160652
|
+
sessionIds: z24.array(z24.string()).max(1e3),
|
|
160653
|
+
citations: citations3,
|
|
160654
|
+
verdictSeen: swarmSessionVerdictValueSchema3,
|
|
160655
|
+
chainStage: userValueStageSchema3.nullable().describe(vocabulary3(USER_VALUE_STAGES3)),
|
|
160656
|
+
chainStageState: stageStateSchema3.nullable().describe(vocabulary3(STAGE_STATES3)),
|
|
160657
|
+
chainStageBasis: z24.enum(SWARM_FINDING_CHAIN_STAGE_BASES3),
|
|
160658
|
+
disposition: disposition3,
|
|
160659
|
+
tone: z24.enum(SWARM_FINDING_TONES3),
|
|
160660
|
+
coverageNotes: coverageNotes3,
|
|
160661
|
+
outcomePhrase: z24.string().max(100).regex(/^[^0-9]*$/).refine(
|
|
160662
|
+
(value) => value.trim().split(/\s+/).length <= 8,
|
|
160663
|
+
"At most eight words"
|
|
160664
|
+
).nullable(),
|
|
160665
|
+
mechanismPhrase: z24.string().nullable(),
|
|
160666
|
+
fixPhrase: z24.string().nullable(),
|
|
160667
|
+
reportExcerpt: z24.object({ actual: z24.string().max(1800), citations: citations3 }).strict().nullable(),
|
|
160668
|
+
mechanismId: z24.string().nullable()
|
|
160669
|
+
}).strict().refine(toneMatchesDisposition3, toneMismatch3);
|
|
160670
|
+
z24.object({
|
|
160671
|
+
contractVersion: z24.literal(SWARM_FINDING_CONTRACT_VERSION3),
|
|
160672
|
+
generatedAt: count4,
|
|
160673
|
+
sourceRevision: z24.string(),
|
|
160674
|
+
pipelineVersion: count4,
|
|
160675
|
+
extractionVersion: count4,
|
|
160676
|
+
extractionModel: z24.string(),
|
|
160677
|
+
reasoningModel: z24.string(),
|
|
160678
|
+
summaryKind: z24.enum(SWARM_FINDING_SUMMARY_KINDS3).describe(vocabulary3(SWARM_FINDING_SUMMARY_KINDS3)),
|
|
160679
|
+
population: z24.object({
|
|
160680
|
+
configured: count4,
|
|
160681
|
+
started: count4,
|
|
160682
|
+
read: count4,
|
|
160683
|
+
unread: count4,
|
|
160684
|
+
withdrawn: count4,
|
|
160685
|
+
limited: count4,
|
|
160686
|
+
graded: count4
|
|
160687
|
+
}).strict(),
|
|
160688
|
+
coverageNotes: coverageNotes3,
|
|
160689
|
+
disclosure: z24.object({
|
|
160690
|
+
rail: z24.enum(["gateway", "openrouter"]),
|
|
160691
|
+
evidenceSent: z24.array(z24.string())
|
|
160692
|
+
}).strict(),
|
|
160693
|
+
personas: z24.array(
|
|
160694
|
+
z24.object({
|
|
160695
|
+
persona: persona3,
|
|
160696
|
+
disposition: disposition3,
|
|
160697
|
+
tone: z24.enum(SWARM_FINDING_TONES3),
|
|
160698
|
+
goalRunIds: z24.array(z24.string())
|
|
160699
|
+
}).strict().refine(toneMatchesDisposition3, toneMismatch3)
|
|
160700
|
+
),
|
|
160701
|
+
findings: z24.array(swarmJourneyFindingSchema3).max(200)
|
|
160702
|
+
}).strict();
|
|
160703
|
+
z24.object({
|
|
160704
|
+
status: z24.enum(["pending", "completed", "failed", "skipped"]),
|
|
160705
|
+
errorCode: z24.string().optional(),
|
|
160706
|
+
updatedAt: count4
|
|
160707
|
+
}).strict();
|
|
159995
160708
|
var NEXT_ACTION_BY_FAILURE_CATEGORY3 = Object.freeze({
|
|
159996
160709
|
setup: "check the server connection and environment configuration",
|
|
159997
160710
|
metadata: "review the tool metadata and descriptions in the server catalog",
|
|
@@ -160314,10 +161027,10 @@ function assembleEvalRunDecisionSummary2(input) {
|
|
|
160314
161027
|
}
|
|
160315
161028
|
function legacyTrialCounts2(summary) {
|
|
160316
161029
|
if (!summary || typeof summary !== "object") return void 0;
|
|
160317
|
-
const
|
|
160318
|
-
const total =
|
|
160319
|
-
const passed2 =
|
|
160320
|
-
const failed3 =
|
|
161030
|
+
const count32 = (value) => typeof value === "number" && Number.isInteger(value) && value >= 0 ? value : void 0;
|
|
161031
|
+
const total = count32(summary.total);
|
|
161032
|
+
const passed2 = count32(summary.passed);
|
|
161033
|
+
const failed3 = count32(summary.failed);
|
|
160321
161034
|
if (total === void 0 && passed2 === void 0 && failed3 === void 0) {
|
|
160322
161035
|
return void 0;
|
|
160323
161036
|
}
|
|
@@ -161576,248 +162289,28 @@ var evalBacktestContinuationSchema3 = z24.object({
|
|
|
161576
162289
|
evalBacktestDraftSchema3.extend({
|
|
161577
162290
|
continuation: evalBacktestContinuationSchema3.optional()
|
|
161578
162291
|
});
|
|
161579
|
-
var SWARM_SESSION_VERDICT_CONTRACT_VERSION3 = 1;
|
|
161580
|
-
var SWARM_ATTEMPT_STATUSES3 = [
|
|
161581
|
-
"pending",
|
|
161582
|
-
"running",
|
|
161583
|
-
"succeeded",
|
|
161584
|
-
"failed",
|
|
161585
|
-
"rate_limited"
|
|
161586
|
-
];
|
|
161587
|
-
var SWARM_SESSION_LIFECYCLES3 = [
|
|
161588
|
-
"pending",
|
|
161589
|
-
"running",
|
|
161590
|
-
"ran",
|
|
161591
|
-
"broke",
|
|
161592
|
-
"limited",
|
|
161593
|
-
"withdrawn"
|
|
161594
|
-
];
|
|
161595
|
-
var SWARM_SESSION_VERDICTS3 = [
|
|
161596
|
-
"passed",
|
|
161597
|
-
"failed",
|
|
161598
|
-
"inconclusive",
|
|
161599
|
-
"notEstablished"
|
|
161600
|
-
];
|
|
161601
|
-
var SWARM_GRADING_STATES3 = [
|
|
161602
|
-
"notRequested",
|
|
161603
|
-
"queued",
|
|
161604
|
-
"running",
|
|
161605
|
-
"settled",
|
|
161606
|
-
"unavailable"
|
|
161607
|
-
];
|
|
161608
|
-
var SWARM_SESSION_VERDICT_OF_REASON3 = {
|
|
161609
|
-
attemptPending: "notEstablished",
|
|
161610
|
-
attemptRunning: "notEstablished",
|
|
161611
|
-
withdrawn: "notEstablished",
|
|
161612
|
-
spendCapReached: "notEstablished",
|
|
161613
|
-
notRun: "notEstablished",
|
|
161614
|
-
executionFailed: "notEstablished",
|
|
161615
|
-
ungraded: "notEstablished",
|
|
161616
|
-
gradingNotClaimed: "notEstablished",
|
|
161617
|
-
criteriaPending: "notEstablished",
|
|
161618
|
-
judgePending: "notEstablished",
|
|
161619
|
-
gatingCriterionFailed: "failed",
|
|
161620
|
-
judgeFailed: "failed",
|
|
161621
|
-
criteriaGradingErrored: "inconclusive",
|
|
161622
|
-
gatingCriterionUnmeasured: "inconclusive",
|
|
161623
|
-
judgeErrored: "inconclusive",
|
|
161624
|
-
gradingUnavailable: "inconclusive",
|
|
161625
|
-
allGatingCriteriaPassed: "passed",
|
|
161626
|
-
judgePassed: "passed",
|
|
161627
|
-
allGradersPassed: "passed"
|
|
161628
|
-
};
|
|
161629
|
-
var SWARM_SESSION_VERDICT_REASONS3 = Object.keys(
|
|
161630
|
-
SWARM_SESSION_VERDICT_OF_REASON3
|
|
161631
|
-
);
|
|
161632
|
-
var swarmSessionLifecycleSchema3 = z24.enum(SWARM_SESSION_LIFECYCLES3);
|
|
161633
|
-
var swarmSessionVerdictValueSchema3 = z24.enum(SWARM_SESSION_VERDICTS3);
|
|
161634
|
-
var swarmGraderRoleSchema3 = z24.enum(["advisory", "required"]);
|
|
161635
|
-
var swarmSessionAttemptInputSchema3 = z24.object({
|
|
161636
|
-
status: z24.enum(SWARM_ATTEMPT_STATUSES3),
|
|
161637
|
-
errorCode: z24.string().nullable().optional()
|
|
161638
|
-
}).strict();
|
|
161639
|
-
var swarmGradingReadinessSchema3 = z24.object({
|
|
161640
|
-
/** Durable readiness for decisive grading only, not background observations. */
|
|
161641
|
-
state: z24.enum(SWARM_GRADING_STATES3),
|
|
161642
|
-
reasonCode: z24.string().min(1).optional()
|
|
161643
|
-
}).strict();
|
|
161644
|
-
var criterionResultSchema3 = z24.object({
|
|
161645
|
-
criterionId: z24.string().min(1),
|
|
161646
|
-
passed: z24.boolean(),
|
|
161647
|
-
status: z24.enum(["scored", "error"]).optional()
|
|
161648
|
-
}).strict();
|
|
161649
|
-
z24.object({
|
|
161650
|
-
attempt: swarmSessionAttemptInputSchema3.nullable(),
|
|
161651
|
-
hasTranscript: z24.boolean(),
|
|
161652
|
-
rubric: z24.array(
|
|
161653
|
-
z24.object({
|
|
161654
|
-
id: z24.string().min(1),
|
|
161655
|
-
role: swarmGraderRoleSchema3,
|
|
161656
|
-
predicateType: z24.string().min(1).optional()
|
|
161657
|
-
}).strict()
|
|
161658
|
-
),
|
|
161659
|
-
criteria: z24.object({
|
|
161660
|
-
status: z24.enum(["pending", "completed", "failed"]),
|
|
161661
|
-
criterionIds: z24.array(z24.string().min(1)).optional(),
|
|
161662
|
-
results: z24.array(criterionResultSchema3).optional()
|
|
161663
|
-
}).strict().nullable(),
|
|
161664
|
-
goalScore: z24.discriminatedUnion("status", [
|
|
161665
|
-
z24.object({ status: z24.literal("completed"), passed: z24.boolean() }).strict(),
|
|
161666
|
-
z24.object({ status: z24.literal("running") }).strict(),
|
|
161667
|
-
z24.object({ status: z24.literal("failed") }).strict()
|
|
161668
|
-
]).nullable(),
|
|
161669
|
-
judge: z24.object({
|
|
161670
|
-
automatic: z24.boolean(),
|
|
161671
|
-
role: swarmGraderRoleSchema3,
|
|
161672
|
-
/** Explicit on-demand intent; automatic:false alone does not mean a judge was requested. */
|
|
161673
|
-
requested: z24.boolean().optional()
|
|
161674
|
-
}).strict(),
|
|
161675
|
-
grading: swarmGradingReadinessSchema3
|
|
161676
|
-
}).strict().superRefine((value, ctx) => {
|
|
161677
|
-
const definitions = new Set(value.rubric.map((entry3) => entry3.id));
|
|
161678
|
-
if (definitions.size !== value.rubric.length) {
|
|
161679
|
-
ctx.addIssue({
|
|
161680
|
-
code: "custom",
|
|
161681
|
-
path: ["rubric"],
|
|
161682
|
-
message: "Duplicate criterion definition"
|
|
161683
|
-
});
|
|
161684
|
-
}
|
|
161685
|
-
for (const [field, ids] of [
|
|
161686
|
-
["criterionIds", value.criteria?.criterionIds ?? []],
|
|
161687
|
-
["results", value.criteria?.results?.map((row2) => row2.criterionId) ?? []]
|
|
161688
|
-
]) {
|
|
161689
|
-
if (new Set(ids).size !== ids.length || ids.some((id3) => !definitions.has(id3))) {
|
|
161690
|
-
ctx.addIssue({
|
|
161691
|
-
code: "custom",
|
|
161692
|
-
path: ["criteria", field],
|
|
161693
|
-
message: "Criterion IDs must be unique and defined in the snapshot"
|
|
161694
|
-
});
|
|
161695
|
-
}
|
|
161696
|
-
}
|
|
161697
|
-
const claimed = value.criteria?.criterionIds;
|
|
161698
|
-
if (claimed && value.criteria?.results?.some((row2) => !claimed.includes(row2.criterionId))) {
|
|
161699
|
-
ctx.addIssue({
|
|
161700
|
-
code: "custom",
|
|
161701
|
-
path: ["criteria", "results"],
|
|
161702
|
-
message: "Result outside claimed scope"
|
|
161703
|
-
});
|
|
161704
|
-
}
|
|
161705
|
-
});
|
|
161706
|
-
var SWARM_SESSION_TRIAL_STATUSES3 = [
|
|
161707
|
-
"pending",
|
|
161708
|
-
"running",
|
|
161709
|
-
"completed",
|
|
161710
|
-
"failed",
|
|
161711
|
-
"setup_failed",
|
|
161712
|
-
"cancelled"
|
|
161713
|
-
];
|
|
161714
|
-
var swarmSessionTrialSchema3 = z24.object({
|
|
161715
|
-
status: z24.enum(SWARM_SESSION_TRIAL_STATUSES3),
|
|
161716
|
-
taskVerdict: z24.enum(["passed", "failed"]).optional(),
|
|
161717
|
-
evaluatorError: z24.literal(true).optional()
|
|
161718
|
-
}).strict().superRefine((value, ctx) => {
|
|
161719
|
-
const verdict = value.taskVerdict !== void 0;
|
|
161720
|
-
const error = value.evaluatorError === true;
|
|
161721
|
-
if (value.status === "completed" ? verdict === error : verdict || error) {
|
|
161722
|
-
ctx.addIssue({
|
|
161723
|
-
code: "custom",
|
|
161724
|
-
message: "Only completed trials carry exactly one verdict or evaluator error"
|
|
161725
|
-
});
|
|
161726
|
-
}
|
|
161727
|
-
});
|
|
161728
|
-
var swarmSessionGraderCountsSchema3 = z24.object({
|
|
161729
|
-
gating: z24.number().int().nonnegative(),
|
|
161730
|
-
gatingPassed: z24.number().int().nonnegative(),
|
|
161731
|
-
gatingFailed: z24.number().int().nonnegative(),
|
|
161732
|
-
gatingUnmeasured: z24.number().int().nonnegative(),
|
|
161733
|
-
advisoryFailed: z24.number().int().nonnegative()
|
|
161734
|
-
}).strict();
|
|
161735
|
-
var swarmSessionVerdictSchema3 = z24.object({
|
|
161736
|
-
contractVersion: z24.literal(SWARM_SESSION_VERDICT_CONTRACT_VERSION3),
|
|
161737
|
-
lifecycle: swarmSessionLifecycleSchema3,
|
|
161738
|
-
verdict: swarmSessionVerdictValueSchema3,
|
|
161739
|
-
reason: z24.enum(SWARM_SESSION_VERDICT_REASONS3),
|
|
161740
|
-
verdictSource: z24.enum([
|
|
161741
|
-
"goalJudge",
|
|
161742
|
-
"requiredAssertions",
|
|
161743
|
-
"combined",
|
|
161744
|
-
"none"
|
|
161745
|
-
]),
|
|
161746
|
-
grading: swarmGradingReadinessSchema3,
|
|
161747
|
-
graders: z24.object({
|
|
161748
|
-
criteria: z24.enum([
|
|
161749
|
-
"notConfigured",
|
|
161750
|
-
"notClaimed",
|
|
161751
|
-
"pending",
|
|
161752
|
-
"scored",
|
|
161753
|
-
"errored"
|
|
161754
|
-
]),
|
|
161755
|
-
judge: z24.enum([
|
|
161756
|
-
"notConfigured",
|
|
161757
|
-
"silent",
|
|
161758
|
-
"pending",
|
|
161759
|
-
"scored",
|
|
161760
|
-
"errored"
|
|
161761
|
-
])
|
|
161762
|
-
}).strict(),
|
|
161763
|
-
counts: swarmSessionGraderCountsSchema3,
|
|
161764
|
-
trial: swarmSessionTrialSchema3.nullable()
|
|
161765
|
-
}).strict().superRefine((value, ctx) => {
|
|
161766
|
-
const fail3 = (message) => ctx.addIssue({ code: "custom", message });
|
|
161767
|
-
if (SWARM_SESSION_VERDICT_OF_REASON3[value.reason] !== value.verdict)
|
|
161768
|
-
fail3("Reason contradicts verdict");
|
|
161769
|
-
const c = value.counts;
|
|
161770
|
-
if (c.gating !== c.gatingPassed + c.gatingFailed + c.gatingUnmeasured)
|
|
161771
|
-
fail3("Required measurement counts do not add up");
|
|
161772
|
-
const trial = value.trial;
|
|
161773
|
-
if (trial?.status === "completed") {
|
|
161774
|
-
if (value.lifecycle !== "ran")
|
|
161775
|
-
fail3("Only a completed execution can be a completed trial");
|
|
161776
|
-
if (trial.taskVerdict !== void 0 && trial.taskVerdict !== value.verdict)
|
|
161777
|
-
fail3("Trial contradicts goal result");
|
|
161778
|
-
if (trial.evaluatorError && value.verdict !== "inconclusive")
|
|
161779
|
-
fail3("Evaluator error requires inconclusive grading");
|
|
161780
|
-
}
|
|
161781
|
-
const expected = {
|
|
161782
|
-
pending: "pending",
|
|
161783
|
-
running: "running",
|
|
161784
|
-
broke: "failed",
|
|
161785
|
-
limited: "setup_failed",
|
|
161786
|
-
withdrawn: "cancelled"
|
|
161787
|
-
};
|
|
161788
|
-
if (value.lifecycle !== "ran" && trial?.status !== expected[value.lifecycle])
|
|
161789
|
-
fail3("Trial must preserve execution lifecycle");
|
|
161790
|
-
if (value.lifecycle === "ran" && trial !== null && trial.status !== "completed")
|
|
161791
|
-
fail3("Completed execution must not become skipped or running");
|
|
161792
|
-
if ((value.verdict === "passed" || value.verdict === "failed") && value.verdictSource === "none")
|
|
161793
|
-
fail3("A measured goal verdict needs a source");
|
|
161794
|
-
if (value.lifecycle === "ran" && value.verdict !== "notEstablished" && trial === null)
|
|
161795
|
-
fail3("Settled grading on completed execution needs an observation");
|
|
161796
|
-
if (value.verdict === "notEstablished" && value.verdictSource !== "none")
|
|
161797
|
-
fail3("An undecided goal has no verdict source");
|
|
161798
|
-
});
|
|
161799
162292
|
var SWARM_REPORT_CONTRACT_VERSION3 = 1;
|
|
161800
|
-
var
|
|
162293
|
+
var count23 = z24.number().int().nonnegative();
|
|
161801
162294
|
var journeyRunVerdictSummarySchema3 = z24.discriminatedUnion("status", [
|
|
161802
162295
|
z24.object({
|
|
161803
162296
|
status: z24.literal("pending"),
|
|
161804
|
-
pendingSessions:
|
|
161805
|
-
updatedAt:
|
|
162297
|
+
pendingSessions: count23,
|
|
162298
|
+
updatedAt: count23
|
|
161806
162299
|
}).strict(),
|
|
161807
162300
|
z24.object({
|
|
161808
162301
|
status: z24.literal("decided"),
|
|
161809
162302
|
decision: evalVerdictDecisionSchema3,
|
|
161810
|
-
updatedAt:
|
|
162303
|
+
updatedAt: count23
|
|
161811
162304
|
}).strict(),
|
|
161812
162305
|
z24.object({
|
|
161813
162306
|
status: z24.literal("notEstablished"),
|
|
161814
162307
|
reason: z24.literal("gradingNotConfigured"),
|
|
161815
|
-
updatedAt:
|
|
162308
|
+
updatedAt: count23
|
|
161816
162309
|
}).strict(),
|
|
161817
162310
|
z24.object({
|
|
161818
162311
|
status: z24.literal("integrityFailed"),
|
|
161819
162312
|
reason: z24.string().min(1),
|
|
161820
|
-
updatedAt:
|
|
162313
|
+
updatedAt: count23
|
|
161821
162314
|
}).strict()
|
|
161822
162315
|
]);
|
|
161823
162316
|
var swarmObservationInputSchema3 = z24.object({
|
|
@@ -161855,7 +162348,7 @@ var swarmReportSessionSchema3 = z24.object({
|
|
|
161855
162348
|
z24.object({
|
|
161856
162349
|
runId: z24.string().min(1),
|
|
161857
162350
|
executionComplete: z24.boolean(),
|
|
161858
|
-
configuredSessions:
|
|
162351
|
+
configuredSessions: count23,
|
|
161859
162352
|
verdictSummary: journeyRunVerdictSummarySchema3.nullable(),
|
|
161860
162353
|
/** Frozen run-wide definitions: absent measurement rows still count as unavailable. */
|
|
161861
162354
|
evaluatorDefinitions: z24.array(
|
|
@@ -161910,11 +162403,11 @@ var swarmObservationCoverageSchema3 = z24.object({
|
|
|
161910
162403
|
"The stage this observation was read from, and the only stage it is evidence ABOUT. `connection` \u2014 the server was reachable and the session initialized. `discovery` \u2014 its tools and resources were listed and readable. `selection` \u2014 the model chose the right tool for the request. `call` \u2014 the call was made with usable arguments. `response` \u2014 the server returned data the model could use. `userValue` \u2014 the user's actual request was satisfied."
|
|
161911
162404
|
),
|
|
161912
162405
|
unit: z24.literal("sessions"),
|
|
161913
|
-
total:
|
|
161914
|
-
passed:
|
|
161915
|
-
failed:
|
|
161916
|
-
pending:
|
|
161917
|
-
unavailable:
|
|
162406
|
+
total: count23,
|
|
162407
|
+
passed: count23,
|
|
162408
|
+
failed: count23,
|
|
162409
|
+
pending: count23,
|
|
162410
|
+
unavailable: count23
|
|
161918
162411
|
}).strict().superRefine((value, ctx) => {
|
|
161919
162412
|
if (value.total !== value.passed + value.failed + value.pending + value.unavailable) {
|
|
161920
162413
|
ctx.addIssue({
|
|
@@ -161944,26 +162437,26 @@ z24.object({
|
|
|
161944
162437
|
]).optional(),
|
|
161945
162438
|
execution: z24.object({
|
|
161946
162439
|
unit: z24.literal("sessions"),
|
|
161947
|
-
configured:
|
|
161948
|
-
reported:
|
|
161949
|
-
started:
|
|
161950
|
-
notStarted:
|
|
161951
|
-
unknown:
|
|
161952
|
-
completed:
|
|
161953
|
-
interrupted:
|
|
162440
|
+
configured: count23,
|
|
162441
|
+
reported: count23,
|
|
162442
|
+
started: count23,
|
|
162443
|
+
notStarted: count23,
|
|
162444
|
+
unknown: count23,
|
|
162445
|
+
completed: count23,
|
|
162446
|
+
interrupted: count23,
|
|
161954
162447
|
/** Only explicit complete execution evidence can establish this. */
|
|
161955
162448
|
neverLaunched: z24.boolean()
|
|
161956
162449
|
}).strict(),
|
|
161957
162450
|
goalGrading: z24.object({
|
|
161958
162451
|
unit: z24.literal("sessions"),
|
|
161959
|
-
reported:
|
|
161960
|
-
passed:
|
|
161961
|
-
failed:
|
|
161962
|
-
pending:
|
|
161963
|
-
unavailable:
|
|
161964
|
-
notRequested:
|
|
162452
|
+
reported: count23,
|
|
162453
|
+
passed: count23,
|
|
162454
|
+
failed: count23,
|
|
162455
|
+
pending: count23,
|
|
162456
|
+
unavailable: count23,
|
|
162457
|
+
notRequested: count23,
|
|
161965
162458
|
/** Readiness is separate: a proven failure can exist while other required grading is pending. */
|
|
161966
|
-
waitingForDecisiveGrading:
|
|
162459
|
+
waitingForDecisiveGrading: count23
|
|
161967
162460
|
}).strict(),
|
|
161968
162461
|
observations: z24.array(swarmObservationCoverageSchema3)
|
|
161969
162462
|
}).strict().superRefine((report, ctx) => {
|
|
@@ -162842,27 +163335,27 @@ function evaluateHostCompat(requirements, profile) {
|
|
|
162842
163335
|
(name15) => !requirements.appOnlyWidgets.includes(name15)
|
|
162843
163336
|
);
|
|
162844
163337
|
if (blockedAppOnly.length > 0) {
|
|
162845
|
-
const
|
|
163338
|
+
const count32 = blockedAppOnly.length;
|
|
162846
163339
|
findings.push({
|
|
162847
163340
|
lane: "apps",
|
|
162848
163341
|
severity: "blocker",
|
|
162849
163342
|
code: "app_only_unrenderable",
|
|
162850
163343
|
tools: blockedAppOnly,
|
|
162851
|
-
title: `${
|
|
162852
|
-
detail: `${formatToolNames(blockedAppOnly)} only ${
|
|
163344
|
+
title: `${count32 === 1 ? "Interactive tool" : `${count32} interactive tools`} unavailable`,
|
|
163345
|
+
detail: `${formatToolNames(blockedAppOnly)} only ${count32 === 1 ? "works" : "work"} inside a widget. ${profile.label} does not render ${count32 === 1 ? "it" : "them"}.`,
|
|
162853
163346
|
remediation,
|
|
162854
163347
|
provenance: profile.provenance
|
|
162855
163348
|
});
|
|
162856
163349
|
}
|
|
162857
163350
|
if (degradedFallback.length > 0) {
|
|
162858
|
-
const
|
|
163351
|
+
const count32 = degradedFallback.length;
|
|
162859
163352
|
findings.push({
|
|
162860
163353
|
lane: "apps",
|
|
162861
163354
|
severity: "degraded",
|
|
162862
163355
|
code: "widget_text_fallback",
|
|
162863
163356
|
tools: degradedFallback,
|
|
162864
|
-
title: `${
|
|
162865
|
-
detail: `${formatToolNames(degradedFallback)} provide${
|
|
163357
|
+
title: `${count32 === 1 ? "Interactive view" : `${count32} interactive views`} unavailable`,
|
|
163358
|
+
detail: `${formatToolNames(degradedFallback)} provide${count32 === 1 ? "s" : ""} an interactive view. ${profile.label} shows the plain-text result instead.`,
|
|
162866
163359
|
remediation,
|
|
162867
163360
|
provenance: profile.provenance
|
|
162868
163361
|
});
|
|
@@ -175105,11 +175598,11 @@ var getPersonaOperation = {
|
|
|
175105
175598
|
{ client: client3, signal, onScopeResolved },
|
|
175106
175599
|
input.project
|
|
175107
175600
|
);
|
|
175108
|
-
const
|
|
175601
|
+
const persona22 = await client3.getPersona(
|
|
175109
175602
|
{ projectId: project.id, personaId: input.persona },
|
|
175110
175603
|
{ signal }
|
|
175111
175604
|
);
|
|
175112
|
-
return { project: toSelectedProjectInfo(project), persona };
|
|
175605
|
+
return { project: toSelectedProjectInfo(project), persona: persona22 };
|
|
175113
175606
|
}
|
|
175114
175607
|
};
|
|
175115
175608
|
var createPersonaInput = z24.object({
|
|
@@ -175139,7 +175632,7 @@ var createPersonaOperation = {
|
|
|
175139
175632
|
{ client: client3, signal, onScopeResolved },
|
|
175140
175633
|
input.project
|
|
175141
175634
|
);
|
|
175142
|
-
const
|
|
175635
|
+
const persona22 = await client3.createPersona(
|
|
175143
175636
|
{
|
|
175144
175637
|
projectId: project.id,
|
|
175145
175638
|
name: input.name,
|
|
@@ -175151,7 +175644,7 @@ var createPersonaOperation = {
|
|
|
175151
175644
|
...input.idempotencyKey ? { idempotencyKey: input.idempotencyKey } : {}
|
|
175152
175645
|
}
|
|
175153
175646
|
);
|
|
175154
|
-
return { project: toSelectedProjectInfo(project), persona };
|
|
175647
|
+
return { project: toSelectedProjectInfo(project), persona: persona22 };
|
|
175155
175648
|
}
|
|
175156
175649
|
};
|
|
175157
175650
|
var updatePersonaInput = personaSelectorInput.extend({
|
|
@@ -175175,7 +175668,7 @@ var updatePersonaOperation = {
|
|
|
175175
175668
|
{ client: client3, signal, onScopeResolved },
|
|
175176
175669
|
input.project
|
|
175177
175670
|
);
|
|
175178
|
-
const
|
|
175671
|
+
const persona22 = await client3.updatePersona(
|
|
175179
175672
|
{
|
|
175180
175673
|
projectId: project.id,
|
|
175181
175674
|
personaId: input.persona,
|
|
@@ -175185,7 +175678,7 @@ var updatePersonaOperation = {
|
|
|
175185
175678
|
},
|
|
175186
175679
|
{ signal }
|
|
175187
175680
|
);
|
|
175188
|
-
return { project: toSelectedProjectInfo(project), persona };
|
|
175681
|
+
return { project: toSelectedProjectInfo(project), persona: persona22 };
|
|
175189
175682
|
}
|
|
175190
175683
|
};
|
|
175191
175684
|
var deletePersonaOperation = {
|
|
@@ -175201,11 +175694,11 @@ var deletePersonaOperation = {
|
|
|
175201
175694
|
{ client: client3, signal, onScopeResolved },
|
|
175202
175695
|
input.project
|
|
175203
175696
|
);
|
|
175204
|
-
const
|
|
175697
|
+
const persona22 = await client3.deletePersona(
|
|
175205
175698
|
{ projectId: project.id, personaId: input.persona },
|
|
175206
175699
|
{ signal }
|
|
175207
175700
|
);
|
|
175208
|
-
return { project: toSelectedProjectInfo(project), persona };
|
|
175701
|
+
return { project: toSelectedProjectInfo(project), persona: persona22 };
|
|
175209
175702
|
}
|
|
175210
175703
|
};
|
|
175211
175704
|
var secretSelectorInput = z24.object({
|
|
@@ -187960,24 +188453,24 @@ function appendVary(c) {
|
|
|
187960
188453
|
if (names.includes(EVAL_VOCABULARY_HEADER2)) return;
|
|
187961
188454
|
c.header?.("Vary", `${existing}, ${EVAL_VOCABULARY_HEADER2}`);
|
|
187962
188455
|
}
|
|
187963
|
-
function projectRoleForVocabulary(role,
|
|
188456
|
+
function projectRoleForVocabulary(role, vocabulary4) {
|
|
187964
188457
|
if (role === void 0) return void 0;
|
|
187965
188458
|
if (role === "advisory") return "advisory";
|
|
187966
188459
|
if (!isRequiredRole(role)) return role;
|
|
187967
|
-
return
|
|
188460
|
+
return vocabulary4 === 2 ? "required" : "gating";
|
|
187968
188461
|
}
|
|
187969
|
-
function refusesCanonicalRole(role,
|
|
187970
|
-
return
|
|
188462
|
+
function refusesCanonicalRole(role, vocabulary4) {
|
|
188463
|
+
return vocabulary4 === 1 && role === "required";
|
|
187971
188464
|
}
|
|
187972
188465
|
function canonicalRoleRefusalMessage(path8) {
|
|
187973
188466
|
return `${path8}: role "required" needs ${EVAL_VOCABULARY_HEADER2}: 2. Without the header this endpoint speaks today's contract, where the spelling is "gating".`;
|
|
187974
188467
|
}
|
|
187975
|
-
function normalizeCheckRolesForVocabulary(checks2,
|
|
188468
|
+
function normalizeCheckRolesForVocabulary(checks2, vocabulary4, path8) {
|
|
187976
188469
|
if (!Array.isArray(checks2)) return checks2;
|
|
187977
188470
|
let changed = false;
|
|
187978
188471
|
const out = checks2.map((check, index) => {
|
|
187979
188472
|
const role = check?.role;
|
|
187980
|
-
if (refusesCanonicalRole(role,
|
|
188473
|
+
if (refusesCanonicalRole(role, vocabulary4)) {
|
|
187981
188474
|
throw new WebRouteError(
|
|
187982
188475
|
400,
|
|
187983
188476
|
ErrorCode.VALIDATION_ERROR,
|
|
@@ -187991,18 +188484,18 @@ function normalizeCheckRolesForVocabulary(checks2, vocabulary, path8) {
|
|
|
187991
188484
|
});
|
|
187992
188485
|
return changed ? out : checks2;
|
|
187993
188486
|
}
|
|
187994
|
-
function normalizeCheckRolesInOverrideForVocabulary(override,
|
|
188487
|
+
function normalizeCheckRolesInOverrideForVocabulary(override, vocabulary4, path8) {
|
|
187995
188488
|
if (!override || typeof override !== "object") return override;
|
|
187996
188489
|
const list = override.list;
|
|
187997
188490
|
if (!Array.isArray(list)) return override;
|
|
187998
188491
|
const next = normalizeCheckRolesForVocabulary(
|
|
187999
188492
|
list,
|
|
188000
|
-
|
|
188493
|
+
vocabulary4,
|
|
188001
188494
|
`${path8}.list`
|
|
188002
188495
|
);
|
|
188003
188496
|
return next === list ? override : { ...override, list: next };
|
|
188004
188497
|
}
|
|
188005
|
-
function normalizeStepRolesForVocabulary(steps,
|
|
188498
|
+
function normalizeStepRolesForVocabulary(steps, vocabulary4, path8) {
|
|
188006
188499
|
if (!Array.isArray(steps)) return steps;
|
|
188007
188500
|
let changed = false;
|
|
188008
188501
|
const out = steps.map((step, index) => {
|
|
@@ -188010,7 +188503,7 @@ function normalizeStepRolesForVocabulary(steps, vocabulary, path8) {
|
|
|
188010
188503
|
if (!row2 || row2.kind !== "assert") return step;
|
|
188011
188504
|
const assertion2 = row2.assertion;
|
|
188012
188505
|
const role = assertion2?.role;
|
|
188013
|
-
if (refusesCanonicalRole(role,
|
|
188506
|
+
if (refusesCanonicalRole(role, vocabulary4)) {
|
|
188014
188507
|
throw new WebRouteError(
|
|
188015
188508
|
400,
|
|
188016
188509
|
ErrorCode.VALIDATION_ERROR,
|
|
@@ -188024,10 +188517,10 @@ function normalizeStepRolesForVocabulary(steps, vocabulary, path8) {
|
|
|
188024
188517
|
});
|
|
188025
188518
|
return changed ? out : steps;
|
|
188026
188519
|
}
|
|
188027
|
-
function normalizeJudgeRoleForVocabulary(slot,
|
|
188520
|
+
function normalizeJudgeRoleForVocabulary(slot, vocabulary4, path8) {
|
|
188028
188521
|
if (!slot || typeof slot !== "object") return slot;
|
|
188029
188522
|
const role = slot.role;
|
|
188030
|
-
if (refusesCanonicalRole(role,
|
|
188523
|
+
if (refusesCanonicalRole(role, vocabulary4)) {
|
|
188031
188524
|
throw new WebRouteError(
|
|
188032
188525
|
400,
|
|
188033
188526
|
ErrorCode.VALIDATION_ERROR,
|
|
@@ -188037,9 +188530,9 @@ function normalizeJudgeRoleForVocabulary(slot, vocabulary, path8) {
|
|
|
188037
188530
|
if (role !== "required") return slot;
|
|
188038
188531
|
return { ...slot, role: "gating" };
|
|
188039
188532
|
}
|
|
188040
|
-
function projectCheckRolesForVocabulary(checks2,
|
|
188533
|
+
function projectCheckRolesForVocabulary(checks2, vocabulary4) {
|
|
188041
188534
|
if (!Array.isArray(checks2)) return checks2;
|
|
188042
|
-
if (
|
|
188535
|
+
if (vocabulary4 === 1) return checks2;
|
|
188043
188536
|
let changed = false;
|
|
188044
188537
|
const out = checks2.map((check) => {
|
|
188045
188538
|
const role = check?.role;
|
|
@@ -188049,9 +188542,9 @@ function projectCheckRolesForVocabulary(checks2, vocabulary) {
|
|
|
188049
188542
|
});
|
|
188050
188543
|
return changed ? out : checks2;
|
|
188051
188544
|
}
|
|
188052
|
-
function projectStepRolesForVocabulary(steps,
|
|
188545
|
+
function projectStepRolesForVocabulary(steps, vocabulary4) {
|
|
188053
188546
|
if (!Array.isArray(steps)) return steps;
|
|
188054
|
-
if (
|
|
188547
|
+
if (vocabulary4 === 1) return steps;
|
|
188055
188548
|
let changed = false;
|
|
188056
188549
|
const out = steps.map((step) => {
|
|
188057
188550
|
const row2 = step;
|
|
@@ -188104,7 +188597,7 @@ function addBothSpellingsIssues(body, ctx, pairs) {
|
|
|
188104
188597
|
// server/routes/v1/eval-score-projection.ts
|
|
188105
188598
|
var ITERATION_SCORE_INTEGRITY = /* @__PURE__ */ new Set(["score_integrity_invalid"]);
|
|
188106
188599
|
var RUN_SCORE_INTEGRITY = /* @__PURE__ */ new Set(["valid", "invalid"]);
|
|
188107
|
-
function toScoreProjection(metadata,
|
|
188600
|
+
function toScoreProjection(metadata, vocabulary4 = 1) {
|
|
188108
188601
|
if (!metadata || typeof metadata !== "object") return {};
|
|
188109
188602
|
const record5 = metadata;
|
|
188110
188603
|
const integrity = typeof record5.scoreIntegrity === "string" && ITERATION_SCORE_INTEGRITY.has(record5.scoreIntegrity) ? { scoreIntegrity: record5.scoreIntegrity } : {};
|
|
@@ -188123,7 +188616,7 @@ function toScoreProjection(metadata, vocabulary = 1) {
|
|
|
188123
188616
|
evaluationConfig: {
|
|
188124
188617
|
...config.data,
|
|
188125
188618
|
definitions: config.data.definitions.map((definition) => {
|
|
188126
|
-
const role = projectRoleForVocabulary(definition.role,
|
|
188619
|
+
const role = projectRoleForVocabulary(definition.role, vocabulary4);
|
|
188127
188620
|
return role === definition.role ? definition : { ...definition, role };
|
|
188128
188621
|
})
|
|
188129
188622
|
}
|
|
@@ -188300,12 +188793,12 @@ function costCoverage(value) {
|
|
|
188300
188793
|
const source = value;
|
|
188301
188794
|
const side = (raw) => {
|
|
188302
188795
|
const inner = isRecord29(raw) ? raw : {};
|
|
188303
|
-
const
|
|
188796
|
+
const count5 = (raw2) => {
|
|
188304
188797
|
const n = numOrNull(raw2);
|
|
188305
188798
|
return typeof n === "number" && Number.isSafeInteger(n) && n >= 0 ? n : 0;
|
|
188306
188799
|
};
|
|
188307
|
-
const costed =
|
|
188308
|
-
const total =
|
|
188800
|
+
const costed = count5(inner.costed);
|
|
188801
|
+
const total = count5(inner.total);
|
|
188309
188802
|
return { costed: costed <= total ? costed : 0, total };
|
|
188310
188803
|
};
|
|
188311
188804
|
return { base: side(source.base), compare: side(source.compare) };
|
|
@@ -188441,8 +188934,8 @@ function toRunCompareDto(diff, baseline) {
|
|
|
188441
188934
|
import { ConvexHttpClient as ConvexHttpClient10 } from "convex/browser";
|
|
188442
188935
|
|
|
188443
188936
|
// server/routes/v1/eval-case-vocabulary-2.ts
|
|
188444
|
-
function countFieldNames(
|
|
188445
|
-
return
|
|
188937
|
+
function countFieldNames(vocabulary4) {
|
|
188938
|
+
return vocabulary4 === 2 ? {
|
|
188446
188939
|
floor: "legacyIterations",
|
|
188447
188940
|
exactCount: "iterations",
|
|
188448
188941
|
settingsExactCount: "settings.iterations"
|
|
@@ -188528,8 +189021,8 @@ function renameKeys(value, renames) {
|
|
|
188528
189021
|
}
|
|
188529
189022
|
return out;
|
|
188530
189023
|
}
|
|
188531
|
-
function projectCaseDto(dto,
|
|
188532
|
-
if (
|
|
189024
|
+
function projectCaseDto(dto, vocabulary4) {
|
|
189025
|
+
if (vocabulary4 === 1) return dto;
|
|
188533
189026
|
return renameKeys(dto, CASE_DTO_RENAMES);
|
|
188534
189027
|
}
|
|
188535
189028
|
function suiteSettingsShapeV2(v12) {
|
|
@@ -188567,8 +189060,8 @@ function foldSuiteSettingsV2ToV1(settings) {
|
|
|
188567
189060
|
...exact !== void 0 ? { repetitions: exact } : {}
|
|
188568
189061
|
};
|
|
188569
189062
|
}
|
|
188570
|
-
function projectSuiteDetailDto(dto,
|
|
188571
|
-
if (
|
|
189063
|
+
function projectSuiteDetailDto(dto, vocabulary4) {
|
|
189064
|
+
if (vocabulary4 === 1) return dto;
|
|
188572
189065
|
const settings = renameKeys(dto.settings, { checks: "defaultAssertions" });
|
|
188573
189066
|
if (dto.settings.verdictPolicyDefaults) {
|
|
188574
189067
|
settings.verdictPolicyDefaults = renameKeys(
|
|
@@ -206649,7 +207142,7 @@ var publicInlineTestSchema = z35.strictObject({
|
|
|
206649
207142
|
});
|
|
206650
207143
|
}
|
|
206651
207144
|
});
|
|
206652
|
-
function publicInlineTestToRunTest(input,
|
|
207145
|
+
function publicInlineTestToRunTest(input, vocabulary4) {
|
|
206653
207146
|
const test = foldInlineTestAliases(input);
|
|
206654
207147
|
const derived = stepsToInternalCaseFields(test.steps);
|
|
206655
207148
|
return {
|
|
@@ -206659,7 +207152,7 @@ function publicInlineTestToRunTest(input, vocabulary) {
|
|
|
206659
207152
|
steps: [
|
|
206660
207153
|
...normalizeStepRolesForVocabulary(
|
|
206661
207154
|
withImplicitRenderAssertForSingleToolCall(test.steps),
|
|
206662
|
-
|
|
207155
|
+
vocabulary4,
|
|
206663
207156
|
"tests[].steps"
|
|
206664
207157
|
)
|
|
206665
207158
|
],
|
|
@@ -206678,7 +207171,7 @@ function publicInlineTestToRunTest(input, vocabulary) {
|
|
|
206678
207171
|
...test.predicates !== void 0 ? {
|
|
206679
207172
|
predicates: normalizeCheckRolesInOverrideForVocabulary(
|
|
206680
207173
|
test.predicates,
|
|
206681
|
-
|
|
207174
|
+
vocabulary4,
|
|
206682
207175
|
"tests[].predicates"
|
|
206683
207176
|
)
|
|
206684
207177
|
} : {},
|
|
@@ -206847,7 +207340,7 @@ var syncFileOwnedSuiteSchema = z35.object({
|
|
|
206847
207340
|
});
|
|
206848
207341
|
}
|
|
206849
207342
|
});
|
|
206850
|
-
function normalizeCreateTestsToRunTests(tests, suite,
|
|
207343
|
+
function normalizeCreateTestsToRunTests(tests, suite, vocabulary4) {
|
|
206851
207344
|
return tests.map((input) => {
|
|
206852
207345
|
const test = foldInlineTestAliases(input);
|
|
206853
207346
|
const runs = test.runs ?? 1;
|
|
@@ -206859,7 +207352,7 @@ function normalizeCreateTestsToRunTests(tests, suite, vocabulary) {
|
|
|
206859
207352
|
steps: [
|
|
206860
207353
|
...normalizeStepRolesForVocabulary(
|
|
206861
207354
|
withImplicitRenderAssertForSingleToolCall(test.steps),
|
|
206862
|
-
|
|
207355
|
+
vocabulary4,
|
|
206863
207356
|
"tests[].steps"
|
|
206864
207357
|
)
|
|
206865
207358
|
],
|
|
@@ -206877,7 +207370,7 @@ function normalizeCreateTestsToRunTests(tests, suite, vocabulary) {
|
|
|
206877
207370
|
...test.predicates !== void 0 ? {
|
|
206878
207371
|
predicates: normalizeCheckRolesInOverrideForVocabulary(
|
|
206879
207372
|
test.predicates,
|
|
206880
|
-
|
|
207373
|
+
vocabulary4,
|
|
206881
207374
|
"tests[].predicates"
|
|
206882
207375
|
)
|
|
206883
207376
|
} : {},
|
|
@@ -207460,7 +207953,7 @@ function toImportEligibilityProjection(raw) {
|
|
|
207460
207953
|
}
|
|
207461
207954
|
};
|
|
207462
207955
|
}
|
|
207463
|
-
function toIterationDto(iteration,
|
|
207956
|
+
function toIterationDto(iteration, vocabulary4 = 1) {
|
|
207464
207957
|
const snapshot2 = iteration.testCaseSnapshot ?? {};
|
|
207465
207958
|
const startedAt = typeof iteration.startedAt === "number" ? iteration.startedAt : null;
|
|
207466
207959
|
const isTerminal = TERMINAL_ITERATION_STATUSES.has(iteration.status);
|
|
@@ -207493,7 +207986,7 @@ function toIterationDto(iteration, vocabulary = 1) {
|
|
|
207493
207986
|
expectedToolCalls: snapshot2.expectedToolCalls ?? [],
|
|
207494
207987
|
...snapshot2.isNegativeTest === true ? { isNegativeTest: true } : {},
|
|
207495
207988
|
error: iteration.error ?? null,
|
|
207496
|
-
...toScoreProjection(iteration.metadata,
|
|
207989
|
+
...toScoreProjection(iteration.metadata, vocabulary4),
|
|
207497
207990
|
...toStageProjection(iteration.metadata),
|
|
207498
207991
|
// Observable patterns in this trial's tool calls — a report beside the
|
|
207499
207992
|
// verdict, never an input to one. ABSENT for every iteration that
|
|
@@ -207611,7 +208104,7 @@ function toPublicCaseImportClaim(raw) {
|
|
|
207611
208104
|
...typeof record5.note === "string" && record5.note ? { note: record5.note } : {}
|
|
207612
208105
|
};
|
|
207613
208106
|
}
|
|
207614
|
-
function toCaseDto(testCase,
|
|
208107
|
+
function toCaseDto(testCase, vocabulary4 = 1) {
|
|
207615
208108
|
const importClaim = toPublicCaseImportClaim(testCase.import);
|
|
207616
208109
|
return {
|
|
207617
208110
|
id: String(testCase._id),
|
|
@@ -207629,7 +208122,7 @@ function toCaseDto(testCase, vocabulary = 1) {
|
|
|
207629
208122
|
// the case's own checks — so it projects the same way.
|
|
207630
208123
|
steps: projectStepRolesForVocabulary(
|
|
207631
208124
|
internalCaseToSteps(testCase),
|
|
207632
|
-
|
|
208125
|
+
vocabulary4
|
|
207633
208126
|
),
|
|
207634
208127
|
...testCase.expectedOutput !== void 0 ? { expectedOutput: testCase.expectedOutput } : {},
|
|
207635
208128
|
iterations: typeof testCase.runs === "number" ? testCase.runs : 1,
|
|
@@ -207664,7 +208157,7 @@ function toCaseDto(testCase, vocabulary = 1) {
|
|
|
207664
208157
|
// unannounced rename in a response would empty its gating set.
|
|
207665
208158
|
list: projectCheckRolesForVocabulary(
|
|
207666
208159
|
testCase.predicates.list ?? [],
|
|
207667
|
-
|
|
208160
|
+
vocabulary4
|
|
207668
208161
|
) ?? []
|
|
207669
208162
|
}
|
|
207670
208163
|
} : {},
|
|
@@ -207680,7 +208173,7 @@ function isCiOwnedSuiteDoc(suite) {
|
|
|
207680
208173
|
const declared = suite.declaredSuiteId;
|
|
207681
208174
|
return typeof declared === "string" && declared.length > 0 || suite.source === "sdk";
|
|
207682
208175
|
}
|
|
207683
|
-
function toSuiteDetailDto(suite, execConfig, resolved = {},
|
|
208176
|
+
function toSuiteDetailDto(suite, execConfig, resolved = {}, vocabulary4 = 1) {
|
|
207684
208177
|
const goal = suite.judgePolicy?.effective ?? suite.judgeConfig?.goalCompletion;
|
|
207685
208178
|
return {
|
|
207686
208179
|
id: String(suite._id),
|
|
@@ -207751,7 +208244,7 @@ function toSuiteDetailDto(suite, execConfig, resolved = {}, vocabulary = 1) {
|
|
|
207751
208244
|
matchOptions: toPublicMatchOptions(suite.defaultMatchOptions),
|
|
207752
208245
|
checks: projectCheckRolesForVocabulary(
|
|
207753
208246
|
Array.isArray(suite.defaultPredicates) ? suite.defaultPredicates : [],
|
|
207754
|
-
|
|
208247
|
+
vocabulary4
|
|
207755
208248
|
) ?? [],
|
|
207756
208249
|
// FULLY RESOLVED, every field layered over GOAL_COMPLETION_DEFAULTS —
|
|
207757
208250
|
// the same resolution the backend's `resolveGoalCompletionConfig`
|
|
@@ -207778,7 +208271,7 @@ function toSuiteDetailDto(suite, execConfig, resolved = {}, vocabulary = 1) {
|
|
|
207778
208271
|
// Projected, never echoed raw: after the storage switch this field
|
|
207779
208272
|
// holds `required`, and a published `eval gate` filters judge roles on
|
|
207780
208273
|
// the literal `"gating"`.
|
|
207781
|
-
...goal?.role !== void 0 ? { role: projectRoleForVocabulary(goal.role,
|
|
208274
|
+
...goal?.role !== void 0 ? { role: projectRoleForVocabulary(goal.role, vocabulary4) } : {},
|
|
207782
208275
|
...goal?.severity === "warn" ? { severity: "warn" } : {},
|
|
207783
208276
|
// Stored groundedness, when present. Not resolved over defaults —
|
|
207784
208277
|
// C1 registers no configurable groundedness defaults.
|
|
@@ -208092,10 +208585,10 @@ function parseCreateCasesBatchBody(c, raw) {
|
|
|
208092
208585
|
};
|
|
208093
208586
|
}
|
|
208094
208587
|
function caseResource(c, doc, status = 200) {
|
|
208095
|
-
const
|
|
208588
|
+
const vocabulary4 = vocabularyOf(c);
|
|
208096
208589
|
return v1Resource(
|
|
208097
208590
|
c,
|
|
208098
|
-
projectCaseDto(toCaseDto(doc,
|
|
208591
|
+
projectCaseDto(toCaseDto(doc, vocabulary4), vocabulary4),
|
|
208099
208592
|
status
|
|
208100
208593
|
);
|
|
208101
208594
|
}
|
|
@@ -208363,7 +208856,7 @@ var generateCasesSchema = z35.object({
|
|
|
208363
208856
|
message: "environmentId and servers are mutually exclusive \u2014 an environment supplies its own closed server set."
|
|
208364
208857
|
});
|
|
208365
208858
|
function buildCaseMutationArgs(body, opts) {
|
|
208366
|
-
const
|
|
208859
|
+
const vocabulary4 = opts.vocabulary ?? 1;
|
|
208367
208860
|
const args = {};
|
|
208368
208861
|
let isModelFreeStepsCase = false;
|
|
208369
208862
|
if ("id" in body && body.id !== void 0) args.caseId = body.id;
|
|
@@ -208382,7 +208875,7 @@ function buildCaseMutationArgs(body, opts) {
|
|
|
208382
208875
|
if (body.steps !== void 0) {
|
|
208383
208876
|
const steps = normalizeStepRolesForVocabulary(
|
|
208384
208877
|
withImplicitRenderAssertForSingleToolCall(body.steps),
|
|
208385
|
-
|
|
208878
|
+
vocabulary4,
|
|
208386
208879
|
"steps"
|
|
208387
208880
|
);
|
|
208388
208881
|
args.steps = steps;
|
|
@@ -208430,7 +208923,7 @@ function buildCaseMutationArgs(body, opts) {
|
|
|
208430
208923
|
// not mint two signatures.
|
|
208431
208924
|
list: normalizeCheckRolesForVocabulary(
|
|
208432
208925
|
body.checks.list,
|
|
208433
|
-
|
|
208926
|
+
vocabulary4,
|
|
208434
208927
|
"checks.list"
|
|
208435
208928
|
)
|
|
208436
208929
|
};
|
|
@@ -211109,7 +211602,7 @@ evals.get(
|
|
|
211109
211602
|
return v1Resource(c, toDescriptionExperimentDto(raw));
|
|
211110
211603
|
}
|
|
211111
211604
|
);
|
|
211112
|
-
async function readSuiteDetail(convexAuthToken, projectId, suiteId,
|
|
211605
|
+
async function readSuiteDetail(convexAuthToken, projectId, suiteId, vocabulary4) {
|
|
211113
211606
|
const convex = createConvexReadClient(convexAuthToken);
|
|
211114
211607
|
let suite;
|
|
211115
211608
|
try {
|
|
@@ -211146,7 +211639,7 @@ async function readSuiteDetail(convexAuthToken, projectId, suiteId, vocabulary)
|
|
|
211146
211639
|
suite,
|
|
211147
211640
|
execConfig,
|
|
211148
211641
|
{ computerEnvironmentName },
|
|
211149
|
-
|
|
211642
|
+
vocabulary4
|
|
211150
211643
|
);
|
|
211151
211644
|
}
|
|
211152
211645
|
async function defaultCaseModels(convex, suiteId) {
|
|
@@ -211260,7 +211753,7 @@ function applyVerdictPolicySettings(suite, settings, updateArgs, names = countFi
|
|
|
211260
211753
|
evals.patch("/projects/:projectId/eval-suites/:suiteId", async (c) => {
|
|
211261
211754
|
const projectId = c.req.param("projectId");
|
|
211262
211755
|
const suiteId = evalIdParam(c, "suiteId", "Eval suite");
|
|
211263
|
-
const
|
|
211756
|
+
const vocabulary4 = vocabularyOf(c);
|
|
211264
211757
|
const body = parseUpdateSuiteBody(c, await readJsonObjectBody(c));
|
|
211265
211758
|
const token = await getConvexBearerForRequest(c);
|
|
211266
211759
|
const { convexClient: convexClient6 } = createConvexClients(token);
|
|
@@ -211317,7 +211810,7 @@ evals.patch("/projects/:projectId/eval-suites/:suiteId", async (c) => {
|
|
|
211317
211810
|
if (s.checks !== void 0)
|
|
211318
211811
|
updateArgs.defaultPredicates = s.checks === null ? null : normalizeCheckRolesForVocabulary(
|
|
211319
211812
|
s.checks,
|
|
211320
|
-
|
|
211813
|
+
vocabulary4,
|
|
211321
211814
|
"settings.checks"
|
|
211322
211815
|
);
|
|
211323
211816
|
if (s.judge !== void 0) {
|
|
@@ -211335,7 +211828,7 @@ evals.patch("/projects/:projectId/eval-suites/:suiteId", async (c) => {
|
|
|
211335
211828
|
if (s.judge.role !== void 0)
|
|
211336
211829
|
goalCompletion.role = normalizeJudgeRoleForVocabulary(
|
|
211337
211830
|
{ role: s.judge.role },
|
|
211338
|
-
|
|
211831
|
+
vocabulary4,
|
|
211339
211832
|
"settings.judge"
|
|
211340
211833
|
).role;
|
|
211341
211834
|
if (s.judge.severity !== void 0)
|
|
@@ -212411,13 +212904,13 @@ async function completeGeneratedAuthoringJob(c, convex, job, suiteId) {
|
|
|
212411
212904
|
const docs = reads.flatMap(
|
|
212412
212905
|
(read) => read.status === "fulfilled" && read.value ? [read.value] : []
|
|
212413
212906
|
);
|
|
212414
|
-
const
|
|
212907
|
+
const vocabulary4 = vocabularyOf(c);
|
|
212415
212908
|
return v1Resource(c, {
|
|
212416
212909
|
jobId: job.jobId,
|
|
212417
212910
|
status: job.status,
|
|
212418
212911
|
generationModel: "anthropic/claude-haiku-4.5",
|
|
212419
212912
|
created: docs.map(
|
|
212420
|
-
(doc) => projectCaseDto(toCaseDto(doc,
|
|
212913
|
+
(doc) => projectCaseDto(toCaseDto(doc, vocabulary4), vocabulary4)
|
|
212421
212914
|
),
|
|
212422
212915
|
counts: {
|
|
212423
212916
|
normal: docs.filter((doc) => !doc.isNegativeTest).length,
|
|
@@ -229679,6 +230172,148 @@ var scenarios_default = scenarios;
|
|
|
229679
230172
|
import { Hono as Hono47 } from "hono";
|
|
229680
230173
|
import { z as z45 } from "zod";
|
|
229681
230174
|
|
|
230175
|
+
// shared/error-page.ts
|
|
230176
|
+
var HTML_PREAMBLE = /^(?:|\s|<!--[\s\S]*?-->|<\?xml[\s\S]*?\?>)+/i;
|
|
230177
|
+
var MARKUP_OPENER = /^<(?:!doctype\s+html|html|head|body|title)\b/i;
|
|
230178
|
+
var DOCTYPE_MARKER = /<!doctype\s+html/i;
|
|
230179
|
+
function looksLikeErrorPage(trimmed) {
|
|
230180
|
+
if (DOCTYPE_MARKER.test(trimmed)) return true;
|
|
230181
|
+
if (/<\/html>\s*$/i.test(trimmed)) return true;
|
|
230182
|
+
return MARKUP_OPENER.test(trimmed.replace(HTML_PREAMBLE, ""));
|
|
230183
|
+
}
|
|
230184
|
+
|
|
230185
|
+
// shared/swarm-attempt-error.ts
|
|
230186
|
+
var AGENT_ERROR_ENVELOPE = /^(?:swarm-agent\s+\S+\s+failed\s+\((\d{3})\):|Backend stream error:\s*(\d{3}))\s*/i;
|
|
230187
|
+
var URL_PATTERN2 = /https?:\/\/\S+/g;
|
|
230188
|
+
function isTransientSpendRefusal(code, refusalReason) {
|
|
230189
|
+
return code === "user_rate_limit" && refusalReason === "holds_committed";
|
|
230190
|
+
}
|
|
230191
|
+
var MAX_ATTEMPT_ERROR_CHARS = 500;
|
|
230192
|
+
function parseJsonObject(text2) {
|
|
230193
|
+
const trimmed = text2.trim();
|
|
230194
|
+
if (!trimmed.startsWith("{")) return null;
|
|
230195
|
+
try {
|
|
230196
|
+
const parsed = JSON.parse(trimmed);
|
|
230197
|
+
return parsed && typeof parsed === "object" && !Array.isArray(parsed) ? parsed : null;
|
|
230198
|
+
} catch {
|
|
230199
|
+
return null;
|
|
230200
|
+
}
|
|
230201
|
+
}
|
|
230202
|
+
function str3(value) {
|
|
230203
|
+
return typeof value === "string" && value.trim() ? value.trim() : void 0;
|
|
230204
|
+
}
|
|
230205
|
+
function num2(value) {
|
|
230206
|
+
return typeof value === "number" && Number.isFinite(value) ? value : void 0;
|
|
230207
|
+
}
|
|
230208
|
+
function scrub(text2) {
|
|
230209
|
+
return text2.replace(URL_PATTERN2, "").replace(/\s+/g, " ").trim();
|
|
230210
|
+
}
|
|
230211
|
+
function compose(headline, details) {
|
|
230212
|
+
if (!details) return headline;
|
|
230213
|
+
if (headline.toLowerCase().includes(details.toLowerCase())) return headline;
|
|
230214
|
+
const separator = /[.!?]$/.test(headline) ? " " : ". ";
|
|
230215
|
+
return `${headline}${separator}${details}`;
|
|
230216
|
+
}
|
|
230217
|
+
var SANDBOX_ERROR_CODE_MESSAGES = {
|
|
230218
|
+
sandbox_unavailable: "This session needed an MCPJam cloud sandbox for its computer commands, but this inspector can't run cloud sandboxes \u2014 the session could not execute commands.",
|
|
230219
|
+
sandbox_at_capacity: "MCPJam cloud is at capacity right now \u2014 try the run again in a few minutes.",
|
|
230220
|
+
sandbox_error: "The cloud sandbox for this session hit an error while starting \u2014 try the run again."
|
|
230221
|
+
};
|
|
230222
|
+
var XAA_REASON_FALLBACK_MESSAGES = {
|
|
230223
|
+
[XaaConnectFailureReason.REAUTH_REQUIRED]: "Your sign-in expired before this server's enterprise access token could be issued \u2014 sign in again, then re-run.",
|
|
230224
|
+
[XaaConnectFailureReason.AUTHORIZATION_SERVER_UNKNOWN]: "MCPJam couldn't find the authorization server protecting this server \u2014 set its issuer in the server's auth settings.",
|
|
230225
|
+
[XaaConnectFailureReason.NOT_SUPPORTED_HERE]: "This server's enterprise authorization mode can't run on this deployment \u2014 use pre-registered credentials in the server's auth settings.",
|
|
230226
|
+
[XaaConnectFailureReason.AUTHORIZATION_REJECTED]: "The authorization server rejected MCPJam's access request \u2014 check the server's XAA client credentials and issuer in its auth settings.",
|
|
230227
|
+
[XaaConnectFailureReason.CONFIGURATION_INVALID]: "This server isn't fully configured for enterprise-managed authorization \u2014 finish its XAA settings, or set an explicit auth method.",
|
|
230228
|
+
[XaaConnectFailureReason.HANDSHAKE_FAILED]: "This server couldn't complete its enterprise authorization handshake \u2014 try again, and check its auth settings if it keeps failing."
|
|
230229
|
+
};
|
|
230230
|
+
function humanizeSwarmAttemptError(raw, errorCode2) {
|
|
230231
|
+
if (errorCode2 === "stale_runner") {
|
|
230232
|
+
return {
|
|
230233
|
+
code: errorCode2,
|
|
230234
|
+
message: "The runner stopped reporting progress, so this run was marked interrupted. Sessions may have run before the interruption; inspect their saved traces. The reason contact was lost was not recorded."
|
|
230235
|
+
};
|
|
230236
|
+
}
|
|
230237
|
+
const sandboxMessage = errorCode2 ? SANDBOX_ERROR_CODE_MESSAGES[errorCode2] : void 0;
|
|
230238
|
+
if (sandboxMessage && errorCode2) {
|
|
230239
|
+
return { message: sandboxMessage, code: errorCode2 };
|
|
230240
|
+
}
|
|
230241
|
+
const input = (raw ?? "").trim();
|
|
230242
|
+
if (errorCode2 === "spending_reservation_busy" || input.includes("streamSpendingReservations") && input.includes("changed while this mutation was being run")) {
|
|
230243
|
+
return {
|
|
230244
|
+
code: "spending_reservation_busy",
|
|
230245
|
+
message: "MCPJam could not reserve spending capacity because concurrent requests kept changing it. This is an internal execution failure. Retry this attempt."
|
|
230246
|
+
};
|
|
230247
|
+
}
|
|
230248
|
+
if (isXaaConnectFailureReason(errorCode2)) {
|
|
230249
|
+
return {
|
|
230250
|
+
message: (scrub(input) || XAA_REASON_FALLBACK_MESSAGES[errorCode2]).slice(
|
|
230251
|
+
0,
|
|
230252
|
+
MAX_ATTEMPT_ERROR_CHARS
|
|
230253
|
+
),
|
|
230254
|
+
code: errorCode2,
|
|
230255
|
+
...isRerunnableXaaFailure(errorCode2) ? { rerunnable: true } : {}
|
|
230256
|
+
};
|
|
230257
|
+
}
|
|
230258
|
+
if (!input) return { message: "The session failed for an unknown reason." };
|
|
230259
|
+
let body = input;
|
|
230260
|
+
let httpStatus2;
|
|
230261
|
+
const envelope = AGENT_ERROR_ENVELOPE.exec(input);
|
|
230262
|
+
if (envelope) {
|
|
230263
|
+
httpStatus2 = Number(envelope[1] ?? envelope[2]);
|
|
230264
|
+
body = input.slice(envelope[0].length).trim();
|
|
230265
|
+
}
|
|
230266
|
+
if (looksLikeErrorPage(body)) {
|
|
230267
|
+
return {
|
|
230268
|
+
code: "upstream_error_page",
|
|
230269
|
+
message: `The request was answered with an HTML error page instead of a response${httpStatus2 ? ` (HTTP ${httpStatus2})` : ""}. This usually means a proxy or CDN blocked it.`,
|
|
230270
|
+
...httpStatus2 !== void 0 ? { httpStatus: httpStatus2 } : {}
|
|
230271
|
+
};
|
|
230272
|
+
}
|
|
230273
|
+
const parsed = parseJsonObject(body);
|
|
230274
|
+
if (!parsed) {
|
|
230275
|
+
const cleaned = scrub(body) || scrub(input);
|
|
230276
|
+
return {
|
|
230277
|
+
message: (cleaned || "The session failed for an unknown reason.").slice(
|
|
230278
|
+
0,
|
|
230279
|
+
MAX_ATTEMPT_ERROR_CHARS
|
|
230280
|
+
),
|
|
230281
|
+
...httpStatus2 !== void 0 ? { httpStatus: httpStatus2 } : {}
|
|
230282
|
+
};
|
|
230283
|
+
}
|
|
230284
|
+
const headline = str3(parsed.error) ?? str3(parsed.message) ?? "The session could not run.";
|
|
230285
|
+
const details = str3(parsed.details);
|
|
230286
|
+
const code = str3(parsed.code);
|
|
230287
|
+
const retryAfterMs = num2(parsed.retryAfter);
|
|
230288
|
+
const canTopUp = parsed.canTopUp === true;
|
|
230289
|
+
return {
|
|
230290
|
+
message: scrub(compose(headline, details)).slice(
|
|
230291
|
+
0,
|
|
230292
|
+
MAX_ATTEMPT_ERROR_CHARS
|
|
230293
|
+
),
|
|
230294
|
+
...code ? { code } : {},
|
|
230295
|
+
...str3(parsed.refusalReason) ? { refusalReason: str3(parsed.refusalReason) } : {},
|
|
230296
|
+
...typeof parsed.isRetryable === "boolean" ? { isRetryable: parsed.isRetryable } : {},
|
|
230297
|
+
...num2(parsed.outstandingHolds) !== void 0 ? { outstandingHolds: num2(parsed.outstandingHolds) } : {},
|
|
230298
|
+
...retryAfterMs !== void 0 ? { retryAfterMs } : {},
|
|
230299
|
+
...canTopUp ? { canTopUp } : {},
|
|
230300
|
+
...httpStatus2 !== void 0 ? { httpStatus: httpStatus2 } : {}
|
|
230301
|
+
};
|
|
230302
|
+
}
|
|
230303
|
+
function humanizeSwarmAttemptErrorMessage(raw) {
|
|
230304
|
+
return humanizeSwarmAttemptError(raw).message;
|
|
230305
|
+
}
|
|
230306
|
+
var ACCOUNT_LIMIT_CODE = /\b(?:user_rate_limit|org_rate_limit|mcpjam_rate_limit|billing_limit_reached|spend_budget_reached|wallet_locked|billing_feature_not_included|free_tier_model_restricted|spend_cap_exceeded|platform_free_budget_exhausted|account_suspended|guest_model_not_allowed|guest_input_too_large)\b/i;
|
|
230307
|
+
function isAccountLimit(message, code) {
|
|
230308
|
+
if (accountLimitCode(message, code)) return true;
|
|
230309
|
+
return !!message && MCPJAM_MODEL_LIMIT_SENTENCE.test(message);
|
|
230310
|
+
}
|
|
230311
|
+
var MCPJAM_MODEL_LIMIT_SENTENCE = /\b(?:Daily|Monthly) MCPJam model limit reached\b/i;
|
|
230312
|
+
function accountLimitCode(message, code) {
|
|
230313
|
+
const match = (code ? ACCOUNT_LIMIT_CODE.exec(code) : null) ?? (message ? ACCOUNT_LIMIT_CODE.exec(message) : null);
|
|
230314
|
+
return match?.[0].toLowerCase();
|
|
230315
|
+
}
|
|
230316
|
+
|
|
229682
230317
|
// shared/swarm-grounding.ts
|
|
229683
230318
|
var GROUNDING_LIMITS = {
|
|
229684
230319
|
tools: 12,
|
|
@@ -229935,6 +230570,53 @@ async function probeReadOnlyTools(args) {
|
|
|
229935
230570
|
}
|
|
229936
230571
|
}
|
|
229937
230572
|
|
|
230573
|
+
// server/services/sessionSimulation/admission-retry.ts
|
|
230574
|
+
function spendRefusalOf(error) {
|
|
230575
|
+
if (!error || typeof error !== "object") return void 0;
|
|
230576
|
+
if ("refusal" in error) return error.refusal;
|
|
230577
|
+
if (error instanceof Error && error.name === "RecordedAssistantTurnError")
|
|
230578
|
+
return void 0;
|
|
230579
|
+
if (!(error instanceof Error)) return void 0;
|
|
230580
|
+
const info = humanizeSwarmAttemptError(error.message);
|
|
230581
|
+
return info.refusalReason ? info : void 0;
|
|
230582
|
+
}
|
|
230583
|
+
var AdmissionWaitBudget = class {
|
|
230584
|
+
constructor(totalMs = 5 * 6e4, maxAttemptsPerCall = 8) {
|
|
230585
|
+
this.maxAttemptsPerCall = maxAttemptsPerCall;
|
|
230586
|
+
this.remainingMs = totalMs;
|
|
230587
|
+
}
|
|
230588
|
+
take(delayMs) {
|
|
230589
|
+
if (delayMs > this.remainingMs) return false;
|
|
230590
|
+
this.remainingMs -= delayMs;
|
|
230591
|
+
return true;
|
|
230592
|
+
}
|
|
230593
|
+
};
|
|
230594
|
+
async function withAdmissionRetry(op, options) {
|
|
230595
|
+
let attempt = 0;
|
|
230596
|
+
for (; ; ) {
|
|
230597
|
+
options.signal?.throwIfAborted();
|
|
230598
|
+
try {
|
|
230599
|
+
return await op();
|
|
230600
|
+
} catch (error) {
|
|
230601
|
+
const refusal = spendRefusalOf(error);
|
|
230602
|
+
if (!refusal || !isTransientSpendRefusal(refusal.code, refusal.refusalReason) || ++attempt >= options.budget.maxAttemptsPerCall)
|
|
230603
|
+
throw error;
|
|
230604
|
+
const base = Number.isFinite(refusal.retryAfterMs) ? refusal.retryAfterMs : 15e3;
|
|
230605
|
+
const delay5 = Math.round(
|
|
230606
|
+
defaultJitter(clampDelay(base * attempt, 5e3, 6e4))
|
|
230607
|
+
);
|
|
230608
|
+
if (!options.budget.take(delay5)) throw error;
|
|
230609
|
+
const resume = options.onWait?.(delay5);
|
|
230610
|
+
try {
|
|
230611
|
+
await abortableSleep(delay5, options.signal);
|
|
230612
|
+
} finally {
|
|
230613
|
+
resume?.();
|
|
230614
|
+
}
|
|
230615
|
+
options.signal?.throwIfAborted();
|
|
230616
|
+
}
|
|
230617
|
+
}
|
|
230618
|
+
}
|
|
230619
|
+
|
|
229938
230620
|
// server/services/sessionSimulation/runner.ts
|
|
229939
230621
|
import { ConvexHttpClient as ConvexHttpClient12 } from "convex/browser";
|
|
229940
230622
|
|
|
@@ -230061,6 +230743,7 @@ function extractToolResults2(messages) {
|
|
|
230061
230743
|
|
|
230062
230744
|
// server/utils/turn-failure-classification.ts
|
|
230063
230745
|
function classifyTurnFailure(message) {
|
|
230746
|
+
if (/\bprovider_empty_response\b/.test(message)) return "failed";
|
|
230064
230747
|
return /rate.?limit|too many requests|(?:^|[^\w.:])429\b|\bspend\b|spend_budget_reached|\bquota\b|\bbudget\b|\bcap\b/i.test(
|
|
230065
230748
|
message
|
|
230066
230749
|
) ? "rate_limited" : "failed";
|
|
@@ -230297,6 +230980,21 @@ async function runSyntheticHostSession(adapter) {
|
|
|
230297
230980
|
);
|
|
230298
230981
|
const sessionSignal = sessionDeadline.signal;
|
|
230299
230982
|
let turnDeadline;
|
|
230983
|
+
const admissionBudget = new AdmissionWaitBudget();
|
|
230984
|
+
const admissionOptions = {
|
|
230985
|
+
budget: admissionBudget,
|
|
230986
|
+
signal: sessionSignal,
|
|
230987
|
+
onWait: () => {
|
|
230988
|
+
turnDeadline?.dispose();
|
|
230989
|
+
return () => {
|
|
230990
|
+
turnDeadline = withDeadline3(
|
|
230991
|
+
sessionSignal,
|
|
230992
|
+
budgets.turnTimeoutMs,
|
|
230993
|
+
"turn"
|
|
230994
|
+
);
|
|
230995
|
+
};
|
|
230996
|
+
}
|
|
230997
|
+
};
|
|
230300
230998
|
let manager;
|
|
230301
230999
|
let dispose;
|
|
230302
231000
|
let browser;
|
|
@@ -230583,7 +231281,10 @@ async function runSyntheticHostSession(adapter) {
|
|
|
230583
231281
|
}
|
|
230584
231282
|
turnDeadline?.dispose();
|
|
230585
231283
|
turnDeadline = withDeadline3(sessionSignal, budgets.turnTimeoutMs, "turn");
|
|
230586
|
-
const next = await
|
|
231284
|
+
const next = await withAdmissionRetry(
|
|
231285
|
+
() => nextPersonaTurn(lastTranscript),
|
|
231286
|
+
admissionOptions
|
|
231287
|
+
);
|
|
230587
231288
|
if (next.endSession) {
|
|
230588
231289
|
if (turn === 0) {
|
|
230589
231290
|
throw Object.assign(
|
|
@@ -230622,130 +231323,133 @@ async function runSyntheticHostSession(adapter) {
|
|
|
230622
231323
|
turnTrace,
|
|
230623
231324
|
modelSource: turnModelSource,
|
|
230624
231325
|
harnessSessionCommit
|
|
230625
|
-
} = await
|
|
230626
|
-
|
|
230627
|
-
|
|
230628
|
-
|
|
230629
|
-
|
|
230630
|
-
|
|
230631
|
-
|
|
230632
|
-
|
|
230633
|
-
|
|
230634
|
-
|
|
230635
|
-
|
|
230636
|
-
|
|
230637
|
-
|
|
230638
|
-
|
|
230639
|
-
|
|
230640
|
-
|
|
230641
|
-
|
|
230642
|
-
|
|
230643
|
-
|
|
230644
|
-
|
|
230645
|
-
|
|
230646
|
-
|
|
230647
|
-
|
|
230648
|
-
|
|
230649
|
-
|
|
230650
|
-
|
|
230651
|
-
|
|
230652
|
-
|
|
230653
|
-
onToolResult: (event) => {
|
|
230654
|
-
void browser.handleEngineToolResult(event);
|
|
230655
|
-
emit?.({
|
|
230656
|
-
type: "tool_result",
|
|
230657
|
-
toolCallId: event.toolCallId,
|
|
230658
|
-
result: event.output
|
|
230659
|
-
});
|
|
230660
|
-
},
|
|
230661
|
-
...browser.prepareAdvertisedTools ? { prepareAdvertisedTools: browser.prepareAdvertisedTools } : {},
|
|
230662
|
-
onToolResultChunk: async (chunk) => {
|
|
230663
|
-
await browser.handleDirectToolResultChunk(chunk);
|
|
230664
|
-
emit?.({
|
|
230665
|
-
type: "tool_result",
|
|
230666
|
-
toolCallId: chunk.toolCallId,
|
|
230667
|
-
result: chunk.output
|
|
230668
|
-
});
|
|
230669
|
-
},
|
|
230670
|
-
onToolCallChunk: (chunk) => {
|
|
230671
|
-
emit?.({
|
|
230672
|
-
type: "tool_call",
|
|
230673
|
-
toolName: chunk.toolName,
|
|
230674
|
-
toolCallId: chunk.toolCallId,
|
|
230675
|
-
args: chunk.input
|
|
230676
|
-
});
|
|
230677
|
-
},
|
|
230678
|
-
...emit ? {
|
|
230679
|
-
onLiveTextDelta: (content) => {
|
|
230680
|
-
emit({ type: "text_delta", content });
|
|
231326
|
+
} = await withAdmissionRetry(
|
|
231327
|
+
() => drainAssistantTurn({
|
|
231328
|
+
messages: messageHistory,
|
|
231329
|
+
modelId: String(modelDefinition.id),
|
|
231330
|
+
modelDefinition,
|
|
231331
|
+
chatSessionId,
|
|
231332
|
+
// Tag the engine-facing turn (usage rows) with THIS surface's source:
|
|
231333
|
+
// "scenario" for the session-simulation surface, "swarm" for the
|
|
231334
|
+
// journey-execution runner. The persist attribution already carries
|
|
231335
|
+
// this; forwarding it keeps hosted + local-BYOK usage rows correctly
|
|
231336
|
+
// sourced instead of hardcoding every journey turn as "scenario".
|
|
231337
|
+
sourceType: persist.sourceType,
|
|
231338
|
+
systemPrompt: prepared.enhancedSystemPrompt,
|
|
231339
|
+
temperature: prepared.resolvedTemperature,
|
|
231340
|
+
...maxSteps !== void 0 ? { maxSteps } : {},
|
|
231341
|
+
// `computer` / `finish_widget` merge into the advertised set; the
|
|
231342
|
+
// prepareAdvertisedTools hook hides them until a widget is mounted.
|
|
231343
|
+
tools: { ...prepared.allTools, ...browser.computerWidgetTools },
|
|
231344
|
+
hooks: {
|
|
231345
|
+
onToolCall: (event) => {
|
|
231346
|
+
browser.noteToolCallInput(event);
|
|
231347
|
+
const args = event.input && typeof event.input === "object" && !Array.isArray(event.input) ? event.input : { value: event.input };
|
|
231348
|
+
emit?.({
|
|
231349
|
+
type: "tool_call",
|
|
231350
|
+
toolName: event.toolName,
|
|
231351
|
+
toolCallId: event.toolCallId,
|
|
231352
|
+
args
|
|
231353
|
+
});
|
|
230681
231354
|
},
|
|
230682
|
-
|
|
230683
|
-
|
|
230684
|
-
|
|
230685
|
-
|
|
230686
|
-
|
|
230687
|
-
|
|
230688
|
-
inputTokens: event.turnUsage.inputTokens ?? 0,
|
|
230689
|
-
outputTokens: event.turnUsage.outputTokens ?? 0
|
|
230690
|
-
}
|
|
230691
|
-
} : {}
|
|
231355
|
+
onToolResult: (event) => {
|
|
231356
|
+
void browser.handleEngineToolResult(event);
|
|
231357
|
+
emit?.({
|
|
231358
|
+
type: "tool_result",
|
|
231359
|
+
toolCallId: event.toolCallId,
|
|
231360
|
+
result: event.output
|
|
230692
231361
|
});
|
|
230693
|
-
}
|
|
230694
|
-
|
|
230695
|
-
|
|
230696
|
-
|
|
230697
|
-
|
|
230698
|
-
|
|
230699
|
-
|
|
230700
|
-
|
|
230701
|
-
|
|
230702
|
-
|
|
230703
|
-
|
|
230704
|
-
|
|
230705
|
-
|
|
230706
|
-
|
|
230707
|
-
|
|
230708
|
-
|
|
230709
|
-
|
|
230710
|
-
|
|
230711
|
-
|
|
230712
|
-
|
|
230713
|
-
|
|
230714
|
-
|
|
230715
|
-
|
|
230716
|
-
|
|
230717
|
-
|
|
230718
|
-
|
|
230719
|
-
|
|
230720
|
-
|
|
230721
|
-
|
|
230722
|
-
|
|
230723
|
-
|
|
230724
|
-
|
|
230725
|
-
|
|
230726
|
-
|
|
230727
|
-
|
|
230728
|
-
|
|
230729
|
-
|
|
230730
|
-
|
|
230731
|
-
|
|
230732
|
-
|
|
230733
|
-
|
|
230734
|
-
|
|
230735
|
-
|
|
230736
|
-
|
|
230737
|
-
|
|
230738
|
-
|
|
230739
|
-
|
|
230740
|
-
|
|
230741
|
-
|
|
230742
|
-
|
|
230743
|
-
|
|
230744
|
-
|
|
230745
|
-
|
|
230746
|
-
|
|
230747
|
-
|
|
230748
|
-
|
|
231362
|
+
},
|
|
231363
|
+
...browser.prepareAdvertisedTools ? { prepareAdvertisedTools: browser.prepareAdvertisedTools } : {},
|
|
231364
|
+
onToolResultChunk: async (chunk) => {
|
|
231365
|
+
await browser.handleDirectToolResultChunk(chunk);
|
|
231366
|
+
emit?.({
|
|
231367
|
+
type: "tool_result",
|
|
231368
|
+
toolCallId: chunk.toolCallId,
|
|
231369
|
+
result: chunk.output
|
|
231370
|
+
});
|
|
231371
|
+
},
|
|
231372
|
+
onToolCallChunk: (chunk) => {
|
|
231373
|
+
emit?.({
|
|
231374
|
+
type: "tool_call",
|
|
231375
|
+
toolName: chunk.toolName,
|
|
231376
|
+
toolCallId: chunk.toolCallId,
|
|
231377
|
+
args: chunk.input
|
|
231378
|
+
});
|
|
231379
|
+
},
|
|
231380
|
+
...emit ? {
|
|
231381
|
+
onLiveTextDelta: (content) => {
|
|
231382
|
+
emit({ type: "text_delta", content });
|
|
231383
|
+
},
|
|
231384
|
+
onStepFinish: (event) => {
|
|
231385
|
+
emit({
|
|
231386
|
+
type: "step_finish",
|
|
231387
|
+
stepNumber: event.stepIndex,
|
|
231388
|
+
...event.turnUsage ? {
|
|
231389
|
+
usage: {
|
|
231390
|
+
inputTokens: event.turnUsage.inputTokens ?? 0,
|
|
231391
|
+
outputTokens: event.turnUsage.outputTokens ?? 0
|
|
231392
|
+
}
|
|
231393
|
+
} : {}
|
|
231394
|
+
});
|
|
231395
|
+
}
|
|
231396
|
+
} : {}
|
|
231397
|
+
},
|
|
231398
|
+
progressivePlan: prepared.progressivePlan,
|
|
231399
|
+
discoveryState: prepared.discoveryState,
|
|
231400
|
+
mcpClientManager: manager,
|
|
231401
|
+
selectedServers: selectedServerIds,
|
|
231402
|
+
requireToolApproval,
|
|
231403
|
+
...harness2 ? { harness: harness2 } : {},
|
|
231404
|
+
// Harness MCP-proxy plane (harness hosts with MCP servers) + swarm
|
|
231405
|
+
// continuity identity (`swarm-chat` owner lane). `harnessMcpProxy` is
|
|
231406
|
+
// resolved once above; `journeyRunId`/`hostId` are the swarm run + pinned
|
|
231407
|
+
// host. All three are inert for the emulated engine / non-swarm surfaces.
|
|
231408
|
+
...harnessMcpProxy ? { harnessMcpProxy } : {},
|
|
231409
|
+
// Pinned harness skills (env-based swarm target running a real
|
|
231410
|
+
// harness): the harness turn skips the live skills fetch and delivers
|
|
231411
|
+
// exactly these artifacts (skillsHash derives from their fingerprints).
|
|
231412
|
+
// Passed even when EMPTY — an empty authoritative set means the
|
|
231413
|
+
// harness must run skill-less, not fall back to the live pool.
|
|
231414
|
+
...harness2 && pinnedSkills !== void 0 ? { pinnedHarnessSkills: pinnedSkills } : {},
|
|
231415
|
+
// The attempt's own disposable box for the HARNESS turn. Only meaningful
|
|
231416
|
+
// when a harness is selected — the emulated engine's shell binds through
|
|
231417
|
+
// `resolveHostTools` above instead.
|
|
231418
|
+
...harness2 && harnessSandboxBinding ? { harnessSandboxBinding } : {},
|
|
231419
|
+
// The target's Project Environment — the GRANT BOUNDARY the harness
|
|
231420
|
+
// turn checks a BROKERED external-account credential against. Harness
|
|
231421
|
+
// only: the emulated engine resolves no such credential, and this
|
|
231422
|
+
// runner delivers no materialized secrets on either path (see the
|
|
231423
|
+
// `runtimeSecrets` contract on `MCPJamHandlerOptions`), so brokered
|
|
231424
|
+
// delivery is the only one a swarm attempt can use.
|
|
231425
|
+
...harness2 && environmentId ? { environmentId } : {},
|
|
231426
|
+
// Server-executed built-ins (`web_search`, …) for the HARNESS turn.
|
|
231427
|
+
// The emulated engine already receives them merged into `tools` via
|
|
231428
|
+
// prepareChatV2's `allTools`; the harness reads them off this separate
|
|
231429
|
+
// option instead, because it hands them to the runtime as specs and
|
|
231430
|
+
// executes them here, while MCP-server tools go via `.mcp.json`. Only
|
|
231431
|
+
// for a harness target — passing them on the emulated path would
|
|
231432
|
+
// duplicate what `allTools` already carries.
|
|
231433
|
+
...harness2 && builtInTools2 && Object.keys(builtInTools2).length > 0 ? { builtInTools: builtInTools2 } : {},
|
|
231434
|
+
...persist.hostId ? { hostId: persist.hostId } : {},
|
|
231435
|
+
// Scenario surface only. The scenario runtime-config redeem returns an
|
|
231436
|
+
// accessVersion that /stream/org/resolve uses to authorize the actor
|
|
231437
|
+
// against the versioned scenario; threading it (instead of undefined)
|
|
231438
|
+
// matches what real-visitor synthetic-equivalent chats send. The swarm
|
|
231439
|
+
// surface authorizes via project membership and leaves both undefined.
|
|
231440
|
+
...scenarioId ? { scenarioId } : {},
|
|
231441
|
+
accessVersion,
|
|
231442
|
+
projectId,
|
|
231443
|
+
authHeader,
|
|
231444
|
+
abortSignal: turnDeadline.signal,
|
|
231445
|
+
// Threaded into the per-step /stream (or /stream/org) body and the
|
|
231446
|
+
// /stream/org/local-usage writeback so the backend BYOK and JAM-paid
|
|
231447
|
+
// writers can stamp the run id onto llmUsageRecord for per-run spend
|
|
231448
|
+
// attribution.
|
|
231449
|
+
...persist.journeyRunId ? { journeyRunId: persist.journeyRunId } : {}
|
|
231450
|
+
}),
|
|
231451
|
+
admissionOptions
|
|
231452
|
+
).catch((error) => {
|
|
230749
231453
|
if (!(error instanceof RecordedAssistantTurnError)) throw error;
|
|
230750
231454
|
failedTurn = error;
|
|
230751
231455
|
return error.turn;
|
|
@@ -230896,13 +231600,14 @@ async function runSyntheticHostSession(adapter) {
|
|
|
230896
231600
|
};
|
|
230897
231601
|
}
|
|
230898
231602
|
const message = error instanceof Error ? error.message : String(error);
|
|
231603
|
+
const errorRefusal = error instanceof RecordedAssistantTurnError ? error.errorRefusal : spendRefusalOf(error);
|
|
230899
231604
|
if (classifyTurnFailure(message) === "rate_limited") {
|
|
230900
231605
|
emit?.({
|
|
230901
231606
|
type: "session_complete",
|
|
230902
231607
|
status: "rate_limited",
|
|
230903
231608
|
errorMessage: message
|
|
230904
231609
|
});
|
|
230905
|
-
return { outcome: "rate_limited", errorMessage: message };
|
|
231610
|
+
return { outcome: "rate_limited", errorMessage: message, errorRefusal };
|
|
230906
231611
|
}
|
|
230907
231612
|
logger.warn("[sessionSimulation.runner] session failed", {
|
|
230908
231613
|
runId,
|
|
@@ -230919,7 +231624,8 @@ async function runSyntheticHostSession(adapter) {
|
|
|
230919
231624
|
return {
|
|
230920
231625
|
outcome: "failed",
|
|
230921
231626
|
errorMessage: message,
|
|
230922
|
-
...errorReason ? { errorReason } : {}
|
|
231627
|
+
...errorReason ? { errorReason } : {},
|
|
231628
|
+
errorRefusal
|
|
230923
231629
|
};
|
|
230924
231630
|
} finally {
|
|
230925
231631
|
turnDeadline?.dispose();
|
|
@@ -231254,12 +231960,24 @@ async function drainAssistantTurn(args) {
|
|
|
231254
231960
|
newMessageCount: result.newMessages.length
|
|
231255
231961
|
});
|
|
231256
231962
|
if (turnFailure && !args.abortSignal?.aborted) {
|
|
231257
|
-
const failed3 = (message) =>
|
|
231258
|
-
|
|
231259
|
-
|
|
231260
|
-
|
|
231261
|
-
|
|
231262
|
-
|
|
231963
|
+
const failed3 = (message) => {
|
|
231964
|
+
const error = new RecordedAssistantTurnError(message, {
|
|
231965
|
+
history: result.messages,
|
|
231966
|
+
turnTrace: result.turnTrace,
|
|
231967
|
+
modelSource: rt.modelSource,
|
|
231968
|
+
...result.harnessSessionCommit ? { harnessSessionCommit: result.harnessSessionCommit } : {}
|
|
231969
|
+
});
|
|
231970
|
+
error.errorRefusal = lastEngineError ? {
|
|
231971
|
+
code: lastEngineError.code,
|
|
231972
|
+
refusalReason: lastEngineError.refusalReason,
|
|
231973
|
+
retryAfterMs: lastEngineError.retryAfterMs,
|
|
231974
|
+
httpStatus: lastEngineError.httpStatus,
|
|
231975
|
+
stepIndex: lastEngineError.stepIndex,
|
|
231976
|
+
outstandingHolds: lastEngineError.outstandingHolds
|
|
231977
|
+
} : void 0;
|
|
231978
|
+
if (result.newMessages.length === 0) error.refusal = error.errorRefusal;
|
|
231979
|
+
return error;
|
|
231980
|
+
};
|
|
231263
231981
|
if (lastEngineError) {
|
|
231264
231982
|
const detail = [
|
|
231265
231983
|
lastEngineError.code,
|
|
@@ -232457,125 +233175,6 @@ function swarmAttemptChatSessionId(runId, target, sessionIndex) {
|
|
|
232457
233175
|
return `synth_${runId}_${target.hostId}_${sessionIndex}`;
|
|
232458
233176
|
}
|
|
232459
233177
|
|
|
232460
|
-
// shared/swarm-attempt-error.ts
|
|
232461
|
-
var AGENT_ERROR_ENVELOPE = /^(?:swarm-agent\s+\S+\s+failed\s+\((\d{3})\):|Backend stream error:\s*(\d{3}))\s*/i;
|
|
232462
|
-
var URL_PATTERN2 = /https?:\/\/\S+/g;
|
|
232463
|
-
var MAX_ATTEMPT_ERROR_CHARS = 500;
|
|
232464
|
-
function parseJsonObject(text2) {
|
|
232465
|
-
const trimmed = text2.trim();
|
|
232466
|
-
if (!trimmed.startsWith("{")) return null;
|
|
232467
|
-
try {
|
|
232468
|
-
const parsed = JSON.parse(trimmed);
|
|
232469
|
-
return parsed && typeof parsed === "object" && !Array.isArray(parsed) ? parsed : null;
|
|
232470
|
-
} catch {
|
|
232471
|
-
return null;
|
|
232472
|
-
}
|
|
232473
|
-
}
|
|
232474
|
-
function str3(value) {
|
|
232475
|
-
return typeof value === "string" && value.trim() ? value.trim() : void 0;
|
|
232476
|
-
}
|
|
232477
|
-
function num2(value) {
|
|
232478
|
-
return typeof value === "number" && Number.isFinite(value) ? value : void 0;
|
|
232479
|
-
}
|
|
232480
|
-
function scrub(text2) {
|
|
232481
|
-
return text2.replace(URL_PATTERN2, "").replace(/\s+/g, " ").trim();
|
|
232482
|
-
}
|
|
232483
|
-
function compose(headline, details) {
|
|
232484
|
-
if (!details) return headline;
|
|
232485
|
-
if (headline.toLowerCase().includes(details.toLowerCase())) return headline;
|
|
232486
|
-
const separator = /[.!?]$/.test(headline) ? " " : ". ";
|
|
232487
|
-
return `${headline}${separator}${details}`;
|
|
232488
|
-
}
|
|
232489
|
-
var SANDBOX_ERROR_CODE_MESSAGES = {
|
|
232490
|
-
sandbox_unavailable: "This session needed an MCPJam cloud sandbox for its computer commands, but this inspector can't run cloud sandboxes \u2014 the session could not execute commands.",
|
|
232491
|
-
sandbox_at_capacity: "MCPJam cloud is at capacity right now \u2014 try the run again in a few minutes.",
|
|
232492
|
-
sandbox_error: "The cloud sandbox for this session hit an error while starting \u2014 try the run again."
|
|
232493
|
-
};
|
|
232494
|
-
var XAA_REASON_FALLBACK_MESSAGES = {
|
|
232495
|
-
[XaaConnectFailureReason.REAUTH_REQUIRED]: "Your sign-in expired before this server's enterprise access token could be issued \u2014 sign in again, then re-run.",
|
|
232496
|
-
[XaaConnectFailureReason.AUTHORIZATION_SERVER_UNKNOWN]: "MCPJam couldn't find the authorization server protecting this server \u2014 set its issuer in the server's auth settings.",
|
|
232497
|
-
[XaaConnectFailureReason.NOT_SUPPORTED_HERE]: "This server's enterprise authorization mode can't run on this deployment \u2014 use pre-registered credentials in the server's auth settings.",
|
|
232498
|
-
[XaaConnectFailureReason.AUTHORIZATION_REJECTED]: "The authorization server rejected MCPJam's access request \u2014 check the server's XAA client credentials and issuer in its auth settings.",
|
|
232499
|
-
[XaaConnectFailureReason.CONFIGURATION_INVALID]: "This server isn't fully configured for enterprise-managed authorization \u2014 finish its XAA settings, or set an explicit auth method.",
|
|
232500
|
-
[XaaConnectFailureReason.HANDSHAKE_FAILED]: "This server couldn't complete its enterprise authorization handshake \u2014 try again, and check its auth settings if it keeps failing."
|
|
232501
|
-
};
|
|
232502
|
-
function humanizeSwarmAttemptError(raw, errorCode2) {
|
|
232503
|
-
if (errorCode2 === "stale_runner") {
|
|
232504
|
-
return {
|
|
232505
|
-
code: errorCode2,
|
|
232506
|
-
message: "The runner stopped reporting progress, so this run was marked interrupted. Sessions may have run before the interruption; inspect their saved traces. The reason contact was lost was not recorded."
|
|
232507
|
-
};
|
|
232508
|
-
}
|
|
232509
|
-
const sandboxMessage = errorCode2 ? SANDBOX_ERROR_CODE_MESSAGES[errorCode2] : void 0;
|
|
232510
|
-
if (sandboxMessage && errorCode2) {
|
|
232511
|
-
return { message: sandboxMessage, code: errorCode2 };
|
|
232512
|
-
}
|
|
232513
|
-
const input = (raw ?? "").trim();
|
|
232514
|
-
if (errorCode2 === "spending_reservation_busy" || input.includes("streamSpendingReservations") && input.includes("changed while this mutation was being run")) {
|
|
232515
|
-
return {
|
|
232516
|
-
code: "spending_reservation_busy",
|
|
232517
|
-
message: "MCPJam could not reserve spending capacity because concurrent requests kept changing it. This is an internal execution failure. Retry this attempt."
|
|
232518
|
-
};
|
|
232519
|
-
}
|
|
232520
|
-
if (isXaaConnectFailureReason(errorCode2)) {
|
|
232521
|
-
return {
|
|
232522
|
-
message: (scrub(input) || XAA_REASON_FALLBACK_MESSAGES[errorCode2]).slice(
|
|
232523
|
-
0,
|
|
232524
|
-
MAX_ATTEMPT_ERROR_CHARS
|
|
232525
|
-
),
|
|
232526
|
-
code: errorCode2,
|
|
232527
|
-
...isRerunnableXaaFailure(errorCode2) ? { rerunnable: true } : {}
|
|
232528
|
-
};
|
|
232529
|
-
}
|
|
232530
|
-
if (!input) return { message: "The session failed for an unknown reason." };
|
|
232531
|
-
let body = input;
|
|
232532
|
-
let httpStatus2;
|
|
232533
|
-
const envelope = AGENT_ERROR_ENVELOPE.exec(input);
|
|
232534
|
-
if (envelope) {
|
|
232535
|
-
httpStatus2 = Number(envelope[1] ?? envelope[2]);
|
|
232536
|
-
body = input.slice(envelope[0].length).trim();
|
|
232537
|
-
}
|
|
232538
|
-
const parsed = parseJsonObject(body);
|
|
232539
|
-
if (!parsed) {
|
|
232540
|
-
const cleaned = scrub(body) || scrub(input);
|
|
232541
|
-
return {
|
|
232542
|
-
message: (cleaned || "The session failed for an unknown reason.").slice(
|
|
232543
|
-
0,
|
|
232544
|
-
MAX_ATTEMPT_ERROR_CHARS
|
|
232545
|
-
),
|
|
232546
|
-
...httpStatus2 !== void 0 ? { httpStatus: httpStatus2 } : {}
|
|
232547
|
-
};
|
|
232548
|
-
}
|
|
232549
|
-
const headline = str3(parsed.error) ?? str3(parsed.message) ?? "The session could not run.";
|
|
232550
|
-
const details = str3(parsed.details);
|
|
232551
|
-
const code = str3(parsed.code);
|
|
232552
|
-
const retryAfterMs = num2(parsed.retryAfter);
|
|
232553
|
-
const canTopUp = parsed.canTopUp === true;
|
|
232554
|
-
return {
|
|
232555
|
-
message: scrub(compose(headline, details)).slice(
|
|
232556
|
-
0,
|
|
232557
|
-
MAX_ATTEMPT_ERROR_CHARS
|
|
232558
|
-
),
|
|
232559
|
-
...code ? { code } : {},
|
|
232560
|
-
...retryAfterMs !== void 0 ? { retryAfterMs } : {},
|
|
232561
|
-
...canTopUp ? { canTopUp } : {},
|
|
232562
|
-
...httpStatus2 !== void 0 ? { httpStatus: httpStatus2 } : {}
|
|
232563
|
-
};
|
|
232564
|
-
}
|
|
232565
|
-
function humanizeSwarmAttemptErrorMessage(raw) {
|
|
232566
|
-
return humanizeSwarmAttemptError(raw).message;
|
|
232567
|
-
}
|
|
232568
|
-
var ACCOUNT_LIMIT_CODE = /\b(?:user_rate_limit|org_rate_limit|mcpjam_rate_limit|billing_limit_reached|spend_budget_reached|wallet_locked|billing_feature_not_included|free_tier_model_restricted|spend_cap_exceeded|platform_free_budget_exhausted|account_suspended|guest_model_not_allowed|guest_input_too_large)\b/i;
|
|
232569
|
-
function isAccountLimit(message, code) {
|
|
232570
|
-
if (accountLimitCode(message, code)) return true;
|
|
232571
|
-
return !!message && MCPJAM_MODEL_LIMIT_SENTENCE.test(message);
|
|
232572
|
-
}
|
|
232573
|
-
var MCPJAM_MODEL_LIMIT_SENTENCE = /\b(?:Daily|Monthly) MCPJam model limit reached\b/i;
|
|
232574
|
-
function accountLimitCode(message, code) {
|
|
232575
|
-
const match = (code ? ACCOUNT_LIMIT_CODE.exec(code) : null) ?? (message ? ACCOUNT_LIMIT_CODE.exec(message) : null);
|
|
232576
|
-
return match?.[0].toLowerCase();
|
|
232577
|
-
}
|
|
232578
|
-
|
|
232579
233178
|
// server/services/sessionSimulation/swarm-stream-hub.ts
|
|
232580
233179
|
var RING_BUFFER_SIZE = 200;
|
|
232581
233180
|
var MAX_RETAINED_FRAME_SESSIONS = 64;
|
|
@@ -232741,7 +233340,8 @@ function terminalForOutcome(outcome, errorMessage7, errorReason) {
|
|
|
232741
233340
|
if (outcome === "succeeded") {
|
|
232742
233341
|
return { status: "succeeded" };
|
|
232743
233342
|
}
|
|
232744
|
-
const
|
|
233343
|
+
const info = errorMessage7 ? humanizeSwarmAttemptError(errorMessage7) : void 0;
|
|
233344
|
+
const safeMessage = info?.message;
|
|
232745
233345
|
if (outcome === "rate_limited") {
|
|
232746
233346
|
return {
|
|
232747
233347
|
status: "rate_limited",
|
|
@@ -232755,11 +233355,14 @@ function terminalForOutcome(outcome, errorMessage7, errorReason) {
|
|
|
232755
233355
|
}
|
|
232756
233356
|
return {
|
|
232757
233357
|
status: "failed",
|
|
232758
|
-
errorCode: errorReason ?? "session_failed",
|
|
233358
|
+
errorCode: errorReason ?? info?.code ?? "session_failed",
|
|
232759
233359
|
...safeMessage ? { errorMessage: safeMessage } : {}
|
|
232760
233360
|
};
|
|
232761
233361
|
}
|
|
232762
|
-
function classifyRateLimit(message) {
|
|
233362
|
+
function classifyRateLimit(message, hint) {
|
|
233363
|
+
const refusal = hint ?? humanizeSwarmAttemptError(message);
|
|
233364
|
+
if (isTransientSpendRefusal(refusal.code, refusal.refusalReason))
|
|
233365
|
+
return "transient_capacity";
|
|
232763
233366
|
if (!message) return "provider_rate_limit";
|
|
232764
233367
|
if (isAccountLimit(message)) return "org_spend_cap";
|
|
232765
233368
|
if (/\bspend\b|\bcap\b|\bquota\b|\bbudget\b/i.test(message)) {
|
|
@@ -233051,7 +233654,7 @@ async function runJourneyFanOut(opts) {
|
|
|
233051
233654
|
emit({
|
|
233052
233655
|
type: "attempt_status",
|
|
233053
233656
|
status: "failed",
|
|
233054
|
-
errorMessage: message
|
|
233657
|
+
errorMessage: humanizeSwarmAttemptErrorMessage(message)
|
|
233055
233658
|
});
|
|
233056
233659
|
await reportAttempt(convexHttpUrl3, bearer, {
|
|
233057
233660
|
projectId,
|
|
@@ -233062,7 +233665,7 @@ async function runJourneyFanOut(opts) {
|
|
|
233062
233665
|
status: "failed",
|
|
233063
233666
|
chatSessionId,
|
|
233064
233667
|
errorCode: "sandbox_unavailable",
|
|
233065
|
-
errorMessage: message
|
|
233668
|
+
errorMessage: humanizeSwarmAttemptErrorMessage(message)
|
|
233066
233669
|
}).catch((err) => {
|
|
233067
233670
|
logger.error(
|
|
233068
233671
|
"[swarm.runner] failed to report sandbox-unavailable terminal",
|
|
@@ -233119,9 +233722,8 @@ async function runJourneyFanOut(opts) {
|
|
|
233119
233722
|
const failure = {
|
|
233120
233723
|
status: "failed",
|
|
233121
233724
|
errorCode: provisioned.code,
|
|
233122
|
-
errorMessage:
|
|
233123
|
-
|
|
233124
|
-
MAX_ATTEMPT_ERROR_CHARS
|
|
233725
|
+
errorMessage: humanizeSwarmAttemptErrorMessage(
|
|
233726
|
+
provisioned.message
|
|
233125
233727
|
)
|
|
233126
233728
|
};
|
|
233127
233729
|
emit({
|
|
@@ -233289,7 +233891,7 @@ async function runJourneyFanOut(opts) {
|
|
|
233289
233891
|
});
|
|
233290
233892
|
}
|
|
233291
233893
|
});
|
|
233292
|
-
const { outcome, errorMessage: errorMessage7, errorReason } = sessionResult;
|
|
233894
|
+
const { outcome, errorMessage: errorMessage7, errorReason, errorRefusal } = sessionResult;
|
|
233293
233895
|
if (stoppedByBackend) return;
|
|
233294
233896
|
const abortedBySpendCap = spendCapTripped && outcome === "failed" && sessionSignal.aborted;
|
|
233295
233897
|
const abortedByRunDeadline = outcome === "failed" && sessionSignal.aborted && runDeadline.firedClock() === "run";
|
|
@@ -233398,7 +234000,7 @@ async function runJourneyFanOut(opts) {
|
|
|
233398
234000
|
});
|
|
233399
234001
|
const accountLimitFailure = outcome === "failed" && !abortedBySpendCap && isAccountLimit(errorMessage7, errorReason);
|
|
233400
234002
|
if (outcome === "rate_limited" || accountLimitFailure) {
|
|
233401
|
-
const cause = classifyRateLimit(errorMessage7);
|
|
234003
|
+
const cause = classifyRateLimit(errorMessage7, errorRefusal);
|
|
233402
234004
|
if (cause === "org_spend_cap") {
|
|
233403
234005
|
spendCapTripped = true;
|
|
233404
234006
|
spendCapMessage = errorMessage7 ? humanizeSwarmAttemptErrorMessage(errorMessage7) : void 0;
|
|
@@ -233421,7 +234023,8 @@ async function runJourneyFanOut(opts) {
|
|
|
233421
234023
|
await markRemainingTargetAttemptsRateLimited(
|
|
233422
234024
|
{ convexHttpUrl: convexHttpUrl3, bearer, projectId, runId, target },
|
|
233423
234025
|
sessionIdx + 1,
|
|
233424
|
-
sessionsPerTarget
|
|
234026
|
+
sessionsPerTarget,
|
|
234027
|
+
cause === "transient_capacity" ? humanizeSwarmAttemptErrorMessage(errorMessage7) : void 0
|
|
233425
234028
|
);
|
|
233426
234029
|
return;
|
|
233427
234030
|
}
|
|
@@ -233635,7 +234238,7 @@ async function resolveTargetPinnedSkills(args) {
|
|
|
233635
234238
|
}
|
|
233636
234239
|
return artifacts;
|
|
233637
234240
|
}
|
|
233638
|
-
async function markRemainingTargetAttemptsRateLimited(ctx, fromIdx, toIdx) {
|
|
234241
|
+
async function markRemainingTargetAttemptsRateLimited(ctx, fromIdx, toIdx, transientMessage) {
|
|
233639
234242
|
const { convexHttpUrl: convexHttpUrl3, bearer, projectId, runId, target } = ctx;
|
|
233640
234243
|
const { hostId, targetId: targetId2 } = target;
|
|
233641
234244
|
for (let sessionIdx = fromIdx; sessionIdx < toIdx; sessionIdx++) {
|
|
@@ -233662,7 +234265,8 @@ async function markRemainingTargetAttemptsRateLimited(ctx, fromIdx, toIdx) {
|
|
|
233662
234265
|
sessionIdx,
|
|
233663
234266
|
status: "rate_limited",
|
|
233664
234267
|
chatSessionId,
|
|
233665
|
-
errorCode: "rate_limited"
|
|
234268
|
+
errorCode: transientMessage ? "user_rate_limit" : "rate_limited",
|
|
234269
|
+
...transientMessage ? { errorMessage: transientMessage } : {}
|
|
233666
234270
|
});
|
|
233667
234271
|
} catch (err) {
|
|
233668
234272
|
logger.warn(
|
|
@@ -233803,6 +234407,20 @@ async function resolveTargetPluginServers(getConvexClient, args) {
|
|
|
233803
234407
|
}
|
|
233804
234408
|
|
|
233805
234409
|
// server/services/sessionSimulation/launch-journey-run.ts
|
|
234410
|
+
var BROWSER_TOOL_ID2 = "browser";
|
|
234411
|
+
async function withoutBrowserOutsideRollout(hosts, workosUserId) {
|
|
234412
|
+
const wantsBrowser = (host) => (host.builtInToolIds ?? []).includes(BROWSER_TOOL_ID2);
|
|
234413
|
+
if (!hosts.some(wantsBrowser)) return hosts;
|
|
234414
|
+
if (workosUserId && await rolloutEnabled(false, workosUserId)) return hosts;
|
|
234415
|
+
return hosts.map(
|
|
234416
|
+
(host) => wantsBrowser(host) ? {
|
|
234417
|
+
...host,
|
|
234418
|
+
builtInToolIds: (host.builtInToolIds ?? []).filter(
|
|
234419
|
+
(id3) => id3 !== BROWSER_TOOL_ID2
|
|
234420
|
+
)
|
|
234421
|
+
} : host
|
|
234422
|
+
);
|
|
234423
|
+
}
|
|
233806
234424
|
var MAX_PASSTHROUGH_REASON_LENGTH = 300;
|
|
233807
234425
|
var LINE_BREAK = /[\r\n\f\u2028\u2029]/;
|
|
233808
234426
|
function launchFailureMessage(err) {
|
|
@@ -233886,12 +234504,15 @@ async function launchJourneyRun(deps, input) {
|
|
|
233886
234504
|
429: ErrorCode.RATE_LIMITED
|
|
233887
234505
|
};
|
|
233888
234506
|
const code = CODE_BY_STATUS[err.status] ?? ErrorCode.VALIDATION_ERROR;
|
|
233889
|
-
const
|
|
233890
|
-
|
|
233891
|
-
|
|
233892
|
-
|
|
233893
|
-
|
|
233894
|
-
|
|
234507
|
+
const details = launchFailureDetails(err);
|
|
234508
|
+
const modelError = environmentModelRequiredError({
|
|
234509
|
+
data: {
|
|
234510
|
+
code: details?.code,
|
|
234511
|
+
message: launchFailureMessage(err),
|
|
234512
|
+
details
|
|
234513
|
+
}
|
|
234514
|
+
});
|
|
234515
|
+
const routeError = modelError ?? new WebRouteError(err.status, code, launchFailureMessage(err), details);
|
|
233895
234516
|
throw err.retryAfter ? routeError.withHeaders({ "Retry-After": err.retryAfter }) : routeError;
|
|
233896
234517
|
}
|
|
233897
234518
|
throw err;
|
|
@@ -233913,11 +234534,15 @@ async function launchJourneyRun(deps, input) {
|
|
|
233913
234534
|
}
|
|
233914
234535
|
const hosts = snapshot2.hosts;
|
|
233915
234536
|
const getPluginRegateClient = async () => createConvexClient(await deps.getRunBearer());
|
|
233916
|
-
setImmediate(() => {
|
|
234537
|
+
setImmediate(async () => {
|
|
234538
|
+
const runHosts = await withoutBrowserOutsideRollout(
|
|
234539
|
+
hosts,
|
|
234540
|
+
deps.callerContext.workosUserId
|
|
234541
|
+
);
|
|
233917
234542
|
startJourneyRun({
|
|
233918
234543
|
runId,
|
|
233919
234544
|
projectId,
|
|
233920
|
-
hosts,
|
|
234545
|
+
hosts: runHosts,
|
|
233921
234546
|
personaSnapshot: snapshot2.personaSnapshot,
|
|
233922
234547
|
sessionsPerTarget: snapshot2.sessionsPerTarget,
|
|
233923
234548
|
maxTurns: snapshot2.maxTurns,
|
|
@@ -242934,8 +243559,8 @@ function measureClientMessage(buffer2, holdsInput) {
|
|
|
242934
243559
|
return decide(20, true);
|
|
242935
243560
|
case RFB_CLIENT_MESSAGE.SET_ENCODINGS: {
|
|
242936
243561
|
if (buffer2.length < 4) return { kind: "incomplete" };
|
|
242937
|
-
const
|
|
242938
|
-
return decide(4 +
|
|
243562
|
+
const count5 = readU16(buffer2, 2);
|
|
243563
|
+
return decide(4 + count5 * 4, true);
|
|
242939
243564
|
}
|
|
242940
243565
|
case RFB_CLIENT_MESSAGE.FRAMEBUFFER_UPDATE_REQUEST:
|
|
242941
243566
|
return decide(10, true);
|
|
@@ -243045,10 +243670,10 @@ function parseProtocolVersion(bytes2) {
|
|
|
243045
243670
|
}
|
|
243046
243671
|
function parseSecurityTypes(bytes2) {
|
|
243047
243672
|
if (bytes2.length < 1) return null;
|
|
243048
|
-
const
|
|
243049
|
-
if (
|
|
243050
|
-
if (bytes2.length < 1 +
|
|
243051
|
-
return Array.from(bytes2.subarray(1, 1 +
|
|
243673
|
+
const count5 = bytes2[0];
|
|
243674
|
+
if (count5 === 0) return [];
|
|
243675
|
+
if (bytes2.length < 1 + count5) return null;
|
|
243676
|
+
return Array.from(bytes2.subarray(1, 1 + count5));
|
|
243052
243677
|
}
|
|
243053
243678
|
function vncPasswordKey(password) {
|
|
243054
243679
|
const key = Buffer.alloc(8, 0);
|
|
@@ -254513,7 +255138,7 @@ async function runAssertionBacktest(input) {
|
|
|
254513
255138
|
const differences = [];
|
|
254514
255139
|
const seen = /* @__PURE__ */ new Set();
|
|
254515
255140
|
let page3;
|
|
254516
|
-
let
|
|
255141
|
+
let count5 = 0;
|
|
254517
255142
|
for (let n = 0; n < 10; n++) {
|
|
254518
255143
|
if (input.signal?.aborted) throw new Error("Backtest cancelled");
|
|
254519
255144
|
const priorPage = page3;
|
|
@@ -254562,7 +255187,7 @@ async function runAssertionBacktest(input) {
|
|
|
254562
255187
|
throw new Error("Duplicate or invalid backtest iteration identity");
|
|
254563
255188
|
seen.add(row2.iterationId);
|
|
254564
255189
|
differences.push(...backtestIteration(row2, draft));
|
|
254565
|
-
|
|
255190
|
+
count5++;
|
|
254566
255191
|
if (Date.now() > deadline2) throw new Error("Backtest deadline exceeded");
|
|
254567
255192
|
}
|
|
254568
255193
|
if (page3.isDone) break;
|
|
@@ -254575,7 +255200,7 @@ async function runAssertionBacktest(input) {
|
|
|
254575
255200
|
sourceHash: page3.sourceHash,
|
|
254576
255201
|
draftHash,
|
|
254577
255202
|
configRevision: page3.configRevision,
|
|
254578
|
-
complete: page3.isDone &&
|
|
255203
|
+
complete: page3.isDone && count5 > 0 && differences.length > 0 && comparable === differences.length,
|
|
254579
255204
|
continuationAvailable: !page3.isDone,
|
|
254580
255205
|
...!page3.isDone ? {
|
|
254581
255206
|
continuation: {
|
|
@@ -254586,7 +255211,7 @@ async function runAssertionBacktest(input) {
|
|
|
254586
255211
|
}
|
|
254587
255212
|
} : {},
|
|
254588
255213
|
counts: {
|
|
254589
|
-
iterations:
|
|
255214
|
+
iterations: count5,
|
|
254590
255215
|
comparable,
|
|
254591
255216
|
ungradable: differences.length - comparable,
|
|
254592
255217
|
flipped: differences.filter((row2) => row2.flipped).length
|
|
@@ -255290,8 +255915,8 @@ function describeClientImpact(input) {
|
|
|
255290
255915
|
[n("scenarioAttachmentCount"), "scenario attachment"],
|
|
255291
255916
|
[n("activeLegacyJourneyCount"), "active legacy journey"]
|
|
255292
255917
|
];
|
|
255293
|
-
const total = parts.reduce((sum2, [
|
|
255294
|
-
const listed = parts.map(([
|
|
255918
|
+
const total = parts.reduce((sum2, [count5]) => sum2 + count5, 0);
|
|
255919
|
+
const listed = parts.map(([count5, noun]) => `${count5} ${noun}${count5 === 1 ? "" : "s"}`).join(", ");
|
|
255295
255920
|
const affected = total === 0 ? "Nothing durable currently uses this client" : `This will affect ${listed}`;
|
|
255296
255921
|
return `${affected}; future direct client and playground use also follows the edit. ${unchanged}`;
|
|
255297
255922
|
}
|
|
@@ -256330,8 +256955,8 @@ var AGENT_OP_REGISTRY = [
|
|
|
256330
256955
|
tier: "gated",
|
|
256331
256956
|
proposal: {
|
|
256332
256957
|
describe: (input) => {
|
|
256333
|
-
const
|
|
256334
|
-
const name15 =
|
|
256958
|
+
const persona4 = input.persona;
|
|
256959
|
+
const name15 = persona4 && typeof persona4 === "object" ? named(persona4, "name") : void 0;
|
|
256335
256960
|
return name15 ? `Draft journeys for ${name15} with a model` : "Draft journeys with a model";
|
|
256336
256961
|
},
|
|
256337
256962
|
buttonLabel: "Draft them",
|