agent-inspect 6.31.8 → 6.31.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1252,11 +1252,23 @@ var init_trace_event_safety = __esm({
1252
1252
  init_persisted_inspect_event();
1253
1253
  }
1254
1254
  });
1255
+ function getSharedContextStorage() {
1256
+ const host = globalThis;
1257
+ const existing = host[GLOBAL_CONTEXT_ALS_KEY];
1258
+ if (existing instanceof async_hooks.AsyncLocalStorage) {
1259
+ return existing;
1260
+ }
1261
+ const created = new async_hooks.AsyncLocalStorage();
1262
+ host[GLOBAL_CONTEXT_ALS_KEY] = created;
1263
+ return created;
1264
+ }
1265
+ var GLOBAL_CONTEXT_ALS_KEY;
1255
1266
  var init_context = __esm({
1256
1267
  "packages/core/src/context.ts"() {
1257
1268
  init_correlation_metadata();
1258
1269
  init_trace_event_safety();
1259
- new async_hooks.AsyncLocalStorage();
1270
+ GLOBAL_CONTEXT_ALS_KEY = "agent-inspect:execution-context-als:v1";
1271
+ getSharedContextStorage();
1260
1272
  }
1261
1273
  });
1262
1274
  var init_inspector_runtime = __esm({
@@ -6524,6 +6536,45 @@ function createToolOrderingRule(options) {
6524
6536
  requireEndpoints: false
6525
6537
  });
6526
6538
  }
6539
+ function createToolFailureRule(options) {
6540
+ return {
6541
+ id: "tool.failures",
6542
+ category: "tool",
6543
+ defaultSeverity: "error",
6544
+ evaluate(context) {
6545
+ const tools = finishedEvents(context, "TOOL");
6546
+ const failures = tools.filter((event) => event.status === "error");
6547
+ const retries = tools.map((event) => ({ event, count: retryCount(event) })).filter((item) => item.count !== void 0);
6548
+ const findings = [];
6549
+ if (options.maxFailures !== void 0 && failures.length > options.maxFailures) {
6550
+ findings.push(
6551
+ failFinding(
6552
+ "tool.failures",
6553
+ `Tool failure count ${failures.length} exceeded ${options.maxFailures}.`,
6554
+ failures.map((event) => eventEvidence(event)),
6555
+ { maxFailures: options.maxFailures },
6556
+ failures.length
6557
+ )
6558
+ );
6559
+ }
6560
+ if (options.maxRetries !== void 0) {
6561
+ const excessiveRetries = retries.filter((item) => item.count > options.maxRetries);
6562
+ if (excessiveRetries.length > 0) {
6563
+ findings.push(
6564
+ failFinding(
6565
+ "tool.failures",
6566
+ `Tool retry count exceeded ${options.maxRetries}.`,
6567
+ excessiveRetries.map((item) => eventEvidence(item.event, "attributes.retryCount")),
6568
+ { maxRetries: options.maxRetries },
6569
+ excessiveRetries.map((item) => ({ tool: toolName(item.event), retries: item.count }))
6570
+ )
6571
+ );
6572
+ }
6573
+ }
6574
+ return findings;
6575
+ }
6576
+ };
6577
+ }
6527
6578
  function createLlmUsageRule(options) {
6528
6579
  return {
6529
6580
  id: "llm.usage",
@@ -14297,18 +14348,36 @@ var init_readers = __esm({
14297
14348
  function diagnostic5(code, message, severity = "error", caseId) {
14298
14349
  return { code, message, severity, ...caseId !== void 0 ? { caseId } : {} };
14299
14350
  }
14300
- function buildCaseRules(suiteCase, config) {
14351
+ function buildCaseAssertions(suiteCase, config) {
14301
14352
  const rules = [];
14302
14353
  const select = new Set(config.checks?.select ?? []);
14303
- if (select.has("run.status")) {
14354
+ const configDiagnostics = [];
14355
+ const evalConfig = config.eval;
14356
+ const declaredSelect = [...select];
14357
+ for (const id of declaredSelect) {
14358
+ if (!KNOWN_SUITE_SELECT_IDS.has(id)) {
14359
+ configDiagnostics.push(
14360
+ diagnostic5(
14361
+ "AI_SUITE_UNKNOWN_SELECTOR",
14362
+ `Unknown or unsupported suite check selector "${id}".`,
14363
+ "error",
14364
+ suiteCase.id
14365
+ )
14366
+ );
14367
+ }
14368
+ }
14369
+ if (select.has("run.status") || evalConfig?.requireSuccess === true) {
14304
14370
  rules.push(createRunStatusRule());
14371
+ select.add("run.status");
14305
14372
  }
14306
14373
  const requiredTools = [
14307
14374
  ...config.checks?.tool?.required ?? [],
14375
+ ...evalConfig?.requiredTools ?? [],
14308
14376
  ...suiteCase.requireTools ?? []
14309
14377
  ];
14310
14378
  const forbiddenTools = [
14311
14379
  ...config.checks?.tool?.forbidden ?? [],
14380
+ ...evalConfig?.forbiddenTools ?? [],
14312
14381
  ...suiteCase.forbidTools ?? []
14313
14382
  ];
14314
14383
  if (requiredTools.length > 0 || forbiddenTools.length > 0) {
@@ -14320,20 +14389,51 @@ function buildCaseRules(suiteCase, config) {
14320
14389
  );
14321
14390
  select.add("tool.usage");
14322
14391
  }
14323
- const maxDurationMs = suiteCase.maxDurationMs ?? config.checks?.run?.maxDurationMs ?? config.eval?.maxDurationMs;
14392
+ const maxDurationMs = suiteCase.maxDurationMs ?? config.checks?.run?.maxDurationMs ?? evalConfig?.maxDurationMs;
14324
14393
  if (maxDurationMs !== void 0) {
14325
14394
  rules.push(createRunDurationRule({ maxDurationMs }));
14326
14395
  select.add("run.duration");
14327
14396
  }
14328
- const llm = config.checks?.llm;
14329
- if (llm?.allowedModels !== void 0 || llm?.maxTotalTokens !== void 0) {
14330
- rules.push(createLlmUsageRule(llm));
14397
+ const maxDepth = config.checks?.run?.maxDepth ?? evalConfig?.maxDepth;
14398
+ if (maxDepth !== void 0) {
14399
+ rules.push(createRunDepthRule({ maxDepth }));
14400
+ select.add("run.depth");
14401
+ }
14402
+ if (evalConfig?.maxRetries !== void 0) {
14403
+ rules.push(createToolFailureRule({ maxRetries: evalConfig.maxRetries }));
14404
+ select.add("tool.failures");
14405
+ }
14406
+ const allowedModels = config.checks?.llm?.allowedModels;
14407
+ const maxTotalTokens = config.checks?.llm?.maxTotalTokens ?? evalConfig?.maxTotalTokens;
14408
+ if (allowedModels !== void 0 || maxTotalTokens !== void 0) {
14409
+ rules.push(
14410
+ createLlmUsageRule({
14411
+ ...allowedModels !== void 0 ? { allowedModels } : {},
14412
+ ...maxTotalTokens !== void 0 ? { maxTotalTokens } : {}
14413
+ })
14414
+ );
14331
14415
  select.add("llm.usage");
14332
14416
  }
14333
14417
  if (select.has("outcome.status")) {
14334
14418
  rules.push(createObservedOutcomeRule({ failOn: ["failed"] }));
14335
14419
  }
14336
- return { rules, select: [...select] };
14420
+ const observationCount = suiteCase.expectedObservations?.length ?? 0;
14421
+ if (rules.length === 0 && observationCount === 0 && configDiagnostics.length === 0) {
14422
+ configDiagnostics.push(
14423
+ diagnostic5(
14424
+ "AI_SUITE_NO_ASSERTIONS",
14425
+ `Suite case "${suiteCase.id}" declares no effective checks, eval controls, or expected observations.`,
14426
+ "error",
14427
+ suiteCase.id
14428
+ )
14429
+ );
14430
+ }
14431
+ return {
14432
+ rules,
14433
+ select: [...select],
14434
+ observationCount,
14435
+ configDiagnostics
14436
+ };
14337
14437
  }
14338
14438
  function outcomesFromRead(read) {
14339
14439
  const persistedOutcomes = extractOutcomesFromPersistedEvents(
@@ -14414,8 +14514,20 @@ async function runSuiteCase(suiteCase, config, options) {
14414
14514
  ]
14415
14515
  };
14416
14516
  }
14417
- const { rules, select } = buildCaseRules(suiteCase, config);
14418
- const checkResult = rules.length > 0 ? runTraceChecks({ read }, { rules, select }) : {
14517
+ const compiled = buildCaseAssertions(suiteCase, config);
14518
+ if (compiled.configDiagnostics.length > 0) {
14519
+ return {
14520
+ id: suiteCase.id,
14521
+ status: "error",
14522
+ tracePath: resolved.tracePath,
14523
+ ...resolved.runId !== void 0 ? { runId: resolved.runId } : {},
14524
+ checkOk: false,
14525
+ ...config.eval?.requireSuccess === true ? { evalOk: false } : {},
14526
+ message: compiled.configDiagnostics.map((item) => item.message).join("; "),
14527
+ diagnostics: compiled.configDiagnostics
14528
+ };
14529
+ }
14530
+ const checkResult = compiled.rules.length > 0 ? runTraceChecks({ read }, { rules: compiled.rules, select: compiled.select }) : {
14419
14531
  ok: true,
14420
14532
  status: "pass",
14421
14533
  format: read.format,
@@ -14443,6 +14555,9 @@ async function runSuiteCase(suiteCase, config, options) {
14443
14555
  ];
14444
14556
  const checkOk = checkResult.ok;
14445
14557
  const observationsOk = observationResult.ok;
14558
+ const evalOk = config.eval?.requireSuccess === true ? checkResult.findings.every(
14559
+ (finding) => finding.ruleId !== "run.status" || finding.status !== "fail"
14560
+ ) && checkResult.diagnostics.every((item) => item.severity !== "error") : void 0;
14446
14561
  const ok = checkOk && observationsOk;
14447
14562
  const status = ok ? "pass" : checkResult.status === "error" ? "error" : "fail";
14448
14563
  return {
@@ -14451,6 +14566,7 @@ async function runSuiteCase(suiteCase, config, options) {
14451
14566
  tracePath: resolved.tracePath,
14452
14567
  ...resolved.runId !== void 0 ? { runId: resolved.runId } : {},
14453
14568
  checkOk,
14569
+ ...evalOk !== void 0 ? { evalOk } : {},
14454
14570
  observationsOk,
14455
14571
  diagnostics,
14456
14572
  ...ok ? {} : {
@@ -14495,6 +14611,7 @@ async function runSuite(options = {}) {
14495
14611
  diagnostics
14496
14612
  };
14497
14613
  }
14614
+ var KNOWN_SUITE_SELECT_IDS;
14498
14615
  var init_run = __esm({
14499
14616
  "packages/core/src/suite/run.ts"() {
14500
14617
  init_checks2();
@@ -14502,6 +14619,15 @@ var init_run = __esm({
14502
14619
  init_readers();
14503
14620
  init_load2();
14504
14621
  init_resolve2();
14622
+ KNOWN_SUITE_SELECT_IDS = /* @__PURE__ */ new Set([
14623
+ "run.status",
14624
+ "run.duration",
14625
+ "run.depth",
14626
+ "outcome.status",
14627
+ "tool.usage",
14628
+ "tool.failures",
14629
+ "llm.usage"
14630
+ ]);
14505
14631
  }
14506
14632
  });
14507
14633
 
@@ -16249,7 +16375,7 @@ var init_src = __esm({
16249
16375
  });
16250
16376
 
16251
16377
  // package.json
16252
- var version = "6.31.8";
16378
+ var version = "6.31.9";
16253
16379
 
16254
16380
  // packages/cli/src/list.ts
16255
16381
  init_advanced();