agent-inspect 6.31.8 → 6.31.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,11 @@
1
1
  # Changelog
2
2
 
3
+ ## 6.31.9
4
+
5
+ ### Patch Changes
6
+
7
+ - f71ec7e: Fail closed when suite cases declare no effective assertions or unknown selectors, wire eval.requireSuccess and related eval controls into real check rules, and share AsyncLocalStorage across packed CJS root and /advanced entrypoints so guarded wrappers see the active run context.
8
+
3
9
  ## 6.31.8
4
10
 
5
11
  ### Patch Changes
package/README.md CHANGED
@@ -213,7 +213,7 @@ The root package is enough for custom capture, the CLI, checks, and Evidence wor
213
213
 
214
214
  ## Status and documentation
215
215
 
216
- **Current published baseline:** **6.31.8** · persisted schema `1.0` · Node.js `>=20` · MIT.
216
+ **Current published baseline:** **6.31.9** · persisted schema `1.0` · Node.js `>=20` · MIT.
217
217
 
218
218
  Legacy v0.1 and v0.2 traces remain readable. Check the npm badge and [changelog](CHANGELOG.md) for the current published version.
219
219
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "agent-inspect",
3
- "version": "6.31.8",
3
+ "version": "6.31.9",
4
4
  "license": "MIT",
5
5
  "type": "module",
6
6
  "description": "Local evidence debugger and trajectory-test toolkit for TypeScript AI agents — execution trees, TraceContract checks, Evidence v2, and read-only MCP",
@@ -1627,7 +1627,18 @@ function extractCorrelationMetadata(record) {
1627
1627
  }
1628
1628
 
1629
1629
  // packages/core/src/context.ts
1630
- new AsyncLocalStorage();
1630
+ var GLOBAL_CONTEXT_ALS_KEY = "agent-inspect:execution-context-als:v1";
1631
+ function getSharedContextStorage() {
1632
+ const host = globalThis;
1633
+ const existing = host[GLOBAL_CONTEXT_ALS_KEY];
1634
+ if (existing instanceof AsyncLocalStorage) {
1635
+ return existing;
1636
+ }
1637
+ const created = new AsyncLocalStorage();
1638
+ host[GLOBAL_CONTEXT_ALS_KEY] = created;
1639
+ return created;
1640
+ }
1641
+ getSharedContextStorage();
1631
1642
 
1632
1643
  // packages/core/src/terminal.ts
1633
1644
  var TERMINAL_INDENT = " ";
@@ -8407,6 +8418,45 @@ function createToolOrderingRule(options) {
8407
8418
  requireEndpoints: false
8408
8419
  });
8409
8420
  }
8421
+ function createToolFailureRule(options) {
8422
+ return {
8423
+ id: "tool.failures",
8424
+ category: "tool",
8425
+ defaultSeverity: "error",
8426
+ evaluate(context) {
8427
+ const tools = finishedEvents(context, "TOOL");
8428
+ const failures = tools.filter((event) => event.status === "error");
8429
+ const retries = tools.map((event) => ({ event, count: retryCount(event) })).filter((item) => item.count !== void 0);
8430
+ const findings = [];
8431
+ if (options.maxFailures !== void 0 && failures.length > options.maxFailures) {
8432
+ findings.push(
8433
+ failFinding(
8434
+ "tool.failures",
8435
+ `Tool failure count ${failures.length} exceeded ${options.maxFailures}.`,
8436
+ failures.map((event) => eventEvidence3(event)),
8437
+ { maxFailures: options.maxFailures },
8438
+ failures.length
8439
+ )
8440
+ );
8441
+ }
8442
+ if (options.maxRetries !== void 0) {
8443
+ const excessiveRetries = retries.filter((item) => item.count > options.maxRetries);
8444
+ if (excessiveRetries.length > 0) {
8445
+ findings.push(
8446
+ failFinding(
8447
+ "tool.failures",
8448
+ `Tool retry count exceeded ${options.maxRetries}.`,
8449
+ excessiveRetries.map((item) => eventEvidence3(item.event, "attributes.retryCount")),
8450
+ { maxRetries: options.maxRetries },
8451
+ excessiveRetries.map((item) => ({ tool: toolName(item.event), retries: item.count }))
8452
+ )
8453
+ );
8454
+ }
8455
+ }
8456
+ return findings;
8457
+ }
8458
+ };
8459
+ }
8410
8460
  function createLlmUsageRule(options) {
8411
8461
  return {
8412
8462
  id: "llm.usage",
@@ -13649,21 +13699,48 @@ async function resolveSuiteCaseTrace(suiteCase, options) {
13649
13699
  }
13650
13700
 
13651
13701
  // packages/core/src/suite/run.ts
13702
+ var KNOWN_SUITE_SELECT_IDS = /* @__PURE__ */ new Set([
13703
+ "run.status",
13704
+ "run.duration",
13705
+ "run.depth",
13706
+ "outcome.status",
13707
+ "tool.usage",
13708
+ "tool.failures",
13709
+ "llm.usage"
13710
+ ]);
13652
13711
  function diagnostic5(code, message, severity = "error", caseId) {
13653
13712
  return { code, message, severity, ...caseId !== void 0 ? { caseId } : {} };
13654
13713
  }
13655
- function buildCaseRules(suiteCase, config) {
13714
+ function buildCaseAssertions(suiteCase, config) {
13656
13715
  const rules = [];
13657
13716
  const select = new Set(config.checks?.select ?? []);
13658
- if (select.has("run.status")) {
13717
+ const configDiagnostics = [];
13718
+ const evalConfig = config.eval;
13719
+ const declaredSelect = [...select];
13720
+ for (const id of declaredSelect) {
13721
+ if (!KNOWN_SUITE_SELECT_IDS.has(id)) {
13722
+ configDiagnostics.push(
13723
+ diagnostic5(
13724
+ "AI_SUITE_UNKNOWN_SELECTOR",
13725
+ `Unknown or unsupported suite check selector "${id}".`,
13726
+ "error",
13727
+ suiteCase.id
13728
+ )
13729
+ );
13730
+ }
13731
+ }
13732
+ if (select.has("run.status") || evalConfig?.requireSuccess === true) {
13659
13733
  rules.push(createRunStatusRule());
13734
+ select.add("run.status");
13660
13735
  }
13661
13736
  const requiredTools = [
13662
13737
  ...config.checks?.tool?.required ?? [],
13738
+ ...evalConfig?.requiredTools ?? [],
13663
13739
  ...suiteCase.requireTools ?? []
13664
13740
  ];
13665
13741
  const forbiddenTools = [
13666
13742
  ...config.checks?.tool?.forbidden ?? [],
13743
+ ...evalConfig?.forbiddenTools ?? [],
13667
13744
  ...suiteCase.forbidTools ?? []
13668
13745
  ];
13669
13746
  if (requiredTools.length > 0 || forbiddenTools.length > 0) {
@@ -13675,20 +13752,51 @@ function buildCaseRules(suiteCase, config) {
13675
13752
  );
13676
13753
  select.add("tool.usage");
13677
13754
  }
13678
- const maxDurationMs = suiteCase.maxDurationMs ?? config.checks?.run?.maxDurationMs ?? config.eval?.maxDurationMs;
13755
+ const maxDurationMs = suiteCase.maxDurationMs ?? config.checks?.run?.maxDurationMs ?? evalConfig?.maxDurationMs;
13679
13756
  if (maxDurationMs !== void 0) {
13680
13757
  rules.push(createRunDurationRule({ maxDurationMs }));
13681
13758
  select.add("run.duration");
13682
13759
  }
13683
- const llm = config.checks?.llm;
13684
- if (llm?.allowedModels !== void 0 || llm?.maxTotalTokens !== void 0) {
13685
- rules.push(createLlmUsageRule(llm));
13760
+ const maxDepth = config.checks?.run?.maxDepth ?? evalConfig?.maxDepth;
13761
+ if (maxDepth !== void 0) {
13762
+ rules.push(createRunDepthRule({ maxDepth }));
13763
+ select.add("run.depth");
13764
+ }
13765
+ if (evalConfig?.maxRetries !== void 0) {
13766
+ rules.push(createToolFailureRule({ maxRetries: evalConfig.maxRetries }));
13767
+ select.add("tool.failures");
13768
+ }
13769
+ const allowedModels = config.checks?.llm?.allowedModels;
13770
+ const maxTotalTokens = config.checks?.llm?.maxTotalTokens ?? evalConfig?.maxTotalTokens;
13771
+ if (allowedModels !== void 0 || maxTotalTokens !== void 0) {
13772
+ rules.push(
13773
+ createLlmUsageRule({
13774
+ ...allowedModels !== void 0 ? { allowedModels } : {},
13775
+ ...maxTotalTokens !== void 0 ? { maxTotalTokens } : {}
13776
+ })
13777
+ );
13686
13778
  select.add("llm.usage");
13687
13779
  }
13688
13780
  if (select.has("outcome.status")) {
13689
13781
  rules.push(createObservedOutcomeRule({ failOn: ["failed"] }));
13690
13782
  }
13691
- return { rules, select: [...select] };
13783
+ const observationCount = suiteCase.expectedObservations?.length ?? 0;
13784
+ if (rules.length === 0 && observationCount === 0 && configDiagnostics.length === 0) {
13785
+ configDiagnostics.push(
13786
+ diagnostic5(
13787
+ "AI_SUITE_NO_ASSERTIONS",
13788
+ `Suite case "${suiteCase.id}" declares no effective checks, eval controls, or expected observations.`,
13789
+ "error",
13790
+ suiteCase.id
13791
+ )
13792
+ );
13793
+ }
13794
+ return {
13795
+ rules,
13796
+ select: [...select],
13797
+ observationCount,
13798
+ configDiagnostics
13799
+ };
13692
13800
  }
13693
13801
  function outcomesFromRead(read) {
13694
13802
  const persistedOutcomes = extractOutcomesFromPersistedEvents(
@@ -13769,8 +13877,20 @@ async function runSuiteCase(suiteCase, config, options) {
13769
13877
  ]
13770
13878
  };
13771
13879
  }
13772
- const { rules, select } = buildCaseRules(suiteCase, config);
13773
- const checkResult = rules.length > 0 ? runTraceChecks({ read }, { rules, select }) : {
13880
+ const compiled = buildCaseAssertions(suiteCase, config);
13881
+ if (compiled.configDiagnostics.length > 0) {
13882
+ return {
13883
+ id: suiteCase.id,
13884
+ status: "error",
13885
+ tracePath: resolved.tracePath,
13886
+ ...resolved.runId !== void 0 ? { runId: resolved.runId } : {},
13887
+ checkOk: false,
13888
+ ...config.eval?.requireSuccess === true ? { evalOk: false } : {},
13889
+ message: compiled.configDiagnostics.map((item) => item.message).join("; "),
13890
+ diagnostics: compiled.configDiagnostics
13891
+ };
13892
+ }
13893
+ const checkResult = compiled.rules.length > 0 ? runTraceChecks({ read }, { rules: compiled.rules, select: compiled.select }) : {
13774
13894
  ok: true,
13775
13895
  status: "pass",
13776
13896
  format: read.format,
@@ -13798,6 +13918,9 @@ async function runSuiteCase(suiteCase, config, options) {
13798
13918
  ];
13799
13919
  const checkOk = checkResult.ok;
13800
13920
  const observationsOk = observationResult.ok;
13921
+ const evalOk = config.eval?.requireSuccess === true ? checkResult.findings.every(
13922
+ (finding) => finding.ruleId !== "run.status" || finding.status !== "fail"
13923
+ ) && checkResult.diagnostics.every((item) => item.severity !== "error") : void 0;
13801
13924
  const ok = checkOk && observationsOk;
13802
13925
  const status = ok ? "pass" : checkResult.status === "error" ? "error" : "fail";
13803
13926
  return {
@@ -13806,6 +13929,7 @@ async function runSuiteCase(suiteCase, config, options) {
13806
13929
  tracePath: resolved.tracePath,
13807
13930
  ...resolved.runId !== void 0 ? { runId: resolved.runId } : {},
13808
13931
  checkOk,
13932
+ ...evalOk !== void 0 ? { evalOk } : {},
13809
13933
  observationsOk,
13810
13934
  diagnostics,
13811
13935
  ...ok ? {} : {
@@ -14973,5 +15097,5 @@ function renderGateReport(result, options = {}) {
14973
15097
  }
14974
15098
 
14975
15099
  export { ATTRIBUTION_CONFIDENCES, COHORT_METRIC_IDS, DEFAULT_REDACT_KEYS, DEFAULT_SUITE_ARTIFACTS_DIR, EVIDENCE_FORMAT_VERSION, EVIDENCE_HTML_FILENAME, EVIDENCE_MANIFEST_FILENAME, Redactor, TraceDirectory, TraceReadError, TreeBuilder, aggregateBundleSafeStatus, aggregateSessionCheckResults, analyzeCohort, applyProfileMetadataCaps, assertBundlePathContained, assertEvidenceRelativePath, buildActivitySummary, buildBundleMetadata, buildBundleSummaryMarkdown, buildEvidenceCausalFailureViewHtml, buildEvidenceCiPackage, buildEvidenceCircuitViewHtml, buildEvidenceContractsViewHtml, buildEvidenceDiffViewHtml, buildEvidenceHtmlShell, buildEvidenceManifest, buildEvidenceOutcomesViewHtml, buildEvidenceProvenanceViewHtml, buildEvidenceSafetyViewHtml, buildEvidenceTimelineViewHtml, buildEvidenceToolsLlmViewHtml, buildEvidenceTreeViewHtml, buildLocalExplanation, buildPlaceholderArtifact, buildRunSummary, buildRunTimeline, buildRunWhatSummary, buildSessionIndex, buildTraceStats, buildZipArchive, bundleFailsOnSafety, bundleRunAssetRelativePath, collectTraceSchemaVersions, compactAttributes, createBaselineRegressionRule, createLlmUsageRule, createMaxStepDurationRule, createObservedOutcomeRule, createRequireCompletedRule, createRunDepthRule, createRunDurationRule, createRunStatusRule, createSafetyOversizedAttributeRule, createSafetyRawContentRule, createSafetyRedactionRule, createSafetySecretPatternRule, createStallDetectionRule, createStructureCycleRule, createStructureOrphanRule, createStructureParallelWidthRule, createStructureRelationshipRule, createToolUsageRule, defaultBundleOutputPath, defaultSuiteConfigTemplate, defineTraceContract, diffRuns, diffTraceEvents, enrichSessionRunRecord, escapeHtml, escapeMarkdown, evaluateTraceContract, extractMetadata, extractOutcomesFromTraceEvents, filterMetasBySessionScope, filterTraces, flattenTree, formatDuration2 as formatDuration, formatStepLabel, formatTimestamp, gateHasThresholds, getIndent, getTraceFilePath, inferEvidenceFileRole, isAgentInspectTrace, isAttributionConfidence, isCredentialSensitiveKey, isPersistedInspectEvent, loadSessionRunRecords, loadSuiteConfig, loadTraceMetadataList, manualTraceEventsToComparableRun, nanoid, normalizeBundleOutputPath, openTrace, parseCohortMetricList, parseDuration, parseDurationFilter, parseGateList, parseTraceJsonl, persistedInspectEventsToRunTrees, persistedInspectEventsToTraceEvents, projectLogicalEvents, renderActivitySummaryHuman, renderCohortReport, renderErrorLine, renderGateReport, renderObservedOutcomesHtml, renderObservedOutcomesMarkdown, renderRunDiff, renderRunWhat, renderStepLine, renderSuiteReport, renderTimeline, renderTraceStats, resolveBundleRunIds, resolveRedactionProfile, resolveSuiteTemplate, resolveTraceDir, runGate, runSuite, runTraceChecks, safeString, sanitizeBundleRunId, searchTraces, serializeEvidenceManifest, sha256Hex, stableJson, stringContainsHighConfidenceCredential, summarizeObservedOutcomes, summarizeSemanticParity, traceEventToPersistedInspectEvent, truncateName, truncateStringForProfile, validateEvent, validateSuiteConfig, verifyEvidenceDirectory, zeroKinds };
14976
- //# sourceMappingURL=chunk-6FE47XSF.mjs.map
14977
- //# sourceMappingURL=chunk-6FE47XSF.mjs.map
15100
+ //# sourceMappingURL=chunk-SWUZRRDT.mjs.map
15101
+ //# sourceMappingURL=chunk-SWUZRRDT.mjs.map