agent-inspect 6.31.8 → 6.31.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1269,7 +1269,7 @@ interface EvidenceCiPackageFiles {
1269
1269
  declare function buildEvidenceCiPackage(input: EvidenceCiPackageInput): EvidenceCiPackageFiles;
1270
1270
 
1271
1271
  type SuiteCaseStatus = "pass" | "fail" | "error" | "skipped";
1272
- type SuiteDiagnosticCode = "AI_SUITE_CONFIG_INVALID" | "AI_SUITE_CONFIG_LOAD_FAILED" | "AI_SUITE_CASE_TRACE_MISSING" | "AI_SUITE_CASE_CHECK_FAILED" | "AI_SUITE_CASE_EVAL_FAILED" | "AI_SUITE_CASE_OBSERVATION_FAILED" | "AI_SUITE_TRACE_UNREADABLE";
1272
+ type SuiteDiagnosticCode = "AI_SUITE_CONFIG_INVALID" | "AI_SUITE_CONFIG_LOAD_FAILED" | "AI_SUITE_CASE_TRACE_MISSING" | "AI_SUITE_CASE_CHECK_FAILED" | "AI_SUITE_CASE_EVAL_FAILED" | "AI_SUITE_CASE_OBSERVATION_FAILED" | "AI_SUITE_NO_ASSERTIONS" | "AI_SUITE_UNKNOWN_SELECTOR" | "AI_SUITE_TRACE_UNREADABLE";
1273
1273
  interface SuiteDiagnostic {
1274
1274
  code: SuiteDiagnosticCode;
1275
1275
  severity: "error" | "warning" | "info";
@@ -1269,7 +1269,7 @@ interface EvidenceCiPackageFiles {
1269
1269
  declare function buildEvidenceCiPackage(input: EvidenceCiPackageInput): EvidenceCiPackageFiles;
1270
1270
 
1271
1271
  type SuiteCaseStatus = "pass" | "fail" | "error" | "skipped";
1272
- type SuiteDiagnosticCode = "AI_SUITE_CONFIG_INVALID" | "AI_SUITE_CONFIG_LOAD_FAILED" | "AI_SUITE_CASE_TRACE_MISSING" | "AI_SUITE_CASE_CHECK_FAILED" | "AI_SUITE_CASE_EVAL_FAILED" | "AI_SUITE_CASE_OBSERVATION_FAILED" | "AI_SUITE_TRACE_UNREADABLE";
1272
+ type SuiteDiagnosticCode = "AI_SUITE_CONFIG_INVALID" | "AI_SUITE_CONFIG_LOAD_FAILED" | "AI_SUITE_CASE_TRACE_MISSING" | "AI_SUITE_CASE_CHECK_FAILED" | "AI_SUITE_CASE_EVAL_FAILED" | "AI_SUITE_CASE_OBSERVATION_FAILED" | "AI_SUITE_NO_ASSERTIONS" | "AI_SUITE_UNKNOWN_SELECTOR" | "AI_SUITE_TRACE_UNREADABLE";
1273
1273
  interface SuiteDiagnostic {
1274
1274
  code: SuiteDiagnosticCode;
1275
1275
  severity: "error" | "warning" | "info";
@@ -1,6 +1,6 @@
1
- import { extractSessionWorkflowMetadata, sessionKeyForRun, defineTraceContract, runTraceChecks, createRunStatusRule, createToolUsageRule, createRunDurationRule, createLlmUsageRule, createObservedOutcomeRule } from './chunk-2CCRUI42.mjs';
1
+ import { extractSessionWorkflowMetadata, sessionKeyForRun, defineTraceContract, runTraceChecks, createRunStatusRule, createToolUsageRule, createRunDurationRule, createRunDepthRule, createToolFailureRule, createLlmUsageRule, createObservedOutcomeRule } from './chunk-2CCRUI42.mjs';
2
2
  export { extractSessionWorkflowMetadata, sessionKeyForRun } from './chunk-2CCRUI42.mjs';
3
- export { MAX_TERMINAL_DEPTH, MAX_TERMINAL_NAME_LENGTH, TERMINAL_INDENT, TraceDirectory, createInspector, createInspectorRuntime, formatTerminalName, getCurrentContext, getCurrentCorrelationMetadata, getCurrentDepth, getCurrentRunId, getCurrentRunName, getCurrentStepId, getIndent, getParentStepId, getTraceDirFromContext, getTraceSafetyFromContext, hasActiveContext, isAgentInspectEnabled, isSilentContext, maybeInspectRun, printError, printFailedAt, printRunComplete, printRunStart, printStepComplete, printStepStart, renderErrorLine, renderRunSummary, renderStepLine, resolveTraceDir, runWithContext, runWithStepContext } from './chunk-VQUB3WRI.mjs';
3
+ export { MAX_TERMINAL_DEPTH, MAX_TERMINAL_NAME_LENGTH, TERMINAL_INDENT, TraceDirectory, createInspector, createInspectorRuntime, formatTerminalName, getCurrentContext, getCurrentCorrelationMetadata, getCurrentDepth, getCurrentRunId, getCurrentRunName, getCurrentStepId, getIndent, getParentStepId, getTraceDirFromContext, getTraceSafetyFromContext, hasActiveContext, isAgentInspectEnabled, isSilentContext, maybeInspectRun, printError, printFailedAt, printRunComplete, printRunStart, printStepComplete, printStepStart, renderErrorLine, renderRunSummary, renderStepLine, resolveTraceDir, runWithContext, runWithStepContext } from './chunk-3SPBELJB.mjs';
4
4
  import { buildRunTimeline, filterTraces, extractMetadata, buildRunSummary } from './chunk-7YYS742B.mjs';
5
5
  export { buildRunSummary, buildRunTimeline, buildRunWhatSummary, buildTraceStats, extractMetadata, filterTraces, formatStepLabel, renderRunWhat, renderTimeline, renderTraceStats } from './chunk-7YYS742B.mjs';
6
6
  import { extractOutcomesFromTraceEvents, parseObservationFilter, summarizeObservedOutcomes, extractOutcomesFromPersistedEvents, renderObservedOutcomesHtml } from './chunk-N5PZAI6Q.mjs';
@@ -3626,21 +3626,48 @@ async function resolveSuiteCaseTrace(suiteCase, options) {
3626
3626
  reason: `no trace found for run id "${runKey}" under ${options.tracesDir}`
3627
3627
  };
3628
3628
  }
3629
+ var KNOWN_SUITE_SELECT_IDS = /* @__PURE__ */ new Set([
3630
+ "run.status",
3631
+ "run.duration",
3632
+ "run.depth",
3633
+ "outcome.status",
3634
+ "tool.usage",
3635
+ "tool.failures",
3636
+ "llm.usage"
3637
+ ]);
3629
3638
  function diagnostic3(code, message, severity = "error", caseId) {
3630
3639
  return { code, message, severity, ...caseId !== void 0 ? { caseId } : {} };
3631
3640
  }
3632
- function buildCaseRules(suiteCase, config) {
3641
+ function buildCaseAssertions(suiteCase, config) {
3633
3642
  const rules = [];
3634
3643
  const select = new Set(config.checks?.select ?? []);
3635
- if (select.has("run.status")) {
3644
+ const configDiagnostics = [];
3645
+ const evalConfig = config.eval;
3646
+ const declaredSelect = [...select];
3647
+ for (const id of declaredSelect) {
3648
+ if (!KNOWN_SUITE_SELECT_IDS.has(id)) {
3649
+ configDiagnostics.push(
3650
+ diagnostic3(
3651
+ "AI_SUITE_UNKNOWN_SELECTOR",
3652
+ `Unknown or unsupported suite check selector "${id}".`,
3653
+ "error",
3654
+ suiteCase.id
3655
+ )
3656
+ );
3657
+ }
3658
+ }
3659
+ if (select.has("run.status") || evalConfig?.requireSuccess === true) {
3636
3660
  rules.push(createRunStatusRule());
3661
+ select.add("run.status");
3637
3662
  }
3638
3663
  const requiredTools = [
3639
3664
  ...config.checks?.tool?.required ?? [],
3665
+ ...evalConfig?.requiredTools ?? [],
3640
3666
  ...suiteCase.requireTools ?? []
3641
3667
  ];
3642
3668
  const forbiddenTools = [
3643
3669
  ...config.checks?.tool?.forbidden ?? [],
3670
+ ...evalConfig?.forbiddenTools ?? [],
3644
3671
  ...suiteCase.forbidTools ?? []
3645
3672
  ];
3646
3673
  if (requiredTools.length > 0 || forbiddenTools.length > 0) {
@@ -3652,20 +3679,51 @@ function buildCaseRules(suiteCase, config) {
3652
3679
  );
3653
3680
  select.add("tool.usage");
3654
3681
  }
3655
- const maxDurationMs = suiteCase.maxDurationMs ?? config.checks?.run?.maxDurationMs ?? config.eval?.maxDurationMs;
3682
+ const maxDurationMs = suiteCase.maxDurationMs ?? config.checks?.run?.maxDurationMs ?? evalConfig?.maxDurationMs;
3656
3683
  if (maxDurationMs !== void 0) {
3657
3684
  rules.push(createRunDurationRule({ maxDurationMs }));
3658
3685
  select.add("run.duration");
3659
3686
  }
3660
- const llm = config.checks?.llm;
3661
- if (llm?.allowedModels !== void 0 || llm?.maxTotalTokens !== void 0) {
3662
- rules.push(createLlmUsageRule(llm));
3687
+ const maxDepth = config.checks?.run?.maxDepth ?? evalConfig?.maxDepth;
3688
+ if (maxDepth !== void 0) {
3689
+ rules.push(createRunDepthRule({ maxDepth }));
3690
+ select.add("run.depth");
3691
+ }
3692
+ if (evalConfig?.maxRetries !== void 0) {
3693
+ rules.push(createToolFailureRule({ maxRetries: evalConfig.maxRetries }));
3694
+ select.add("tool.failures");
3695
+ }
3696
+ const allowedModels = config.checks?.llm?.allowedModels;
3697
+ const maxTotalTokens = config.checks?.llm?.maxTotalTokens ?? evalConfig?.maxTotalTokens;
3698
+ if (allowedModels !== void 0 || maxTotalTokens !== void 0) {
3699
+ rules.push(
3700
+ createLlmUsageRule({
3701
+ ...allowedModels !== void 0 ? { allowedModels } : {},
3702
+ ...maxTotalTokens !== void 0 ? { maxTotalTokens } : {}
3703
+ })
3704
+ );
3663
3705
  select.add("llm.usage");
3664
3706
  }
3665
3707
  if (select.has("outcome.status")) {
3666
3708
  rules.push(createObservedOutcomeRule({ failOn: ["failed"] }));
3667
3709
  }
3668
- return { rules, select: [...select] };
3710
+ const observationCount = suiteCase.expectedObservations?.length ?? 0;
3711
+ if (rules.length === 0 && observationCount === 0 && configDiagnostics.length === 0) {
3712
+ configDiagnostics.push(
3713
+ diagnostic3(
3714
+ "AI_SUITE_NO_ASSERTIONS",
3715
+ `Suite case "${suiteCase.id}" declares no effective checks, eval controls, or expected observations.`,
3716
+ "error",
3717
+ suiteCase.id
3718
+ )
3719
+ );
3720
+ }
3721
+ return {
3722
+ rules,
3723
+ select: [...select],
3724
+ observationCount,
3725
+ configDiagnostics
3726
+ };
3669
3727
  }
3670
3728
  function outcomesFromRead(read) {
3671
3729
  const persistedOutcomes = extractOutcomesFromPersistedEvents(
@@ -3746,8 +3804,20 @@ async function runSuiteCase(suiteCase, config, options) {
3746
3804
  ]
3747
3805
  };
3748
3806
  }
3749
- const { rules, select } = buildCaseRules(suiteCase, config);
3750
- const checkResult = rules.length > 0 ? runTraceChecks({ read }, { rules, select }) : {
3807
+ const compiled = buildCaseAssertions(suiteCase, config);
3808
+ if (compiled.configDiagnostics.length > 0) {
3809
+ return {
3810
+ id: suiteCase.id,
3811
+ status: "error",
3812
+ tracePath: resolved.tracePath,
3813
+ ...resolved.runId !== void 0 ? { runId: resolved.runId } : {},
3814
+ checkOk: false,
3815
+ ...config.eval?.requireSuccess === true ? { evalOk: false } : {},
3816
+ message: compiled.configDiagnostics.map((item) => item.message).join("; "),
3817
+ diagnostics: compiled.configDiagnostics
3818
+ };
3819
+ }
3820
+ const checkResult = compiled.rules.length > 0 ? runTraceChecks({ read }, { rules: compiled.rules, select: compiled.select }) : {
3751
3821
  ok: true,
3752
3822
  status: "pass",
3753
3823
  format: read.format,
@@ -3775,6 +3845,9 @@ async function runSuiteCase(suiteCase, config, options) {
3775
3845
  ];
3776
3846
  const checkOk = checkResult.ok;
3777
3847
  const observationsOk = observationResult.ok;
3848
+ const evalOk = config.eval?.requireSuccess === true ? checkResult.findings.every(
3849
+ (finding) => finding.ruleId !== "run.status" || finding.status !== "fail"
3850
+ ) && checkResult.diagnostics.every((item) => item.severity !== "error") : void 0;
3778
3851
  const ok = checkOk && observationsOk;
3779
3852
  const status = ok ? "pass" : checkResult.status === "error" ? "error" : "fail";
3780
3853
  return {
@@ -3783,6 +3856,7 @@ async function runSuiteCase(suiteCase, config, options) {
3783
3856
  tracePath: resolved.tracePath,
3784
3857
  ...resolved.runId !== void 0 ? { runId: resolved.runId } : {},
3785
3858
  checkOk,
3859
+ ...evalOk !== void 0 ? { evalOk } : {},
3786
3860
  observationsOk,
3787
3861
  diagnostics,
3788
3862
  ...ok ? {} : {