agent-inspect 6.31.8 → 6.31.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +6 -0
- package/README.md +1 -1
- package/package.json +1 -1
- package/packages/cli/dist/{chunk-6FE47XSF.mjs → chunk-SWUZRRDT.mjs} +136 -12
- package/packages/cli/dist/chunk-SWUZRRDT.mjs.map +1 -0
- package/packages/cli/dist/index.cjs +137 -11
- package/packages/cli/dist/index.cjs.map +1 -1
- package/packages/cli/dist/index.mjs +4 -4
- package/packages/cli/dist/index.mjs.map +1 -1
- package/packages/cli/dist/{src-NXAYHG34.mjs → src-XFIPYODW.mjs} +3 -3
- package/packages/cli/dist/{src-NXAYHG34.mjs.map → src-XFIPYODW.mjs.map} +1 -1
- package/packages/core/dist/advanced.cjs +178 -15
- package/packages/core/dist/advanced.cjs.map +1 -1
- package/packages/core/dist/advanced.d.cts +1 -1
- package/packages/core/dist/advanced.d.ts +1 -1
- package/packages/core/dist/advanced.mjs +85 -11
- package/packages/core/dist/advanced.mjs.map +1 -1
- package/packages/core/dist/{chunk-VQUB3WRI.mjs → chunk-3SPBELJB.mjs} +14 -3
- package/packages/core/dist/chunk-3SPBELJB.mjs.map +1 -0
- package/packages/core/dist/index.cjs +12 -1
- package/packages/core/dist/index.cjs.map +1 -1
- package/packages/core/dist/index.mjs +2 -2
- package/packages/cli/dist/chunk-6FE47XSF.mjs.map +0 -1
- package/packages/core/dist/chunk-VQUB3WRI.mjs.map +0 -1
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,11 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 6.31.9
|
|
4
|
+
|
|
5
|
+
### Patch Changes
|
|
6
|
+
|
|
7
|
+
- f71ec7e: Fail closed when suite cases declare no effective assertions or unknown selectors, wire eval.requireSuccess and related eval controls into real check rules, and share AsyncLocalStorage across packed CJS root and /advanced entrypoints so guarded wrappers see the active run context.
|
|
8
|
+
|
|
3
9
|
## 6.31.8
|
|
4
10
|
|
|
5
11
|
### Patch Changes
|
package/README.md
CHANGED
|
@@ -213,7 +213,7 @@ The root package is enough for custom capture, the CLI, checks, and Evidence wor
|
|
|
213
213
|
|
|
214
214
|
## Status and documentation
|
|
215
215
|
|
|
216
|
-
**Current published baseline:** **6.31.
|
|
216
|
+
**Current published baseline:** **6.31.9** · persisted schema `1.0` · Node.js `>=20` · MIT.
|
|
217
217
|
|
|
218
218
|
Legacy v0.1 and v0.2 traces remain readable. Check the npm badge and [changelog](CHANGELOG.md) for the current published version.
|
|
219
219
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "agent-inspect",
|
|
3
|
-
"version": "6.31.
|
|
3
|
+
"version": "6.31.9",
|
|
4
4
|
"license": "MIT",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"description": "Local evidence debugger and trajectory-test toolkit for TypeScript AI agents — execution trees, TraceContract checks, Evidence v2, and read-only MCP",
|
|
@@ -1627,7 +1627,18 @@ function extractCorrelationMetadata(record) {
|
|
|
1627
1627
|
}
|
|
1628
1628
|
|
|
1629
1629
|
// packages/core/src/context.ts
|
|
1630
|
-
|
|
1630
|
+
var GLOBAL_CONTEXT_ALS_KEY = "agent-inspect:execution-context-als:v1";
|
|
1631
|
+
function getSharedContextStorage() {
|
|
1632
|
+
const host = globalThis;
|
|
1633
|
+
const existing = host[GLOBAL_CONTEXT_ALS_KEY];
|
|
1634
|
+
if (existing instanceof AsyncLocalStorage) {
|
|
1635
|
+
return existing;
|
|
1636
|
+
}
|
|
1637
|
+
const created = new AsyncLocalStorage();
|
|
1638
|
+
host[GLOBAL_CONTEXT_ALS_KEY] = created;
|
|
1639
|
+
return created;
|
|
1640
|
+
}
|
|
1641
|
+
getSharedContextStorage();
|
|
1631
1642
|
|
|
1632
1643
|
// packages/core/src/terminal.ts
|
|
1633
1644
|
var TERMINAL_INDENT = " ";
|
|
@@ -8407,6 +8418,45 @@ function createToolOrderingRule(options) {
|
|
|
8407
8418
|
requireEndpoints: false
|
|
8408
8419
|
});
|
|
8409
8420
|
}
|
|
8421
|
+
function createToolFailureRule(options) {
|
|
8422
|
+
return {
|
|
8423
|
+
id: "tool.failures",
|
|
8424
|
+
category: "tool",
|
|
8425
|
+
defaultSeverity: "error",
|
|
8426
|
+
evaluate(context) {
|
|
8427
|
+
const tools = finishedEvents(context, "TOOL");
|
|
8428
|
+
const failures = tools.filter((event) => event.status === "error");
|
|
8429
|
+
const retries = tools.map((event) => ({ event, count: retryCount(event) })).filter((item) => item.count !== void 0);
|
|
8430
|
+
const findings = [];
|
|
8431
|
+
if (options.maxFailures !== void 0 && failures.length > options.maxFailures) {
|
|
8432
|
+
findings.push(
|
|
8433
|
+
failFinding(
|
|
8434
|
+
"tool.failures",
|
|
8435
|
+
`Tool failure count ${failures.length} exceeded ${options.maxFailures}.`,
|
|
8436
|
+
failures.map((event) => eventEvidence3(event)),
|
|
8437
|
+
{ maxFailures: options.maxFailures },
|
|
8438
|
+
failures.length
|
|
8439
|
+
)
|
|
8440
|
+
);
|
|
8441
|
+
}
|
|
8442
|
+
if (options.maxRetries !== void 0) {
|
|
8443
|
+
const excessiveRetries = retries.filter((item) => item.count > options.maxRetries);
|
|
8444
|
+
if (excessiveRetries.length > 0) {
|
|
8445
|
+
findings.push(
|
|
8446
|
+
failFinding(
|
|
8447
|
+
"tool.failures",
|
|
8448
|
+
`Tool retry count exceeded ${options.maxRetries}.`,
|
|
8449
|
+
excessiveRetries.map((item) => eventEvidence3(item.event, "attributes.retryCount")),
|
|
8450
|
+
{ maxRetries: options.maxRetries },
|
|
8451
|
+
excessiveRetries.map((item) => ({ tool: toolName(item.event), retries: item.count }))
|
|
8452
|
+
)
|
|
8453
|
+
);
|
|
8454
|
+
}
|
|
8455
|
+
}
|
|
8456
|
+
return findings;
|
|
8457
|
+
}
|
|
8458
|
+
};
|
|
8459
|
+
}
|
|
8410
8460
|
function createLlmUsageRule(options) {
|
|
8411
8461
|
return {
|
|
8412
8462
|
id: "llm.usage",
|
|
@@ -13649,21 +13699,48 @@ async function resolveSuiteCaseTrace(suiteCase, options) {
|
|
|
13649
13699
|
}
|
|
13650
13700
|
|
|
13651
13701
|
// packages/core/src/suite/run.ts
|
|
13702
|
+
var KNOWN_SUITE_SELECT_IDS = /* @__PURE__ */ new Set([
|
|
13703
|
+
"run.status",
|
|
13704
|
+
"run.duration",
|
|
13705
|
+
"run.depth",
|
|
13706
|
+
"outcome.status",
|
|
13707
|
+
"tool.usage",
|
|
13708
|
+
"tool.failures",
|
|
13709
|
+
"llm.usage"
|
|
13710
|
+
]);
|
|
13652
13711
|
function diagnostic5(code, message, severity = "error", caseId) {
|
|
13653
13712
|
return { code, message, severity, ...caseId !== void 0 ? { caseId } : {} };
|
|
13654
13713
|
}
|
|
13655
|
-
function
|
|
13714
|
+
function buildCaseAssertions(suiteCase, config) {
|
|
13656
13715
|
const rules = [];
|
|
13657
13716
|
const select = new Set(config.checks?.select ?? []);
|
|
13658
|
-
|
|
13717
|
+
const configDiagnostics = [];
|
|
13718
|
+
const evalConfig = config.eval;
|
|
13719
|
+
const declaredSelect = [...select];
|
|
13720
|
+
for (const id of declaredSelect) {
|
|
13721
|
+
if (!KNOWN_SUITE_SELECT_IDS.has(id)) {
|
|
13722
|
+
configDiagnostics.push(
|
|
13723
|
+
diagnostic5(
|
|
13724
|
+
"AI_SUITE_UNKNOWN_SELECTOR",
|
|
13725
|
+
`Unknown or unsupported suite check selector "${id}".`,
|
|
13726
|
+
"error",
|
|
13727
|
+
suiteCase.id
|
|
13728
|
+
)
|
|
13729
|
+
);
|
|
13730
|
+
}
|
|
13731
|
+
}
|
|
13732
|
+
if (select.has("run.status") || evalConfig?.requireSuccess === true) {
|
|
13659
13733
|
rules.push(createRunStatusRule());
|
|
13734
|
+
select.add("run.status");
|
|
13660
13735
|
}
|
|
13661
13736
|
const requiredTools = [
|
|
13662
13737
|
...config.checks?.tool?.required ?? [],
|
|
13738
|
+
...evalConfig?.requiredTools ?? [],
|
|
13663
13739
|
...suiteCase.requireTools ?? []
|
|
13664
13740
|
];
|
|
13665
13741
|
const forbiddenTools = [
|
|
13666
13742
|
...config.checks?.tool?.forbidden ?? [],
|
|
13743
|
+
...evalConfig?.forbiddenTools ?? [],
|
|
13667
13744
|
...suiteCase.forbidTools ?? []
|
|
13668
13745
|
];
|
|
13669
13746
|
if (requiredTools.length > 0 || forbiddenTools.length > 0) {
|
|
@@ -13675,20 +13752,51 @@ function buildCaseRules(suiteCase, config) {
|
|
|
13675
13752
|
);
|
|
13676
13753
|
select.add("tool.usage");
|
|
13677
13754
|
}
|
|
13678
|
-
const maxDurationMs = suiteCase.maxDurationMs ?? config.checks?.run?.maxDurationMs ??
|
|
13755
|
+
const maxDurationMs = suiteCase.maxDurationMs ?? config.checks?.run?.maxDurationMs ?? evalConfig?.maxDurationMs;
|
|
13679
13756
|
if (maxDurationMs !== void 0) {
|
|
13680
13757
|
rules.push(createRunDurationRule({ maxDurationMs }));
|
|
13681
13758
|
select.add("run.duration");
|
|
13682
13759
|
}
|
|
13683
|
-
const
|
|
13684
|
-
if (
|
|
13685
|
-
rules.push(
|
|
13760
|
+
const maxDepth = config.checks?.run?.maxDepth ?? evalConfig?.maxDepth;
|
|
13761
|
+
if (maxDepth !== void 0) {
|
|
13762
|
+
rules.push(createRunDepthRule({ maxDepth }));
|
|
13763
|
+
select.add("run.depth");
|
|
13764
|
+
}
|
|
13765
|
+
if (evalConfig?.maxRetries !== void 0) {
|
|
13766
|
+
rules.push(createToolFailureRule({ maxRetries: evalConfig.maxRetries }));
|
|
13767
|
+
select.add("tool.failures");
|
|
13768
|
+
}
|
|
13769
|
+
const allowedModels = config.checks?.llm?.allowedModels;
|
|
13770
|
+
const maxTotalTokens = config.checks?.llm?.maxTotalTokens ?? evalConfig?.maxTotalTokens;
|
|
13771
|
+
if (allowedModels !== void 0 || maxTotalTokens !== void 0) {
|
|
13772
|
+
rules.push(
|
|
13773
|
+
createLlmUsageRule({
|
|
13774
|
+
...allowedModels !== void 0 ? { allowedModels } : {},
|
|
13775
|
+
...maxTotalTokens !== void 0 ? { maxTotalTokens } : {}
|
|
13776
|
+
})
|
|
13777
|
+
);
|
|
13686
13778
|
select.add("llm.usage");
|
|
13687
13779
|
}
|
|
13688
13780
|
if (select.has("outcome.status")) {
|
|
13689
13781
|
rules.push(createObservedOutcomeRule({ failOn: ["failed"] }));
|
|
13690
13782
|
}
|
|
13691
|
-
|
|
13783
|
+
const observationCount = suiteCase.expectedObservations?.length ?? 0;
|
|
13784
|
+
if (rules.length === 0 && observationCount === 0 && configDiagnostics.length === 0) {
|
|
13785
|
+
configDiagnostics.push(
|
|
13786
|
+
diagnostic5(
|
|
13787
|
+
"AI_SUITE_NO_ASSERTIONS",
|
|
13788
|
+
`Suite case "${suiteCase.id}" declares no effective checks, eval controls, or expected observations.`,
|
|
13789
|
+
"error",
|
|
13790
|
+
suiteCase.id
|
|
13791
|
+
)
|
|
13792
|
+
);
|
|
13793
|
+
}
|
|
13794
|
+
return {
|
|
13795
|
+
rules,
|
|
13796
|
+
select: [...select],
|
|
13797
|
+
observationCount,
|
|
13798
|
+
configDiagnostics
|
|
13799
|
+
};
|
|
13692
13800
|
}
|
|
13693
13801
|
function outcomesFromRead(read) {
|
|
13694
13802
|
const persistedOutcomes = extractOutcomesFromPersistedEvents(
|
|
@@ -13769,8 +13877,20 @@ async function runSuiteCase(suiteCase, config, options) {
|
|
|
13769
13877
|
]
|
|
13770
13878
|
};
|
|
13771
13879
|
}
|
|
13772
|
-
const
|
|
13773
|
-
|
|
13880
|
+
const compiled = buildCaseAssertions(suiteCase, config);
|
|
13881
|
+
if (compiled.configDiagnostics.length > 0) {
|
|
13882
|
+
return {
|
|
13883
|
+
id: suiteCase.id,
|
|
13884
|
+
status: "error",
|
|
13885
|
+
tracePath: resolved.tracePath,
|
|
13886
|
+
...resolved.runId !== void 0 ? { runId: resolved.runId } : {},
|
|
13887
|
+
checkOk: false,
|
|
13888
|
+
...config.eval?.requireSuccess === true ? { evalOk: false } : {},
|
|
13889
|
+
message: compiled.configDiagnostics.map((item) => item.message).join("; "),
|
|
13890
|
+
diagnostics: compiled.configDiagnostics
|
|
13891
|
+
};
|
|
13892
|
+
}
|
|
13893
|
+
const checkResult = compiled.rules.length > 0 ? runTraceChecks({ read }, { rules: compiled.rules, select: compiled.select }) : {
|
|
13774
13894
|
ok: true,
|
|
13775
13895
|
status: "pass",
|
|
13776
13896
|
format: read.format,
|
|
@@ -13798,6 +13918,9 @@ async function runSuiteCase(suiteCase, config, options) {
|
|
|
13798
13918
|
];
|
|
13799
13919
|
const checkOk = checkResult.ok;
|
|
13800
13920
|
const observationsOk = observationResult.ok;
|
|
13921
|
+
const evalOk = config.eval?.requireSuccess === true ? checkResult.findings.every(
|
|
13922
|
+
(finding) => finding.ruleId !== "run.status" || finding.status !== "fail"
|
|
13923
|
+
) && checkResult.diagnostics.every((item) => item.severity !== "error") : void 0;
|
|
13801
13924
|
const ok = checkOk && observationsOk;
|
|
13802
13925
|
const status = ok ? "pass" : checkResult.status === "error" ? "error" : "fail";
|
|
13803
13926
|
return {
|
|
@@ -13806,6 +13929,7 @@ async function runSuiteCase(suiteCase, config, options) {
|
|
|
13806
13929
|
tracePath: resolved.tracePath,
|
|
13807
13930
|
...resolved.runId !== void 0 ? { runId: resolved.runId } : {},
|
|
13808
13931
|
checkOk,
|
|
13932
|
+
...evalOk !== void 0 ? { evalOk } : {},
|
|
13809
13933
|
observationsOk,
|
|
13810
13934
|
diagnostics,
|
|
13811
13935
|
...ok ? {} : {
|
|
@@ -14973,5 +15097,5 @@ function renderGateReport(result, options = {}) {
|
|
|
14973
15097
|
}
|
|
14974
15098
|
|
|
14975
15099
|
export { ATTRIBUTION_CONFIDENCES, COHORT_METRIC_IDS, DEFAULT_REDACT_KEYS, DEFAULT_SUITE_ARTIFACTS_DIR, EVIDENCE_FORMAT_VERSION, EVIDENCE_HTML_FILENAME, EVIDENCE_MANIFEST_FILENAME, Redactor, TraceDirectory, TraceReadError, TreeBuilder, aggregateBundleSafeStatus, aggregateSessionCheckResults, analyzeCohort, applyProfileMetadataCaps, assertBundlePathContained, assertEvidenceRelativePath, buildActivitySummary, buildBundleMetadata, buildBundleSummaryMarkdown, buildEvidenceCausalFailureViewHtml, buildEvidenceCiPackage, buildEvidenceCircuitViewHtml, buildEvidenceContractsViewHtml, buildEvidenceDiffViewHtml, buildEvidenceHtmlShell, buildEvidenceManifest, buildEvidenceOutcomesViewHtml, buildEvidenceProvenanceViewHtml, buildEvidenceSafetyViewHtml, buildEvidenceTimelineViewHtml, buildEvidenceToolsLlmViewHtml, buildEvidenceTreeViewHtml, buildLocalExplanation, buildPlaceholderArtifact, buildRunSummary, buildRunTimeline, buildRunWhatSummary, buildSessionIndex, buildTraceStats, buildZipArchive, bundleFailsOnSafety, bundleRunAssetRelativePath, collectTraceSchemaVersions, compactAttributes, createBaselineRegressionRule, createLlmUsageRule, createMaxStepDurationRule, createObservedOutcomeRule, createRequireCompletedRule, createRunDepthRule, createRunDurationRule, createRunStatusRule, createSafetyOversizedAttributeRule, createSafetyRawContentRule, createSafetyRedactionRule, createSafetySecretPatternRule, createStallDetectionRule, createStructureCycleRule, createStructureOrphanRule, createStructureParallelWidthRule, createStructureRelationshipRule, createToolUsageRule, defaultBundleOutputPath, defaultSuiteConfigTemplate, defineTraceContract, diffRuns, diffTraceEvents, enrichSessionRunRecord, escapeHtml, escapeMarkdown, evaluateTraceContract, extractMetadata, extractOutcomesFromTraceEvents, filterMetasBySessionScope, filterTraces, flattenTree, formatDuration2 as formatDuration, formatStepLabel, formatTimestamp, gateHasThresholds, getIndent, getTraceFilePath, inferEvidenceFileRole, isAgentInspectTrace, isAttributionConfidence, isCredentialSensitiveKey, isPersistedInspectEvent, loadSessionRunRecords, loadSuiteConfig, loadTraceMetadataList, manualTraceEventsToComparableRun, nanoid, normalizeBundleOutputPath, openTrace, parseCohortMetricList, parseDuration, parseDurationFilter, parseGateList, parseTraceJsonl, persistedInspectEventsToRunTrees, persistedInspectEventsToTraceEvents, projectLogicalEvents, renderActivitySummaryHuman, renderCohortReport, renderErrorLine, renderGateReport, renderObservedOutcomesHtml, renderObservedOutcomesMarkdown, renderRunDiff, renderRunWhat, renderStepLine, renderSuiteReport, renderTimeline, renderTraceStats, resolveBundleRunIds, resolveRedactionProfile, resolveSuiteTemplate, resolveTraceDir, runGate, runSuite, runTraceChecks, safeString, sanitizeBundleRunId, searchTraces, serializeEvidenceManifest, sha256Hex, stableJson, stringContainsHighConfidenceCredential, summarizeObservedOutcomes, summarizeSemanticParity, traceEventToPersistedInspectEvent, truncateName, truncateStringForProfile, validateEvent, validateSuiteConfig, verifyEvidenceDirectory, zeroKinds };
|
|
14976
|
-
//# sourceMappingURL=chunk-
|
|
14977
|
-
//# sourceMappingURL=chunk-
|
|
15100
|
+
//# sourceMappingURL=chunk-SWUZRRDT.mjs.map
|
|
15101
|
+
//# sourceMappingURL=chunk-SWUZRRDT.mjs.map
|