@tangle-network/agent-eval 0.137.0 → 0.138.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +48 -0
- package/README.md +33 -0
- package/dist/analyst/index.d.ts +473 -39
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +11 -593
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-PVtnfjvA.d.ts → analyze-runs-CPYxfPWT.d.ts} +5 -5
- package/dist/{analyze-runs-PVtnfjvA.d.ts.map → analyze-runs-CPYxfPWT.d.ts.map} +1 -1
- package/dist/{benchmark-YDrpumqB.js → benchmark-D8dkki-J.js} +299 -159
- package/dist/benchmark-D8dkki-J.js.map +1 -0
- package/dist/{benchmark-CHX4orG7.d.ts → benchmark-DlQgU_XI.d.ts} +67 -15
- package/dist/benchmark-DlQgU_XI.d.ts.map +1 -0
- package/dist/benchmark-command-CMqVqReF.js +4332 -0
- package/dist/benchmark-command-CMqVqReF.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-DCLkQOmc.js → benchmarks-BJ_xK5rQ.js} +4 -3
- package/dist/{benchmarks-DCLkQOmc.js.map → benchmarks-BJ_xK5rQ.js.map} +1 -1
- package/dist/campaign/index.d.ts +5 -5
- package/dist/campaign/index.js +3 -3
- package/dist/{campaign-lgObcHFC.js → campaign-BIBS-NHV.js} +16 -9
- package/dist/campaign-BIBS-NHV.js.map +1 -0
- package/dist/cli.js +9 -2
- package/dist/cli.js.map +1 -1
- package/dist/{client-C8L6h6Wf.d.ts → client-BwPKohkJ.d.ts} +4 -4
- package/dist/{client-C8L6h6Wf.d.ts.map → client-BwPKohkJ.d.ts.map} +1 -1
- package/dist/{completion-verifier-DSyRNVzU.d.ts → completion-verifier-B4-IMYcS.d.ts} +3 -3
- package/dist/{completion-verifier-DSyRNVzU.d.ts.map → completion-verifier-B4-IMYcS.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +10 -10
- package/dist/contract/index.js +8 -8
- package/dist/control.d.ts +2 -2
- package/dist/{cost-ledger-D2o6JOrL.d.ts → cost-ledger-B1D3COAc.d.ts} +5 -4
- package/dist/{cost-ledger-D2o6JOrL.d.ts.map → cost-ledger-B1D3COAc.d.ts.map} +1 -1
- package/dist/{cost-ledger-D-5_-dhi.js → cost-ledger-CHDLA0Ss.js} +90 -45
- package/dist/cost-ledger-CHDLA0Ss.js.map +1 -0
- package/dist/{default-registry-Dc5D_Loc.d.ts → default-registry-PUhIVRWz.d.ts} +18 -5
- package/dist/default-registry-PUhIVRWz.d.ts.map +1 -0
- package/dist/{default-registry-CLXbRt0f.js → default-registry-lp5R0lve.js} +1503 -258
- package/dist/default-registry-lp5R0lve.js.map +1 -0
- package/dist/{eval-campaign-CHqfLnff.js → eval-campaign-9MozgKL7.js} +2 -2
- package/dist/{eval-campaign-CHqfLnff.js.map → eval-campaign-9MozgKL7.js.map} +1 -1
- package/dist/exact-types-Dpw2LeHA.d.ts +234 -0
- package/dist/exact-types-Dpw2LeHA.d.ts.map +1 -0
- package/dist/{extract-usage-p-56bh8q.js → extract-usage-CS391dOE.js} +2 -2
- package/dist/{extract-usage-p-56bh8q.js.map → extract-usage-CS391dOE.js.map} +1 -1
- package/dist/{feedback-trajectory-N_F0PwHz.d.ts → feedback-trajectory-CoNep7rl.d.ts} +3 -2
- package/dist/feedback-trajectory-CoNep7rl.d.ts.map +1 -0
- package/dist/fuzz.d.ts +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-U3RHOShi.d.ts → index-B2-IxCMB.d.ts} +2 -2
- package/dist/{index-U3RHOShi.d.ts.map → index-B2-IxCMB.d.ts.map} +1 -1
- package/dist/{index-BnP1QJUv.d.ts → index-CjVYlVBK.d.ts} +5 -5
- package/dist/{index-BnP1QJUv.d.ts.map → index-CjVYlVBK.d.ts.map} +1 -1
- package/dist/{index-C-Pr4OWg.d.ts → index-D0cxAdaV.d.ts} +11 -10
- package/dist/index-D0cxAdaV.d.ts.map +1 -0
- package/dist/index-DEb46kc6.d.ts.map +1 -1
- package/dist/{index-DRNl6g_N.d.ts → index-sMN_hI4E.d.ts} +3 -3
- package/dist/{index-DRNl6g_N.d.ts.map → index-sMN_hI4E.d.ts.map} +1 -1
- package/dist/index.d.ts +24 -23
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +19 -353
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-B9ooYH_g.d.ts → insight-report-CXd8VBDR.d.ts} +4 -4
- package/dist/{insight-report-B9ooYH_g.d.ts.map → insight-report-CXd8VBDR.d.ts.map} +1 -1
- package/dist/{integrity-CKxosZ5Z.d.ts → integrity-B-MLFz0I.d.ts} +2 -2
- package/dist/{integrity-CKxosZ5Z.d.ts.map → integrity-B-MLFz0I.d.ts.map} +1 -1
- package/dist/ledger-core/index.js +1 -1
- package/dist/{ledger-core-t6sItivm.js → ledger-core-C0Yx1I14.js} +220 -27
- package/dist/ledger-core-C0Yx1I14.js.map +1 -0
- package/dist/{llm-client-DKB25jV8.js → llm-client-Cj3c7PEm.js} +5 -5
- package/dist/llm-client-Cj3c7PEm.js.map +1 -0
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{proposal-findings-DCawte-y.js → proposal-findings-2GIUo1et.js} +2 -68
- package/dist/proposal-findings-2GIUo1et.js.map +1 -0
- package/dist/{registry-BdM7SuTr.d.ts → registry-C4yJTza7.d.ts} +60 -6
- package/dist/registry-C4yJTza7.d.ts.map +1 -0
- package/dist/{release-report-CofgVNZt.d.ts → release-report-CoyvyLBs.d.ts} +3 -3
- package/dist/{release-report-CofgVNZt.d.ts.map → release-report-CoyvyLBs.d.ts.map} +1 -1
- package/dist/{replay-Bju0T8Ls.js → replay-Cb-4Vf0k.js} +8 -7
- package/dist/replay-Cb-4Vf0k.js.map +1 -0
- package/dist/{replay-K8FaC0CB.d.ts → replay-DbIYwso6.d.ts} +7 -7
- package/dist/{replay-K8FaC0CB.d.ts.map → replay-DbIYwso6.d.ts.map} +1 -1
- package/dist/reporting.d.ts +4 -4
- package/dist/{researcher-Da0Wj-bt.d.ts → researcher-BCeOEjtR.d.ts} +5 -5
- package/dist/{researcher-Da0Wj-bt.d.ts.map → researcher-BCeOEjtR.d.ts.map} +1 -1
- package/dist/{reward-hacking-CQ3hTCO3.d.ts → reward-hacking-sE2l_NV6.d.ts} +2 -2
- package/dist/{reward-hacking-CQ3hTCO3.d.ts.map → reward-hacking-sE2l_NV6.d.ts.map} +1 -1
- package/dist/rl.d.ts +5 -5
- package/dist/rl.js +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/{rubric-predictive-validity-C4sztLR3.d.ts → rubric-predictive-validity-w2klGv1u.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-C4sztLR3.d.ts.map → rubric-predictive-validity-w2klGv1u.d.ts.map} +1 -1
- package/dist/{run-evidence-BDIircdA.d.ts → run-evidence-CbE0A8Xg.d.ts} +3 -3
- package/dist/{run-evidence-BDIircdA.d.ts.map → run-evidence-CbE0A8Xg.d.ts.map} +1 -1
- package/dist/{run-record-BPCa2rQ8.d.ts → run-record-DwHMk1Ai.d.ts} +2 -2
- package/dist/{run-record-BPCa2rQ8.d.ts.map → run-record-DwHMk1Ai.d.ts.map} +1 -1
- package/dist/{semantic-concept-judge-Bz64IckK.js → semantic-concept-judge-DYXDPZW0.js} +11 -5
- package/dist/semantic-concept-judge-DYXDPZW0.js.map +1 -0
- package/dist/{server-KjXZZUDX.js → server-DLEvyW2z.js} +3 -3
- package/dist/{server-KjXZZUDX.js.map → server-DLEvyW2z.js.map} +1 -1
- package/dist/single-run-lock-D_bS5xhj.js +318 -0
- package/dist/single-run-lock-D_bS5xhj.js.map +1 -0
- package/dist/{skill-usage-CFDLLlhF.d.ts → skill-usage-Bv3G4VkA.d.ts} +18 -8
- package/dist/skill-usage-Bv3G4VkA.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-f4o9sUT4.js → skillopt-optimization-method-CjKMZy0d.js} +7 -182
- package/dist/skillopt-optimization-method-CjKMZy0d.js.map +1 -0
- package/dist/{skillopt-optimization-method-BpbnlvAZ.d.ts → skillopt-optimization-method-CzfnA8O-.d.ts} +10 -10
- package/dist/{skillopt-optimization-method-BpbnlvAZ.d.ts.map → skillopt-optimization-method-CzfnA8O-.d.ts.map} +1 -1
- package/dist/{statistics-_7P642CN.d.ts → statistics-mf70aXKp.d.ts} +2 -2
- package/dist/{statistics-_7P642CN.d.ts.map → statistics-mf70aXKp.d.ts.map} +1 -1
- package/dist/{tools-DZk2Jn64.js → store-otlp-BenKynPE.js} +4 -192
- package/dist/store-otlp-BenKynPE.js.map +1 -0
- package/dist/{summary-report-DHipz9Kx.d.ts → summary-report-BKinV4yD.d.ts} +3 -3
- package/dist/{summary-report-DHipz9Kx.d.ts.map → summary-report-BKinV4yD.d.ts.map} +1 -1
- package/dist/tools-DZGdROtG.js +255 -0
- package/dist/tools-DZGdROtG.js.map +1 -0
- package/dist/traces.d.ts +5 -5
- package/dist/traces.js +4 -3
- package/dist/{types-CTvKfr5F.d.ts → types-5q2T25iW.d.ts} +2 -2
- package/dist/{types-CTvKfr5F.d.ts.map → types-5q2T25iW.d.ts.map} +1 -1
- package/dist/{types-CKswbJGO.d.ts → types-BtJhn8v6.d.ts} +4 -4
- package/dist/{types-CKswbJGO.d.ts.map → types-BtJhn8v6.d.ts.map} +1 -1
- package/dist/{types-CTGbIm57.d.ts → types-zFYez3PK.d.ts} +5 -5
- package/dist/{types-CTGbIm57.d.ts.map → types-zFYez3PK.d.ts.map} +1 -1
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/docs/trace-analysis.md +123 -3
- package/package.json +5 -3
- package/dist/benchmark-CHX4orG7.d.ts.map +0 -1
- package/dist/benchmark-YDrpumqB.js.map +0 -1
- package/dist/campaign-lgObcHFC.js.map +0 -1
- package/dist/concurrency-MUjT7VjM.js +0 -109
- package/dist/concurrency-MUjT7VjM.js.map +0 -1
- package/dist/cost-ledger-D-5_-dhi.js.map +0 -1
- package/dist/default-registry-CLXbRt0f.js.map +0 -1
- package/dist/default-registry-Dc5D_Loc.d.ts.map +0 -1
- package/dist/feedback-trajectory-N_F0PwHz.d.ts.map +0 -1
- package/dist/index-C-Pr4OWg.d.ts.map +0 -1
- package/dist/ledger-core-t6sItivm.js.map +0 -1
- package/dist/llm-client-DKB25jV8.js.map +0 -1
- package/dist/proposal-findings-DCawte-y.js.map +0 -1
- package/dist/registry-BdM7SuTr.d.ts.map +0 -1
- package/dist/replay-Bju0T8Ls.js.map +0 -1
- package/dist/semantic-concept-judge-Bz64IckK.js.map +0 -1
- package/dist/skill-usage-CFDLLlhF.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-f4o9sUT4.js.map +0 -1
- package/dist/tools-DZk2Jn64.js.map +0 -1
|
@@ -0,0 +1,318 @@
|
|
|
1
|
+
import { c as ValidationError } from "./errors-D-LKuDhb.js";
|
|
2
|
+
import { i as CostLedger } from "./cost-ledger-CHDLA0Ss.js";
|
|
3
|
+
import { a as appendLedgerLine, d as tryAcquireAtomicFileLock, o as tryWithLedgerFileLock, u as probeAtomicFileLock } from "./ledger-core-C0Yx1I14.js";
|
|
4
|
+
import { createRequire } from "node:module";
|
|
5
|
+
import { join } from "node:path";
|
|
6
|
+
import { createHash } from "node:crypto";
|
|
7
|
+
//#region src/analyst/types.ts
|
|
8
|
+
/**
|
|
9
|
+
* Analyst contract — the missing orchestration layer over agent-eval's
|
|
10
|
+
* existing analyzers (analyzeTraces, MultiLayerVerifier, RunCritic,
|
|
11
|
+
* SemanticConceptJudge, JudgeFn, ...).
|
|
12
|
+
*
|
|
13
|
+
* Each existing primitive returns its own output shape. The Analyst
|
|
14
|
+
* contract is the single envelope every primitive lifts into, so a
|
|
15
|
+
* registry can run N analysts against a run and a single renderer can
|
|
16
|
+
* compose findings without knowing which analyzer produced them.
|
|
17
|
+
*
|
|
18
|
+
* The contract is intentionally domain-agnostic: nothing here knows
|
|
19
|
+
* about code, voice, RAG, or any particular agent stack. Analysts
|
|
20
|
+
* declare what INPUT KIND they need (a trace store, an artifact dir,
|
|
21
|
+
* a RunRecord, a JudgeInput, or `custom`), and the registry routes
|
|
22
|
+
* the matching input from `AnalystRunInputs`.
|
|
23
|
+
*/
|
|
24
|
+
/**
|
|
25
|
+
* Compute the stable finding_id from the identity-defining fields.
|
|
26
|
+
* Default implementation hashes {analyst_id, area, subject, normalized claim}.
|
|
27
|
+
* Analysts that emit findings whose claim text varies per run (timestamps,
|
|
28
|
+
* counts) SHOULD either: (a) pass an explicit `id_basis` to fix the hash,
|
|
29
|
+
* or (b) move the variable part into `rationale`/`metadata` and keep the
|
|
30
|
+
* `claim` static.
|
|
31
|
+
*/
|
|
32
|
+
function computeFindingId(input) {
|
|
33
|
+
const basis = JSON.stringify({
|
|
34
|
+
a: input.analyst_id,
|
|
35
|
+
r: input.area,
|
|
36
|
+
s: input.subject ?? "",
|
|
37
|
+
c: normalizeClaim(input.id_basis ?? input.claim)
|
|
38
|
+
});
|
|
39
|
+
return `f_${createHash("sha256").update(basis).digest("hex").slice(0, 20)}`;
|
|
40
|
+
}
|
|
41
|
+
function normalizeClaim(c) {
|
|
42
|
+
return c.toLowerCase().replace(/\s+/g, " ").replace(/[.!?;:,]+$/g, "").trim();
|
|
43
|
+
}
|
|
44
|
+
/**
|
|
45
|
+
* Convenience factory: produce a fully-formed AnalystFinding with the
|
|
46
|
+
* id computed automatically. Analyst code stays terse.
|
|
47
|
+
*/
|
|
48
|
+
function makeFinding(init) {
|
|
49
|
+
const { id_basis, produced_at, ...rest } = init;
|
|
50
|
+
return {
|
|
51
|
+
schema_version: "1.0.0",
|
|
52
|
+
finding_id: computeFindingId({
|
|
53
|
+
analyst_id: rest.analyst_id,
|
|
54
|
+
area: rest.area,
|
|
55
|
+
subject: rest.subject,
|
|
56
|
+
claim: rest.claim,
|
|
57
|
+
id_basis
|
|
58
|
+
}),
|
|
59
|
+
produced_at: produced_at ?? (/* @__PURE__ */ new Date()).toISOString(),
|
|
60
|
+
...rest
|
|
61
|
+
};
|
|
62
|
+
}
|
|
63
|
+
/** Build a finding whose source is explicitly allowed during candidate generation. */
|
|
64
|
+
function makeProposalFinding(init) {
|
|
65
|
+
const { proposal_origin, ...finding } = init;
|
|
66
|
+
return {
|
|
67
|
+
...makeFinding(finding),
|
|
68
|
+
proposal_origin
|
|
69
|
+
};
|
|
70
|
+
}
|
|
71
|
+
/** Convert one ledger channel's complete call set into one analyst receipt. */
|
|
72
|
+
function usageReceiptFromCostLedger(ledger, filter = "analyst") {
|
|
73
|
+
const resolvedFilter = typeof filter === "string" ? { channel: filter } : filter;
|
|
74
|
+
const summary = ledger.summary(resolvedFilter);
|
|
75
|
+
const receipts = ledger.list(resolvedFilter);
|
|
76
|
+
const hasReasoningUsage = receipts.some((receipt) => receipt.reasoningTokens !== void 0);
|
|
77
|
+
const hasCacheWriteUsage = receipts.some((receipt) => receipt.cacheWriteTokens !== void 0);
|
|
78
|
+
const cost = summary.costProvenance;
|
|
79
|
+
return {
|
|
80
|
+
calls: summary.totalCalls + summary.pendingCalls,
|
|
81
|
+
tokens: summary.usageComplete ? {
|
|
82
|
+
input: summary.inputTokens,
|
|
83
|
+
output: summary.outputTokens,
|
|
84
|
+
...hasReasoningUsage ? { reasoning: summary.reasoningTokens ?? 0 } : {},
|
|
85
|
+
...summary.cachedTokens > 0 ? { cached: summary.cachedTokens } : {},
|
|
86
|
+
...hasCacheWriteUsage ? { cacheWrite: summary.cacheWriteTokens ?? 0 } : {}
|
|
87
|
+
} : null,
|
|
88
|
+
cost,
|
|
89
|
+
...cost.kind === "uncaptured" ? { knownCostUsd: summary.totalCostUsd } : {}
|
|
90
|
+
};
|
|
91
|
+
}
|
|
92
|
+
/** Wait a bounded time for late provider receipts, then take one immutable snapshot. */
|
|
93
|
+
async function settleUsageReceiptFromCostLedger(ledger, options = {}) {
|
|
94
|
+
const { timeoutMs: requestedTimeoutMs, ...requestedFilter } = options;
|
|
95
|
+
const filter = {
|
|
96
|
+
channel: requestedFilter.channel ?? "analyst",
|
|
97
|
+
...requestedFilter.phase === void 0 ? {} : { phase: requestedFilter.phase },
|
|
98
|
+
...requestedFilter.tags === void 0 ? {} : { tags: requestedFilter.tags }
|
|
99
|
+
};
|
|
100
|
+
const timeoutMs = validateUsageSettlementTimeout(requestedTimeoutMs);
|
|
101
|
+
const waitResult = ledger.summary(filter).pendingCalls === 0 ? true : ledger.waitForIdle ? await ledger.waitForIdle({ timeoutMs }) : false;
|
|
102
|
+
const pendingCalls = ledger.summary(filter).pendingCalls;
|
|
103
|
+
return {
|
|
104
|
+
settled: waitResult && pendingCalls === 0,
|
|
105
|
+
pendingCalls,
|
|
106
|
+
receipt: usageReceiptFromCostLedger(ledger, filter)
|
|
107
|
+
};
|
|
108
|
+
}
|
|
109
|
+
function validateUsageSettlementTimeout(timeoutMs) {
|
|
110
|
+
const resolved = timeoutMs ?? 5e3;
|
|
111
|
+
if (!Number.isSafeInteger(resolved) || resolved < 0 || resolved > 2147483647) throw new TypeError("settlementTimeoutMs must be a non-negative safe integer no greater than 2147483647");
|
|
112
|
+
return resolved;
|
|
113
|
+
}
|
|
114
|
+
function assertValidAnalystUsageReceipt(receipt, context = "AnalystContext.recordUsage") {
|
|
115
|
+
if (receipt.calls !== null && (!Number.isSafeInteger(receipt.calls) || receipt.calls < 0)) throw new Error(`${context}: calls must be a non-negative safe integer or null`);
|
|
116
|
+
if (receipt.tokens) {
|
|
117
|
+
assertNonNegativeSafeInteger(receipt.tokens.input, "tokens.input", context);
|
|
118
|
+
assertNonNegativeSafeInteger(receipt.tokens.output, "tokens.output", context);
|
|
119
|
+
if (receipt.tokens.reasoning !== void 0) {
|
|
120
|
+
assertNonNegativeSafeInteger(receipt.tokens.reasoning, "tokens.reasoning", context);
|
|
121
|
+
if (receipt.tokens.reasoning > receipt.tokens.output) throw new Error(`${context}: tokens.reasoning must not exceed tokens.output`);
|
|
122
|
+
}
|
|
123
|
+
if (receipt.tokens.cached !== void 0) assertNonNegativeSafeInteger(receipt.tokens.cached, "tokens.cached", context);
|
|
124
|
+
if (receipt.tokens.cacheWrite !== void 0) assertNonNegativeSafeInteger(receipt.tokens.cacheWrite, "tokens.cacheWrite", context);
|
|
125
|
+
}
|
|
126
|
+
if (receipt.cost.kind !== "uncaptured") assertNonNegativeFinite(receipt.cost.usd, "cost.usd", context);
|
|
127
|
+
else if (receipt.cost.usd !== null) throw new Error(`${context}: uncaptured cost.usd must be null`);
|
|
128
|
+
if (receipt.knownCostUsd !== void 0) assertNonNegativeFinite(receipt.knownCostUsd, "knownCostUsd", context);
|
|
129
|
+
}
|
|
130
|
+
function assertNonNegativeSafeInteger(value, field, context) {
|
|
131
|
+
if (!Number.isSafeInteger(value) || value < 0) throw new Error(`${context}: ${field} must be a non-negative safe integer`);
|
|
132
|
+
}
|
|
133
|
+
function assertNonNegativeFinite(value, field, context) {
|
|
134
|
+
if (!Number.isFinite(value) || value < 0) throw new Error(`${context}: ${field} must be a non-negative finite number`);
|
|
135
|
+
}
|
|
136
|
+
//#endregion
|
|
137
|
+
//#region src/campaign/search-ledger-errors.ts
|
|
138
|
+
/** Base error for invalid search-ledger input or operations. */
|
|
139
|
+
var SearchLedgerError = class extends ValidationError {};
|
|
140
|
+
/** Error raised when durable search-ledger data fails an integrity check. */
|
|
141
|
+
var SearchLedgerIntegrityError = class extends SearchLedgerError {};
|
|
142
|
+
/** Error raised when an event identifier is reused with different content. */
|
|
143
|
+
var SearchLedgerConflictError = class extends SearchLedgerError {};
|
|
144
|
+
//#endregion
|
|
145
|
+
//#region src/campaign/search-ledger-file.ts
|
|
146
|
+
/** Campaign binding of the generic journal file layer to search-ledger errors. */
|
|
147
|
+
const SEARCH_LEDGER_FILE_CONTEXT = {
|
|
148
|
+
subject: "search ledger",
|
|
149
|
+
integrityError: (message, options) => new SearchLedgerIntegrityError(message, options)
|
|
150
|
+
};
|
|
151
|
+
function appendSearchLedgerLine(path, line) {
|
|
152
|
+
appendLedgerLine(path, line, SEARCH_LEDGER_FILE_CONTEXT);
|
|
153
|
+
}
|
|
154
|
+
function tryWithSearchLedgerFileLock(ledgerPath, run) {
|
|
155
|
+
return tryWithLedgerFileLock(ledgerPath, SEARCH_LEDGER_FILE_CONTEXT, run);
|
|
156
|
+
}
|
|
157
|
+
//#endregion
|
|
158
|
+
//#region src/campaign/storage.ts
|
|
159
|
+
/** Node-filesystem storage — the default. Lazily requires `node:fs` so the
|
|
160
|
+
* module imports cleanly in non-Node runtimes (where the caller passes
|
|
161
|
+
* `inMemoryCampaignStorage` instead and never constructs this).
|
|
162
|
+
*
|
|
163
|
+
* `createRequire(import.meta.url)` is the ESM-native lazy require — a bare
|
|
164
|
+
* `require` is a ReferenceError under `"type": "module"`, which is exactly
|
|
165
|
+
* the shape this package publishes. */
|
|
166
|
+
function fsCampaignStorage() {
|
|
167
|
+
const { existsSync, mkdirSync, readFileSync, statSync, writeFileSync } = createRequire(import.meta.url)("node:fs");
|
|
168
|
+
return {
|
|
169
|
+
ensureDir(dir) {
|
|
170
|
+
if (!existsSync(dir)) mkdirSync(dir, { recursive: true });
|
|
171
|
+
},
|
|
172
|
+
exists(path) {
|
|
173
|
+
return existsSync(path);
|
|
174
|
+
},
|
|
175
|
+
read(path) {
|
|
176
|
+
try {
|
|
177
|
+
return readFileSync(path, "utf8");
|
|
178
|
+
} catch {
|
|
179
|
+
return;
|
|
180
|
+
}
|
|
181
|
+
},
|
|
182
|
+
write(path, content) {
|
|
183
|
+
writeFileSync(path, content);
|
|
184
|
+
},
|
|
185
|
+
append(path, content, expectedBytes) {
|
|
186
|
+
const result = tryWithSearchLedgerFileLock(path, () => {
|
|
187
|
+
let actualBytes = 0;
|
|
188
|
+
try {
|
|
189
|
+
actualBytes = statSync(path).size;
|
|
190
|
+
} catch (error) {
|
|
191
|
+
if (error.code !== "ENOENT") throw error;
|
|
192
|
+
}
|
|
193
|
+
if (actualBytes !== expectedBytes) return void 0;
|
|
194
|
+
appendSearchLedgerLine(path, content);
|
|
195
|
+
return expectedBytes + Buffer.byteLength(content);
|
|
196
|
+
});
|
|
197
|
+
return result.acquired ? result.value : void 0;
|
|
198
|
+
}
|
|
199
|
+
};
|
|
200
|
+
}
|
|
201
|
+
/** In-memory storage for filesystem-less runtimes. Artifacts + trace spans
|
|
202
|
+
* live in a `Map` for the duration of the run; the `CampaignResult` is
|
|
203
|
+
* fully populated, but nothing is persisted to disk. */
|
|
204
|
+
function inMemoryCampaignStorage() {
|
|
205
|
+
const files = /* @__PURE__ */ new Map();
|
|
206
|
+
const dirs = /* @__PURE__ */ new Set();
|
|
207
|
+
return {
|
|
208
|
+
ensureDir(dir) {
|
|
209
|
+
dirs.add(dir);
|
|
210
|
+
},
|
|
211
|
+
exists(path) {
|
|
212
|
+
return files.has(path) || dirs.has(path);
|
|
213
|
+
},
|
|
214
|
+
read(path) {
|
|
215
|
+
const value = files.get(path);
|
|
216
|
+
if (value === void 0) return void 0;
|
|
217
|
+
return typeof value === "string" ? value : new TextDecoder().decode(value);
|
|
218
|
+
},
|
|
219
|
+
write(path, content) {
|
|
220
|
+
files.set(path, content);
|
|
221
|
+
},
|
|
222
|
+
append(path, content, expectedBytes) {
|
|
223
|
+
const current = files.get(path);
|
|
224
|
+
const currentText = current === void 0 ? "" : typeof current === "string" ? current : new TextDecoder().decode(current);
|
|
225
|
+
const currentBytes = new TextEncoder().encode(currentText).byteLength;
|
|
226
|
+
if (currentBytes !== expectedBytes) return void 0;
|
|
227
|
+
files.set(path, `${currentText}${content}`);
|
|
228
|
+
return currentBytes + new TextEncoder().encode(content).byteLength;
|
|
229
|
+
}
|
|
230
|
+
};
|
|
231
|
+
}
|
|
232
|
+
/** Open the durable spend account stored beside a logical run. */
|
|
233
|
+
function createRunCostLedger(input) {
|
|
234
|
+
const path = join(input.runDir, "cost-ledger.jsonl");
|
|
235
|
+
input.storage.ensureDir(input.runDir);
|
|
236
|
+
return new CostLedger({
|
|
237
|
+
costCeilingUsd: input.costCeilingUsd,
|
|
238
|
+
persistence: {
|
|
239
|
+
read: () => {
|
|
240
|
+
const stored = input.storage.read(path);
|
|
241
|
+
if (stored === void 0 && input.storage.exists(path)) throw new Error(`CostLedger: cannot read existing event log '${path}'`);
|
|
242
|
+
const events = stored ?? "";
|
|
243
|
+
return {
|
|
244
|
+
revision: String(new TextEncoder().encode(events).byteLength),
|
|
245
|
+
events
|
|
246
|
+
};
|
|
247
|
+
},
|
|
248
|
+
append: (expectedRevision, event) => {
|
|
249
|
+
const expectedBytes = Number(expectedRevision);
|
|
250
|
+
if (!Number.isSafeInteger(expectedBytes) || expectedBytes < 0) throw new Error(`CostLedger: invalid storage revision '${expectedRevision}'`);
|
|
251
|
+
const next = input.storage.append(path, event, expectedBytes);
|
|
252
|
+
return next === void 0 ? void 0 : String(next);
|
|
253
|
+
}
|
|
254
|
+
}
|
|
255
|
+
});
|
|
256
|
+
}
|
|
257
|
+
//#endregion
|
|
258
|
+
//#region src/campaign/single-run-lock.ts
|
|
259
|
+
/**
|
|
260
|
+
* Single-run lock for evaluations that share one mutable environment.
|
|
261
|
+
*
|
|
262
|
+
* Two concurrent runs against a shared stateful gym silently corrupt each
|
|
263
|
+
* other: each resets/mutates environment state mid-cell of the other, and
|
|
264
|
+
* every score from both becomes garbage that LOOKS like worker variance
|
|
265
|
+
* (agent-lab R357 burned hours on flip-flopping scores before tracing them
|
|
266
|
+
* to exactly this). The fix is a pid lockfile: refuse to start while a live
|
|
267
|
+
* holder exists, reclaim stale locks whose pid is gone, release only if the
|
|
268
|
+
* lock is still ours.
|
|
269
|
+
*
|
|
270
|
+
* `alsoCheck` exists because independent runners can guard the same shared
|
|
271
|
+
* resource with differently named lockfiles; a runner must respect all of
|
|
272
|
+
* them even though it writes only its own.
|
|
273
|
+
*/
|
|
274
|
+
function assertAvailable(path) {
|
|
275
|
+
const unavailable = probeAtomicFileLock({ lockPath: path });
|
|
276
|
+
if (unavailable) throw unavailableError(path, unavailable);
|
|
277
|
+
}
|
|
278
|
+
function unavailableError(path, unavailable) {
|
|
279
|
+
if (unavailable.reason === "recovery") return /* @__PURE__ */ new Error(`single-run lock recovery is already in progress (${path}); refusing a concurrent run`);
|
|
280
|
+
return /* @__PURE__ */ new Error(`single-run lock held by live pid ${unavailable.holder.pid} (${path}); refusing a concurrent run on the shared resource`);
|
|
281
|
+
}
|
|
282
|
+
/**
|
|
283
|
+
* Acquire the lock or throw naming the live holder. A stale lock (holder pid
|
|
284
|
+
* no longer running) is reclaimed by one contender. An interrupted reclaim
|
|
285
|
+
* leaves a marker that fails closed instead of admitting overlapping runs.
|
|
286
|
+
*/
|
|
287
|
+
function acquireSingleRunLock(opts) {
|
|
288
|
+
const pid = opts.pid ?? process.pid;
|
|
289
|
+
if (!Number.isSafeInteger(pid) || pid <= 0) throw new Error("single-run lock pid must be a positive integer");
|
|
290
|
+
for (const path of opts.alsoCheck ?? []) assertAvailable(path);
|
|
291
|
+
const acquisition = tryAcquireAtomicFileLock({
|
|
292
|
+
lockPath: opts.lockPath,
|
|
293
|
+
pid
|
|
294
|
+
});
|
|
295
|
+
if (!acquisition.acquired) throw unavailableError(opts.lockPath, acquisition);
|
|
296
|
+
const releaseOnExit = opts.releaseOnExit ?? true;
|
|
297
|
+
let released = false;
|
|
298
|
+
const release = () => {
|
|
299
|
+
if (released) return;
|
|
300
|
+
released = true;
|
|
301
|
+
if (releaseOnExit) process.off("exit", release);
|
|
302
|
+
try {
|
|
303
|
+
acquisition.lock.release();
|
|
304
|
+
} catch {}
|
|
305
|
+
};
|
|
306
|
+
try {
|
|
307
|
+
for (const path of opts.alsoCheck ?? []) assertAvailable(path);
|
|
308
|
+
} catch (error) {
|
|
309
|
+
release();
|
|
310
|
+
throw error;
|
|
311
|
+
}
|
|
312
|
+
if (releaseOnExit) process.on("exit", release);
|
|
313
|
+
return { release };
|
|
314
|
+
}
|
|
315
|
+
//#endregion
|
|
316
|
+
export { SEARCH_LEDGER_FILE_CONTEXT as a, SearchLedgerIntegrityError as c, usageReceiptFromCostLedger as d, validateUsageSettlementTimeout as f, makeProposalFinding as h, inMemoryCampaignStorage as i, assertValidAnalystUsageReceipt as l, makeFinding as m, createRunCostLedger as n, SearchLedgerConflictError as o, computeFindingId as p, fsCampaignStorage as r, SearchLedgerError as s, acquireSingleRunLock as t, settleUsageReceiptFromCostLedger as u };
|
|
317
|
+
|
|
318
|
+
//# sourceMappingURL=single-run-lock-D_bS5xhj.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"single-run-lock-D_bS5xhj.js","names":[],"sources":["../src/analyst/types.ts","../src/analyst/usage-receipt.ts","../src/campaign/search-ledger-errors.ts","../src/campaign/search-ledger-file.ts","../src/campaign/storage.ts","../src/campaign/single-run-lock.ts"],"sourcesContent":["/**\n * Analyst contract — the missing orchestration layer over agent-eval's\n * existing analyzers (analyzeTraces, MultiLayerVerifier, RunCritic,\n * SemanticConceptJudge, JudgeFn, ...).\n *\n * Each existing primitive returns its own output shape. The Analyst\n * contract is the single envelope every primitive lifts into, so a\n * registry can run N analysts against a run and a single renderer can\n * compose findings without knowing which analyzer produced them.\n *\n * The contract is intentionally domain-agnostic: nothing here knows\n * about code, voice, RAG, or any particular agent stack. Analysts\n * declare what INPUT KIND they need (a trace store, an artifact dir,\n * a RunRecord, a JudgeInput, or `custom`), and the registry routes\n * the matching input from `AnalystRunInputs`.\n */\n\nimport { createHash } from 'node:crypto'\nimport type { CostLedgerHandle } from '../cost-ledger'\nimport type { RunCostProvenance, RunRecord, RunTokenUsage } from '../run-record'\nimport type { TraceAnalysisStore } from '../trace-analyst/store'\nimport type { JudgeInput } from '../types'\nimport type { ChatClient } from './chat-client'\n\n/**\n * Unified envelope every analyst emits. Schema-versioned so renderers\n * and time-series diffs survive future field additions.\n */\nexport interface AnalystFinding {\n schema_version: '1.0.0'\n /**\n * Stable hash over identity-defining fields (analyst_id + canonical\n * claim + area + optional subject). Two findings from two runs that\n * \"are the same finding\" share this id — that's what `diffFindings`\n * uses to compute appeared/disappeared sets across runs.\n */\n finding_id: string\n analyst_id: string\n produced_at: string\n severity: AnalystSeverity\n /**\n * Coarse classification. Renderers group by this. Free-form so\n * domain-specific analysts can introduce categories without a\n * schema change ('agent-reasoning', 'verification', 'cost',\n * 'tool-use', 'safety', 'latency', 'data-quality', ...).\n */\n area: string\n claim: string\n rationale?: string\n evidence_refs: EvidenceRef[]\n recommended_action?: string\n validation_plan?: string\n /** 0..1 — the analyst's own confidence. Not calibrated across analysts. */\n confidence: number\n /**\n * Optional subject the finding is about — leaf id, agent id, request\n * id. Included in finding_id when present so per-subject findings\n * diff cleanly across runs.\n */\n subject?: string\n /** True when this finding was lifted from a judge result rather than observed\n * directly in a trace or artifact. Descriptive only: proposal access is\n * controlled by `ProposalFinding.proposal_origin`. */\n derived_from_judge?: boolean\n /** Analyst-private extras; renderers ignore unless they know the analyst. */\n metadata?: Record<string, unknown>\n}\n\nexport type AnalystSeverity = 'critical' | 'high' | 'medium' | 'low' | 'info'\n\n/** Data sources that candidate generation may intentionally learn from. */\nexport type ProposalFindingOrigin = 'search' | 'production'\n\n/** A finding explicitly admitted as candidate-generation input. */\nexport type ProposalFinding = AnalystFinding & {\n readonly proposal_origin: ProposalFindingOrigin\n}\n\nexport interface EvidenceRef {\n /**\n * Where the evidence lives. `span` and `event` refer to OTLP trace\n * elements; `artifact` to a file inside the run's artifact tree;\n * `finding` to another AnalystFinding (cross-analyst chaining);\n * `metric` to a named scalar reading the renderer knows how to read.\n */\n kind: 'span' | 'event' | 'artifact' | 'finding' | 'metric'\n uri: string\n excerpt?: string\n}\n\n// ── Analyst contract ─────────────────────────────────────────────────\n\n/**\n * The discriminator the registry uses to pass the right input.\n * `custom` is the escape hatch — analysts that need something else\n * (e.g. an embedding cache, a partner SDK handle) read it from\n * `AnalystRunInputs.custom[<analyst id>]`.\n */\nexport type AnalystInputKind =\n | 'trace-store'\n | 'artifact-dir'\n | 'run-record'\n | 'judge-input'\n | 'custom'\n\nexport interface AnalystCost {\n /** `deterministic` analysts MUST NOT call the LLM. */\n kind: 'deterministic' | 'llm'\n /** Optional declared upper bound; the registry can enforce a budget. */\n est_usd_per_run?: number\n /** Models the analyst expects to use (informational). */\n models?: string[]\n /** Maximum post-cancellation wait for provider usage. Model analysts default to 5 seconds. */\n settlement_timeout_ms?: number\n}\n\nexport interface AnalystRequirements {\n /** Min number of shots / samples the analyst needs to produce signal. */\n min_shots?: number\n /** Capabilities the runtime must supply (e.g. ['network', 'gpu']). */\n capabilities?: string[]\n}\n\n/**\n * What's passed to every analyst call. The registry resolves which\n * field the analyst's `inputKind` selects and asserts it's present.\n */\nexport interface AnalystRunInputs {\n traceStore?: TraceAnalysisStore\n artifactDir?: string\n runRecord?: RunRecord\n judgeInput?: JudgeInput\n /** Keyed by analyst id; populated by callers that registered custom analysts. */\n custom?: Record<string, unknown>\n}\n\nexport interface AnalystContext {\n runId: string\n /** Stable correlation id so logs from a single registry.run() share a tag. */\n correlationId: string\n /** Enforced wall-clock deadline (epoch ms). */\n deadlineMs?: number\n /** Per-analyst USD budget. Analysts MAY check before issuing LLM calls. */\n budgetUsd?: number\n /** Shared paid-call account when the analyst runs inside a larger campaign. */\n costLedger?: CostLedgerHandle\n /** Attribution phase used when writing to the shared paid-call account. */\n costPhase?: string\n /**\n * Shared chat client. Analysts that call an LLM go through this so\n * the operator picks transport (sandbox-sdk | router | cli-bridge |\n * direct-provider | mock) at the registry boundary without touching\n * analyst code.\n */\n chat?: ChatClient\n /**\n * Findings from a prior run the operator wants the analyst to see as\n * retrieval context. Kinds that take advantage of cross-run memory\n * (failure-mode \"I saw this cluster last run\", knowledge-gap \"the wiki\n * page I asked for is still missing\") render these into the actor's\n * working set. Filtering is the operator's job: pass the slice that\n * matches the analyst's id, or pass everything and let the kind\n * filter. Empty / absent means no cross-run context.\n */\n priorFindings?: ReadonlyArray<AnalystFinding>\n /**\n * Findings emitted by analysts that completed earlier in this registry run.\n * This is separate from `priorFindings`: upstream findings are dependency\n * context for the current pass, while prior findings are cross-run memory.\n * The registry populates this only when `RegistryRunOpts.chainFindings` is on.\n */\n upstreamFindings?: ReadonlyArray<AnalystFinding>\n /**\n * Report metered work independently of findings. This keeps an empty finding\n * set from erasing token/cost telemetry. Multiple receipts are accumulated.\n */\n recordUsage?: (receipt: AnalystUsageReceipt) => void\n /** Free-form runtime tags (env, host, op). Findings can echo these into metadata. */\n tags?: Record<string, string>\n /** Logger callback — analysts SHOULD prefer this over console.* for testability. */\n log?: (msg: string, fields?: Record<string, unknown>) => void\n /** Optional abort signal. Analysts SHOULD pass it through to LLM calls. */\n signal?: AbortSignal\n}\n\n/**\n * The minimal contract. Concrete analysts can refine `TInput` so\n * implementations stay type-safe (e.g. a trace analyst's `TInput` is\n * `TraceAnalysisStore`); the registry passes the right field from\n * `AnalystRunInputs` based on `inputKind`.\n */\nexport interface Analyst<TInput = unknown> {\n /** Stable identifier — appears in finding_id, telemetry, and registry exclusion lists. */\n readonly id: string\n /** Human-readable. One sentence. */\n readonly description: string\n readonly inputKind: AnalystInputKind\n readonly cost: AnalystCost\n readonly requires?: AnalystRequirements\n /** Bump on breaking changes to claim wording or area so old finding_ids don't collide. */\n readonly version: string\n analyze(input: TInput, ctx: AnalystContext): Promise<AnalystFinding[]>\n}\n\n/** Metered work performed by one analyst call. */\nexport interface AnalystUsageReceipt {\n /** Number of model-usage records observed at the provider boundary. */\n calls: number | null\n /** Null when the provider did not return token accounting. */\n tokens: RunTokenUsage | null\n /** Observed, estimated, or explicitly uncaptured dollar cost. */\n cost: RunCostProvenance\n /** Known lower bound when one or more calls have uncaptured cost. */\n knownCostUsd?: number\n}\n\n// ── finding_id stability ─────────────────────────────────────────────\n\n/**\n * Compute the stable finding_id from the identity-defining fields.\n * Default implementation hashes {analyst_id, area, subject, normalized claim}.\n * Analysts that emit findings whose claim text varies per run (timestamps,\n * counts) SHOULD either: (a) pass an explicit `id_basis` to fix the hash,\n * or (b) move the variable part into `rationale`/`metadata` and keep the\n * `claim` static.\n */\nexport function computeFindingId(input: {\n analyst_id: string\n area: string\n subject?: string\n claim: string\n /** Override the claim for hashing — use when the displayed claim has run-specific bits. */\n id_basis?: string\n}): string {\n const basis = JSON.stringify({\n a: input.analyst_id,\n r: input.area,\n s: input.subject ?? '',\n c: normalizeClaim(input.id_basis ?? input.claim),\n })\n return `f_${createHash('sha256').update(basis).digest('hex').slice(0, 20)}`\n}\n\nfunction normalizeClaim(c: string): string {\n // Lowercase, collapse whitespace, strip trailing punctuation. Goal:\n // \"Leaf X failed install\" and \"Leaf X failed install.\" hash the same.\n return c\n .toLowerCase()\n .replace(/\\s+/g, ' ')\n .replace(/[.!?;:,]+$/g, '')\n .trim()\n}\n\n/**\n * Convenience factory: produce a fully-formed AnalystFinding with the\n * id computed automatically. Analyst code stays terse.\n */\nexport function makeFinding(\n init: Omit<AnalystFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {\n id_basis?: string\n produced_at?: string\n },\n): AnalystFinding {\n const { id_basis, produced_at, ...rest } = init\n return {\n schema_version: '1.0.0',\n finding_id: computeFindingId({\n analyst_id: rest.analyst_id,\n area: rest.area,\n subject: rest.subject,\n claim: rest.claim,\n id_basis,\n }),\n produced_at: produced_at ?? new Date().toISOString(),\n ...rest,\n }\n}\n\n/** Build a finding whose source is explicitly allowed during candidate generation. */\nexport function makeProposalFinding(\n init: Omit<ProposalFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {\n id_basis?: string\n produced_at?: string\n },\n): ProposalFinding {\n const { proposal_origin, ...finding } = init\n return { ...makeFinding(finding), proposal_origin }\n}\n\n// ── Registry result envelope ────────────────────────────────────────\n\nexport interface AnalystRunSummary {\n analyst_id: string\n status: 'ok' | 'skipped' | 'failed'\n /** Why skipped — missing input, budget exceeded, capability unmet. */\n reason?: string\n findings_count: number\n latency_ms: number\n /** Additive model usage and cost provenance for this analyst. */\n usage: AnalystUsageReceipt\n /** When `status='failed'`: the error class + message, never the full stack. */\n error?: { class: string; message: string }\n}\n\nexport interface AnalystRunResult {\n run_id: string\n correlation_id: string\n started_at: string\n ended_at: string\n findings: AnalystFinding[]\n per_analyst: AnalystRunSummary[]\n /** Total LLM cost in USD across all analysts in this registry.run(). */\n total_cost_usd: number\n /**\n * Provenance for `total_cost_usd`. When uncaptured, the numeric field is only\n * the known subtotal and must not be treated as the run's total spend.\n */\n total_cost_provenance?: RunCostProvenance\n}\n\n// ── Streaming event envelope ────────────────────────────────────────\n\n/**\n * Events emitted by `AnalystRegistry.runStream(...)` in real time as\n * the registry executes. UIs subscribe via `for await (const ev of\n * registry.runStream(...))`; `registry.run(...)` is a thin collector\n * over the same stream, so the two surfaces share their invariants.\n *\n * Per-finding events are intentionally omitted — analyzers are batch\n * operations (an Ax actor returns the full `findings:json[]` at the\n * end of the responder), so streaming inside one analyst would only\n * emit partial JSON consumers can't render. The kind-completion event\n * is the right granularity; subscribers wanting per-finding rendering\n * iterate `event.findings` themselves.\n */\nexport type AnalystRunEvent =\n | {\n type: 'run-started'\n run_id: string\n correlation_id: string\n started_at: string\n /** The ordered list of analyst ids the registry will run. */\n analyst_ids: ReadonlyArray<string>\n }\n | {\n type: 'analyst-skipped'\n summary: AnalystRunSummary\n }\n | {\n type: 'analyst-started'\n analyst_id: string\n started_at: string\n }\n | {\n type: 'analyst-completed'\n /** `summary.status` is `'ok'` for clean completion or `'failed'` for thrown analysts. */\n summary: AnalystRunSummary\n findings: ReadonlyArray<AnalystFinding>\n }\n | {\n type: 'run-completed'\n result: AnalystRunResult\n }\n","import type { CostChannel, CostLedgerFilter, CostLedgerHandle } from '../cost-ledger'\nimport type { AnalystUsageReceipt } from './types'\n\nexport const DEFAULT_USAGE_SETTLEMENT_TIMEOUT_MS = 5_000\n\n/** Convert one ledger channel's complete call set into one analyst receipt. */\nexport function usageReceiptFromCostLedger(\n ledger: CostLedgerHandle,\n filter: CostChannel | CostLedgerFilter = 'analyst',\n): AnalystUsageReceipt {\n const resolvedFilter = typeof filter === 'string' ? { channel: filter } : filter\n const summary = ledger.summary(resolvedFilter)\n const receipts = ledger.list(resolvedFilter)\n const hasReasoningUsage = receipts.some((receipt) => receipt.reasoningTokens !== undefined)\n const hasCacheWriteUsage = receipts.some((receipt) => receipt.cacheWriteTokens !== undefined)\n const cost = summary.costProvenance\n return {\n calls: summary.totalCalls + summary.pendingCalls,\n tokens: summary.usageComplete\n ? {\n input: summary.inputTokens,\n output: summary.outputTokens,\n ...(hasReasoningUsage ? { reasoning: summary.reasoningTokens ?? 0 } : {}),\n ...(summary.cachedTokens > 0 ? { cached: summary.cachedTokens } : {}),\n ...(hasCacheWriteUsage ? { cacheWrite: summary.cacheWriteTokens ?? 0 } : {}),\n }\n : null,\n cost,\n ...(cost.kind === 'uncaptured' ? { knownCostUsd: summary.totalCostUsd } : {}),\n }\n}\n\nexport interface SettledUsageReceipt {\n settled: boolean\n pendingCalls: number\n receipt: AnalystUsageReceipt\n}\n\n/** Wait a bounded time for late provider receipts, then take one immutable snapshot. */\nexport async function settleUsageReceiptFromCostLedger(\n ledger: CostLedgerHandle,\n options: CostLedgerFilter & { timeoutMs?: number } = {},\n): Promise<SettledUsageReceipt> {\n const { timeoutMs: requestedTimeoutMs, ...requestedFilter } = options\n const filter: CostLedgerFilter = {\n channel: requestedFilter.channel ?? 'analyst',\n ...(requestedFilter.phase === undefined ? {} : { phase: requestedFilter.phase }),\n ...(requestedFilter.tags === undefined ? {} : { tags: requestedFilter.tags }),\n }\n const timeoutMs = validateUsageSettlementTimeout(requestedTimeoutMs)\n const initial = ledger.summary(filter)\n const waitResult =\n initial.pendingCalls === 0\n ? true\n : ledger.waitForIdle\n ? await ledger.waitForIdle({ timeoutMs })\n : false\n const pendingCalls = ledger.summary(filter).pendingCalls\n return {\n settled: waitResult && pendingCalls === 0,\n pendingCalls,\n receipt: usageReceiptFromCostLedger(ledger, filter),\n }\n}\n\nexport function validateUsageSettlementTimeout(timeoutMs?: number): number {\n const resolved = timeoutMs ?? DEFAULT_USAGE_SETTLEMENT_TIMEOUT_MS\n if (!Number.isSafeInteger(resolved) || resolved < 0 || resolved > 2_147_483_647) {\n throw new TypeError(\n 'settlementTimeoutMs must be a non-negative safe integer no greater than 2147483647',\n )\n }\n return resolved\n}\n\nexport function assertValidAnalystUsageReceipt(\n receipt: AnalystUsageReceipt,\n context = 'AnalystContext.recordUsage',\n): void {\n if (receipt.calls !== null && (!Number.isSafeInteger(receipt.calls) || receipt.calls < 0)) {\n throw new Error(`${context}: calls must be a non-negative safe integer or null`)\n }\n if (receipt.tokens) {\n assertNonNegativeSafeInteger(receipt.tokens.input, 'tokens.input', context)\n assertNonNegativeSafeInteger(receipt.tokens.output, 'tokens.output', context)\n if (receipt.tokens.reasoning !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.reasoning, 'tokens.reasoning', context)\n if (receipt.tokens.reasoning > receipt.tokens.output) {\n throw new Error(`${context}: tokens.reasoning must not exceed tokens.output`)\n }\n }\n if (receipt.tokens.cached !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.cached, 'tokens.cached', context)\n }\n if (receipt.tokens.cacheWrite !== undefined) {\n assertNonNegativeSafeInteger(receipt.tokens.cacheWrite, 'tokens.cacheWrite', context)\n }\n }\n if (receipt.cost.kind !== 'uncaptured') {\n assertNonNegativeFinite(receipt.cost.usd, 'cost.usd', context)\n } else if (receipt.cost.usd !== null) {\n throw new Error(`${context}: uncaptured cost.usd must be null`)\n }\n if (receipt.knownCostUsd !== undefined) {\n assertNonNegativeFinite(receipt.knownCostUsd, 'knownCostUsd', context)\n }\n}\n\nfunction assertNonNegativeSafeInteger(value: number, field: string, context: string): void {\n if (!Number.isSafeInteger(value) || value < 0) {\n throw new Error(`${context}: ${field} must be a non-negative safe integer`)\n }\n}\n\nfunction assertNonNegativeFinite(value: number, field: string, context: string): void {\n if (!Number.isFinite(value) || value < 0) {\n throw new Error(`${context}: ${field} must be a non-negative finite number`)\n }\n}\n","import { ValidationError } from '../errors'\n\n/** Base error for invalid search-ledger input or operations. */\nexport class SearchLedgerError extends ValidationError {}\n\n/** Error raised when durable search-ledger data fails an integrity check. */\nexport class SearchLedgerIntegrityError extends SearchLedgerError {}\n\n/** Error raised when an event identifier is reused with different content. */\nexport class SearchLedgerConflictError extends SearchLedgerError {}\n","/** Campaign binding of the generic journal file layer to search-ledger errors. */\n\nimport {\n appendLedgerLine,\n type FileLockResult,\n type LedgerFileContext,\n tryWithLedgerFileLock,\n withLedgerFileLock,\n} from '../ledger-core'\nimport { SearchLedgerIntegrityError } from './search-ledger-errors'\n\nexport const SEARCH_LEDGER_FILE_CONTEXT: LedgerFileContext = {\n subject: 'search ledger',\n integrityError: (message, options) => new SearchLedgerIntegrityError(message, options),\n}\n\nexport function appendSearchLedgerLine(path: string, line: string): void {\n appendLedgerLine(path, line, SEARCH_LEDGER_FILE_CONTEXT)\n}\n\nexport function withSearchLedgerFileLock<T>(ledgerPath: string, run: () => T): T {\n return withLedgerFileLock(ledgerPath, SEARCH_LEDGER_FILE_CONTEXT, run)\n}\n\nexport type { FileLockResult }\n\nexport function tryWithSearchLedgerFileLock<T>(\n ledgerPath: string,\n run: () => T,\n): FileLockResult<T> {\n return tryWithLedgerFileLock(ledgerPath, SEARCH_LEDGER_FILE_CONTEXT, run)\n}\n","import { createRequire } from 'node:module'\nimport { join } from 'node:path'\nimport { CostLedger } from '../cost-ledger'\nimport { appendSearchLedgerLine, tryWithSearchLedgerFileLock } from './search-ledger-file'\n\n/**\n * `CampaignStorage` — the filesystem seam `runCampaign` writes through\n * (run/cell dirs, the resumability cache, per-cell artifacts, trace spans).\n *\n * The default (`fsCampaignStorage`) is the Node filesystem — identical\n * behavior to the inline `node:fs` calls it replaces, so existing CLI\n * consumers are unaffected. `inMemoryCampaignStorage` keeps everything in a\n * `Map`, so the substrate runs in environments WITHOUT a filesystem\n * (Cloudflare Workers, Deno Deploy, other edge runtimes) — the campaign\n * still produces its `CampaignResult` (cells + aggregates) in memory;\n * artifacts/traces simply aren't persisted to disk.\n *\n * Paths are opaque keys to the in-memory adapter — it does not parse them,\n * so the same `join(...)`-built paths work unchanged across both adapters.\n */\nexport interface CampaignStorage {\n /** Ensure a directory exists (recursive). No-op for in-memory. */\n ensureDir(dir: string): void\n /** Does this path exist (as a written file or an ensured dir)? */\n exists(path: string): boolean\n /** Read a UTF-8 file; `undefined` when missing or unreadable. */\n read(path: string): string | undefined\n /** Write a file (string or bytes). Parent dir is assumed ensured. */\n write(path: string, content: string | Uint8Array): void\n /** Append only when the current UTF-8 byte length matches `expectedBytes`.\n * Returns the new length, or undefined when another writer won. */\n append(path: string, content: string, expectedBytes: number): number | undefined\n}\n\n/** Node-filesystem storage — the default. Lazily requires `node:fs` so the\n * module imports cleanly in non-Node runtimes (where the caller passes\n * `inMemoryCampaignStorage` instead and never constructs this).\n *\n * `createRequire(import.meta.url)` is the ESM-native lazy require — a bare\n * `require` is a ReferenceError under `\"type\": \"module\"`, which is exactly\n * the shape this package publishes. */\nexport function fsCampaignStorage(): CampaignStorage {\n const nodeRequire = createRequire(import.meta.url)\n const { existsSync, mkdirSync, readFileSync, statSync, writeFileSync } = nodeRequire(\n 'node:fs',\n ) as typeof import('node:fs')\n return {\n ensureDir(dir) {\n if (!existsSync(dir)) mkdirSync(dir, { recursive: true })\n },\n exists(path) {\n return existsSync(path)\n },\n read(path) {\n try {\n return readFileSync(path, 'utf8')\n } catch {\n return undefined\n }\n },\n write(path, content) {\n writeFileSync(path, content as Uint8Array)\n },\n append(path, content, expectedBytes) {\n const result = tryWithSearchLedgerFileLock(path, () => {\n let actualBytes = 0\n try {\n actualBytes = statSync(path).size\n } catch (error) {\n if ((error as NodeJS.ErrnoException).code !== 'ENOENT') throw error\n }\n if (actualBytes !== expectedBytes) return undefined\n appendSearchLedgerLine(path, content)\n return expectedBytes + Buffer.byteLength(content)\n })\n return result.acquired ? result.value : undefined\n },\n }\n}\n\n/** In-memory storage for filesystem-less runtimes. Artifacts + trace spans\n * live in a `Map` for the duration of the run; the `CampaignResult` is\n * fully populated, but nothing is persisted to disk. */\nexport function inMemoryCampaignStorage(): CampaignStorage {\n const files = new Map<string, string | Uint8Array>()\n const dirs = new Set<string>()\n return {\n ensureDir(dir) {\n dirs.add(dir)\n },\n exists(path) {\n return files.has(path) || dirs.has(path)\n },\n read(path) {\n const value = files.get(path)\n if (value === undefined) return undefined\n return typeof value === 'string' ? value : new TextDecoder().decode(value)\n },\n write(path, content) {\n files.set(path, content)\n },\n append(path, content, expectedBytes) {\n const current = files.get(path)\n const currentText =\n current === undefined\n ? ''\n : typeof current === 'string'\n ? current\n : new TextDecoder().decode(current)\n const currentBytes = new TextEncoder().encode(currentText).byteLength\n if (currentBytes !== expectedBytes) return undefined\n files.set(path, `${currentText}${content}`)\n return currentBytes + new TextEncoder().encode(content).byteLength\n },\n }\n}\n\n/** Open the durable spend account stored beside a logical run. */\nexport function createRunCostLedger(input: {\n storage: CampaignStorage\n runDir: string\n costCeilingUsd?: number\n}): CostLedger {\n const path = join(input.runDir, 'cost-ledger.jsonl')\n input.storage.ensureDir(input.runDir)\n return new CostLedger({\n costCeilingUsd: input.costCeilingUsd,\n persistence: {\n read: () => {\n const stored = input.storage.read(path)\n if (stored === undefined && input.storage.exists(path)) {\n throw new Error(`CostLedger: cannot read existing event log '${path}'`)\n }\n const events = stored ?? ''\n return {\n revision: String(new TextEncoder().encode(events).byteLength),\n events,\n }\n },\n append: (expectedRevision, event) => {\n const expectedBytes = Number(expectedRevision)\n if (!Number.isSafeInteger(expectedBytes) || expectedBytes < 0) {\n throw new Error(`CostLedger: invalid storage revision '${expectedRevision}'`)\n }\n const next = input.storage.append(path, event, expectedBytes)\n return next === undefined ? undefined : String(next)\n },\n },\n })\n}\n","/**\n * Single-run lock for evaluations that share one mutable environment.\n *\n * Two concurrent runs against a shared stateful gym silently corrupt each\n * other: each resets/mutates environment state mid-cell of the other, and\n * every score from both becomes garbage that LOOKS like worker variance\n * (agent-lab R357 burned hours on flip-flopping scores before tracing them\n * to exactly this). The fix is a pid lockfile: refuse to start while a live\n * holder exists, reclaim stale locks whose pid is gone, release only if the\n * lock is still ours.\n *\n * `alsoCheck` exists because independent runners can guard the same shared\n * resource with differently named lockfiles; a runner must respect all of\n * them even though it writes only its own.\n */\n\nimport {\n type AtomicFileLockUnavailable,\n probeAtomicFileLock,\n tryAcquireAtomicFileLock,\n} from '../ledger-core/atomic-file-lock'\n\nexport interface SingleRunLockOptions {\n /** Lockfile this runner writes (and checks). */\n readonly lockPath: string\n /** Other runners' lockfiles guarding the same resource; checked, never written. */\n readonly alsoCheck?: readonly string[]\n /** Install a process 'exit' hook that releases the lock. Default true. */\n readonly releaseOnExit?: boolean\n /** Owner pid recorded in the lockfile metadata. Default process.pid. */\n readonly pid?: number\n}\n\nexport interface SingleRunLock {\n /** Remove the lockfile if this process still owns it. Idempotent. */\n release(): void\n}\n\nfunction assertAvailable(path: string): void {\n const unavailable = probeAtomicFileLock({ lockPath: path })\n if (unavailable) throw unavailableError(path, unavailable)\n}\n\nfunction unavailableError(path: string, unavailable: AtomicFileLockUnavailable): Error {\n if (unavailable.reason === 'recovery') {\n return new Error(\n `single-run lock recovery is already in progress (${path}); refusing a concurrent run`,\n )\n }\n return new Error(\n `single-run lock held by live pid ${unavailable.holder.pid} (${path}); refusing a concurrent run on the shared resource`,\n )\n}\n\n/**\n * Acquire the lock or throw naming the live holder. A stale lock (holder pid\n * no longer running) is reclaimed by one contender. An interrupted reclaim\n * leaves a marker that fails closed instead of admitting overlapping runs.\n */\nexport function acquireSingleRunLock(opts: SingleRunLockOptions): SingleRunLock {\n const pid = opts.pid ?? process.pid\n if (!Number.isSafeInteger(pid) || pid <= 0)\n throw new Error('single-run lock pid must be a positive integer')\n for (const path of opts.alsoCheck ?? []) assertAvailable(path)\n const acquisition = tryAcquireAtomicFileLock({\n lockPath: opts.lockPath,\n pid,\n })\n if (!acquisition.acquired) throw unavailableError(opts.lockPath, acquisition)\n const releaseOnExit = opts.releaseOnExit ?? true\n let released = false\n const release = (): void => {\n if (released) return\n released = true\n if (releaseOnExit) process.off('exit', release)\n try {\n acquisition.lock.release()\n } catch {\n // release is best-effort; a leftover stale lock is reclaimed on next acquire\n }\n }\n try {\n for (const path of opts.alsoCheck ?? []) assertAvailable(path)\n } catch (error) {\n release()\n throw error\n }\n if (releaseOnExit) process.on('exit', release)\n return { release }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAkOA,SAAgB,iBAAiB,OAOtB;CACT,MAAM,QAAQ,KAAK,UAAU;EAC3B,GAAG,MAAM;EACT,GAAG,MAAM;EACT,GAAG,MAAM,WAAW;EACpB,GAAG,eAAe,MAAM,YAAY,MAAM,KAAK;CACjD,CAAC;CACD,OAAO,KAAK,WAAW,QAAQ,CAAC,CAAC,OAAO,KAAK,CAAC,CAAC,OAAO,KAAK,CAAC,CAAC,MAAM,GAAG,EAAE;AAC1E;AAEA,SAAS,eAAe,GAAmB;CAGzC,OAAO,EACJ,YAAY,CAAC,CACb,QAAQ,QAAQ,GAAG,CAAC,CACpB,QAAQ,eAAe,EAAE,CAAC,CAC1B,KAAK;AACV;;;;;AAMA,SAAgB,YACd,MAIgB;CAChB,MAAM,EAAE,UAAU,aAAa,GAAG,SAAS;CAC3C,OAAO;EACL,gBAAgB;EAChB,YAAY,iBAAiB;GAC3B,YAAY,KAAK;GACjB,MAAM,KAAK;GACX,SAAS,KAAK;GACd,OAAO,KAAK;GACZ;EACF,CAAC;EACD,aAAa,gCAAe,IAAI,KAAK,EAAA,CAAE,YAAY;EACnD,GAAG;CACL;AACF;;AAGA,SAAgB,oBACd,MAIiB;CACjB,MAAM,EAAE,iBAAiB,GAAG,YAAY;CACxC,OAAO;EAAE,GAAG,YAAY,OAAO;EAAG;CAAgB;AACpD;;ACzRA,SAAgB,2BACd,QACA,SAAyC,WACpB;CACrB,MAAM,iBAAiB,OAAO,WAAW,WAAW,EAAE,SAAS,OAAO,IAAI;CAC1E,MAAM,UAAU,OAAO,QAAQ,cAAc;CAC7C,MAAM,WAAW,OAAO,KAAK,cAAc;CAC3C,MAAM,oBAAoB,SAAS,MAAM,YAAY,QAAQ,oBAAoB,KAAA,CAAS;CAC1F,MAAM,qBAAqB,SAAS,MAAM,YAAY,QAAQ,qBAAqB,KAAA,CAAS;CAC5F,MAAM,OAAO,QAAQ;CACrB,OAAO;EACL,OAAO,QAAQ,aAAa,QAAQ;EACpC,QAAQ,QAAQ,gBACZ;GACE,OAAO,QAAQ;GACf,QAAQ,QAAQ;GAChB,GAAI,oBAAoB,EAAE,WAAW,QAAQ,mBAAmB,EAAE,IAAI,CAAC;GACvE,GAAI,QAAQ,eAAe,IAAI,EAAE,QAAQ,QAAQ,aAAa,IAAI,CAAC;GACnE,GAAI,qBAAqB,EAAE,YAAY,QAAQ,oBAAoB,EAAE,IAAI,CAAC;EAC5E,IACA;EACJ;EACA,GAAI,KAAK,SAAS,eAAe,EAAE,cAAc,QAAQ,aAAa,IAAI,CAAC;CAC7E;AACF;;AASA,eAAsB,iCACpB,QACA,UAAqD,CAAC,GACxB;CAC9B,MAAM,EAAE,WAAW,oBAAoB,GAAG,oBAAoB;CAC9D,MAAM,SAA2B;EAC/B,SAAS,gBAAgB,WAAW;EACpC,GAAI,gBAAgB,UAAU,KAAA,IAAY,CAAC,IAAI,EAAE,OAAO,gBAAgB,MAAM;EAC9E,GAAI,gBAAgB,SAAS,KAAA,IAAY,CAAC,IAAI,EAAE,MAAM,gBAAgB,KAAK;CAC7E;CACA,MAAM,YAAY,+BAA+B,kBAAkB;CAEnE,MAAM,aADU,OAAO,QAAQ,MAEvB,CAAC,CAAC,iBAAiB,IACrB,OACA,OAAO,cACL,MAAM,OAAO,YAAY,EAAE,UAAU,CAAC,IACtC;CACR,MAAM,eAAe,OAAO,QAAQ,MAAM,CAAC,CAAC;CAC5C,OAAO;EACL,SAAS,cAAc,iBAAiB;EACxC;EACA,SAAS,2BAA2B,QAAQ,MAAM;CACpD;AACF;AAEA,SAAgB,+BAA+B,WAA4B;CACzE,MAAM,WAAW,aAAA;CACjB,IAAI,CAAC,OAAO,cAAc,QAAQ,KAAK,WAAW,KAAK,WAAW,YAChE,MAAM,IAAI,UACR,oFACF;CAEF,OAAO;AACT;AAEA,SAAgB,+BACd,SACA,UAAU,8BACJ;CACN,IAAI,QAAQ,UAAU,SAAS,CAAC,OAAO,cAAc,QAAQ,KAAK,KAAK,QAAQ,QAAQ,IACrF,MAAM,IAAI,MAAM,GAAG,QAAQ,oDAAoD;CAEjF,IAAI,QAAQ,QAAQ;EAClB,6BAA6B,QAAQ,OAAO,OAAO,gBAAgB,OAAO;EAC1E,6BAA6B,QAAQ,OAAO,QAAQ,iBAAiB,OAAO;EAC5E,IAAI,QAAQ,OAAO,cAAc,KAAA,GAAW;GAC1C,6BAA6B,QAAQ,OAAO,WAAW,oBAAoB,OAAO;GAClF,IAAI,QAAQ,OAAO,YAAY,QAAQ,OAAO,QAC5C,MAAM,IAAI,MAAM,GAAG,QAAQ,iDAAiD;EAEhF;EACA,IAAI,QAAQ,OAAO,WAAW,KAAA,GAC5B,6BAA6B,QAAQ,OAAO,QAAQ,iBAAiB,OAAO;EAE9E,IAAI,QAAQ,OAAO,eAAe,KAAA,GAChC,6BAA6B,QAAQ,OAAO,YAAY,qBAAqB,OAAO;CAExF;CACA,IAAI,QAAQ,KAAK,SAAS,cACxB,wBAAwB,QAAQ,KAAK,KAAK,YAAY,OAAO;MACxD,IAAI,QAAQ,KAAK,QAAQ,MAC9B,MAAM,IAAI,MAAM,GAAG,QAAQ,mCAAmC;CAEhE,IAAI,QAAQ,iBAAiB,KAAA,GAC3B,wBAAwB,QAAQ,cAAc,gBAAgB,OAAO;AAEzE;AAEA,SAAS,6BAA6B,OAAe,OAAe,SAAuB;CACzF,IAAI,CAAC,OAAO,cAAc,KAAK,KAAK,QAAQ,GAC1C,MAAM,IAAI,MAAM,GAAG,QAAQ,IAAI,MAAM,qCAAqC;AAE9E;AAEA,SAAS,wBAAwB,OAAe,OAAe,SAAuB;CACpF,IAAI,CAAC,OAAO,SAAS,KAAK,KAAK,QAAQ,GACrC,MAAM,IAAI,MAAM,GAAG,QAAQ,IAAI,MAAM,sCAAsC;AAE/E;;;;ACnHA,IAAa,oBAAb,cAAuC,gBAAgB,CAAC;;AAGxD,IAAa,6BAAb,cAAgD,kBAAkB,CAAC;;AAGnE,IAAa,4BAAb,cAA+C,kBAAkB,CAAC;;;;ACElE,MAAa,6BAAgD;CAC3D,SAAS;CACT,iBAAiB,SAAS,YAAY,IAAI,2BAA2B,SAAS,OAAO;AACvF;AAEA,SAAgB,uBAAuB,MAAc,MAAoB;CACvE,iBAAiB,MAAM,MAAM,0BAA0B;AACzD;AAQA,SAAgB,4BACd,YACA,KACmB;CACnB,OAAO,sBAAsB,YAAY,4BAA4B,GAAG;AAC1E;;;;;;;;;;ACUA,SAAgB,oBAAqC;CAEnD,MAAM,EAAE,YAAY,WAAW,cAAc,UAAU,kBADnC,cAAc,OAAO,KAAK,GACqC,CAAC,CAClF,SACF;CACA,OAAO;EACL,UAAU,KAAK;GACb,IAAI,CAAC,WAAW,GAAG,GAAG,UAAU,KAAK,EAAE,WAAW,KAAK,CAAC;EAC1D;EACA,OAAO,MAAM;GACX,OAAO,WAAW,IAAI;EACxB;EACA,KAAK,MAAM;GACT,IAAI;IACF,OAAO,aAAa,MAAM,MAAM;GAClC,QAAQ;IACN;GACF;EACF;EACA,MAAM,MAAM,SAAS;GACnB,cAAc,MAAM,OAAqB;EAC3C;EACA,OAAO,MAAM,SAAS,eAAe;GACnC,MAAM,SAAS,4BAA4B,YAAY;IACrD,IAAI,cAAc;IAClB,IAAI;KACF,cAAc,SAAS,IAAI,CAAC,CAAC;IAC/B,SAAS,OAAO;KACd,IAAK,MAAgC,SAAS,UAAU,MAAM;IAChE;IACA,IAAI,gBAAgB,eAAe,OAAO,KAAA;IAC1C,uBAAuB,MAAM,OAAO;IACpC,OAAO,gBAAgB,OAAO,WAAW,OAAO;GAClD,CAAC;GACD,OAAO,OAAO,WAAW,OAAO,QAAQ,KAAA;EAC1C;CACF;AACF;;;;AAKA,SAAgB,0BAA2C;CACzD,MAAM,wBAAQ,IAAI,IAAiC;CACnD,MAAM,uBAAO,IAAI,IAAY;CAC7B,OAAO;EACL,UAAU,KAAK;GACb,KAAK,IAAI,GAAG;EACd;EACA,OAAO,MAAM;GACX,OAAO,MAAM,IAAI,IAAI,KAAK,KAAK,IAAI,IAAI;EACzC;EACA,KAAK,MAAM;GACT,MAAM,QAAQ,MAAM,IAAI,IAAI;GAC5B,IAAI,UAAU,KAAA,GAAW,OAAO,KAAA;GAChC,OAAO,OAAO,UAAU,WAAW,QAAQ,IAAI,YAAY,CAAC,CAAC,OAAO,KAAK;EAC3E;EACA,MAAM,MAAM,SAAS;GACnB,MAAM,IAAI,MAAM,OAAO;EACzB;EACA,OAAO,MAAM,SAAS,eAAe;GACnC,MAAM,UAAU,MAAM,IAAI,IAAI;GAC9B,MAAM,cACJ,YAAY,KAAA,IACR,KACA,OAAO,YAAY,WACjB,UACA,IAAI,YAAY,CAAC,CAAC,OAAO,OAAO;GACxC,MAAM,eAAe,IAAI,YAAY,CAAC,CAAC,OAAO,WAAW,CAAC,CAAC;GAC3D,IAAI,iBAAiB,eAAe,OAAO,KAAA;GAC3C,MAAM,IAAI,MAAM,GAAG,cAAc,SAAS;GAC1C,OAAO,eAAe,IAAI,YAAY,CAAC,CAAC,OAAO,OAAO,CAAC,CAAC;EAC1D;CACF;AACF;;AAGA,SAAgB,oBAAoB,OAIrB;CACb,MAAM,OAAO,KAAK,MAAM,QAAQ,mBAAmB;CACnD,MAAM,QAAQ,UAAU,MAAM,MAAM;CACpC,OAAO,IAAI,WAAW;EACpB,gBAAgB,MAAM;EACtB,aAAa;GACX,YAAY;IACV,MAAM,SAAS,MAAM,QAAQ,KAAK,IAAI;IACtC,IAAI,WAAW,KAAA,KAAa,MAAM,QAAQ,OAAO,IAAI,GACnD,MAAM,IAAI,MAAM,+CAA+C,KAAK,EAAE;IAExE,MAAM,SAAS,UAAU;IACzB,OAAO;KACL,UAAU,OAAO,IAAI,YAAY,CAAC,CAAC,OAAO,MAAM,CAAC,CAAC,UAAU;KAC5D;IACF;GACF;GACA,SAAS,kBAAkB,UAAU;IACnC,MAAM,gBAAgB,OAAO,gBAAgB;IAC7C,IAAI,CAAC,OAAO,cAAc,aAAa,KAAK,gBAAgB,GAC1D,MAAM,IAAI,MAAM,yCAAyC,iBAAiB,EAAE;IAE9E,MAAM,OAAO,MAAM,QAAQ,OAAO,MAAM,OAAO,aAAa;IAC5D,OAAO,SAAS,KAAA,IAAY,KAAA,IAAY,OAAO,IAAI;GACrD;EACF;CACF,CAAC;AACH;;;;;;;;;;;;;;;;;;AC/GA,SAAS,gBAAgB,MAAoB;CAC3C,MAAM,cAAc,oBAAoB,EAAE,UAAU,KAAK,CAAC;CAC1D,IAAI,aAAa,MAAM,iBAAiB,MAAM,WAAW;AAC3D;AAEA,SAAS,iBAAiB,MAAc,aAA+C;CACrF,IAAI,YAAY,WAAW,YACzB,uBAAO,IAAI,MACT,oDAAoD,KAAK,6BAC3D;CAEF,uBAAO,IAAI,MACT,oCAAoC,YAAY,OAAO,IAAI,IAAI,KAAK,oDACtE;AACF;;;;;;AAOA,SAAgB,qBAAqB,MAA2C;CAC9E,MAAM,MAAM,KAAK,OAAO,QAAQ;CAChC,IAAI,CAAC,OAAO,cAAc,GAAG,KAAK,OAAO,GACvC,MAAM,IAAI,MAAM,gDAAgD;CAClE,KAAK,MAAM,QAAQ,KAAK,aAAa,CAAC,GAAG,gBAAgB,IAAI;CAC7D,MAAM,cAAc,yBAAyB;EAC3C,UAAU,KAAK;EACf;CACF,CAAC;CACD,IAAI,CAAC,YAAY,UAAU,MAAM,iBAAiB,KAAK,UAAU,WAAW;CAC5E,MAAM,gBAAgB,KAAK,iBAAiB;CAC5C,IAAI,WAAW;CACf,MAAM,gBAAsB;EAC1B,IAAI,UAAU;EACd,WAAW;EACX,IAAI,eAAe,QAAQ,IAAI,QAAQ,OAAO;EAC9C,IAAI;GACF,YAAY,KAAK,QAAQ;EAC3B,QAAQ,CAER;CACF;CACA,IAAI;EACF,KAAK,MAAM,QAAQ,KAAK,aAAa,CAAC,GAAG,gBAAgB,IAAI;CAC/D,SAAS,OAAO;EACd,QAAQ;EACR,MAAM;CACR;CACA,IAAI,eAAe,QAAQ,GAAG,QAAQ,OAAO;CAC7C,OAAO,EAAE,QAAQ;AACnB"}
|
|
@@ -1,11 +1,12 @@
|
|
|
1
1
|
import { o as Severity } from "./multi-layer-verifier-BHY1gWAc.js";
|
|
2
|
-
import { c as CostLedgerHandle } from "./cost-ledger-
|
|
2
|
+
import { c as CostLedgerHandle } from "./cost-ledger-B1D3COAc.js";
|
|
3
3
|
import { C as TraceEvent, _ as Span, f as Run, n as BudgetLedgerEntry, t as Artifact } from "./schema-BtVldJ3T.js";
|
|
4
|
-
import "./index-
|
|
5
|
-
import { q as LlmClientOptions } from "./types-
|
|
4
|
+
import "./index-sMN_hI4E.js";
|
|
5
|
+
import { q as LlmClientOptions } from "./types-5q2T25iW.js";
|
|
6
6
|
import { s as TraceStore } from "./store-CT9YIIve.js";
|
|
7
|
-
import { i as AnalystFinding, n as AnalystContext
|
|
8
|
-
import { i as TraceAnalystKindSpec } from "./default-registry-
|
|
7
|
+
import { i as AnalystFinding, n as AnalystContext } from "./types-BtJhn8v6.js";
|
|
8
|
+
import { i as TraceAnalystKindSpec } from "./default-registry-PUhIVRWz.js";
|
|
9
|
+
import { l as ExactCapableAnalyst } from "./exact-types-Dpw2LeHA.js";
|
|
9
10
|
import { b as SupervisorRunTree, y as SupervisorRunSources } from "./types-Dea6tiVI.js";
|
|
10
11
|
import { AxAIArgs, AxAIService } from "@ax-llm/ax";
|
|
11
12
|
import { z } from "zod";
|
|
@@ -412,7 +413,7 @@ declare function diffFindings(previous: PersistedFinding[], current: PersistedFi
|
|
|
412
413
|
/** Translate typed supervisor-run integrity issues into the shared analyst envelope. */
|
|
413
414
|
declare function emitControlIntegrityFindings(input: SupervisorRunSources | SupervisorRunTree, producedAt: string): AnalystFinding[];
|
|
414
415
|
/** Deterministic Analyst adapter for `SupervisorRunSources | SupervisorRunTree`. */
|
|
415
|
-
declare class ControlIntegrityAnalyst implements
|
|
416
|
+
declare class ControlIntegrityAnalyst implements ExactCapableAnalyst<SupervisorRunSources | SupervisorRunTree> {
|
|
416
417
|
readonly id = "control-integrity";
|
|
417
418
|
readonly description = "Deterministic supervisor-run integrity checks with explicit unavailable evidence.";
|
|
418
419
|
readonly inputKind: 'custom';
|
|
@@ -421,6 +422,10 @@ declare class ControlIntegrityAnalyst implements Analyst<SupervisorRunSources |
|
|
|
421
422
|
est_usd_per_run: number;
|
|
422
423
|
};
|
|
423
424
|
readonly version = "2.0.0";
|
|
425
|
+
readonly executionConfig: {
|
|
426
|
+
readonly kind: 'control-integrity';
|
|
427
|
+
readonly produced_at_source: 'tags.producedAt-or-system-clock';
|
|
428
|
+
};
|
|
424
429
|
analyze(input: SupervisorRunSources | SupervisorRunTree, ctx: AnalystContext): Promise<AnalystFinding[]>;
|
|
425
430
|
}
|
|
426
431
|
declare const CONTROL_INTEGRITY_ANALYST: ControlIntegrityAnalyst;
|
|
@@ -496,7 +501,7 @@ interface SkillUsageScanConfig {
|
|
|
496
501
|
declare function buildSkillUsageReport(config: SkillUsageScanConfig): SkillUsageReport;
|
|
497
502
|
/** Pure rule pass over a report → findings. Exported for direct/unit use. */
|
|
498
503
|
declare function emitSkillUsageFindings(report: SkillUsageReport, producedAt: string): AnalystFinding[];
|
|
499
|
-
declare class SkillUsageAnalyst implements
|
|
504
|
+
declare class SkillUsageAnalyst implements ExactCapableAnalyst<SkillUsageReport> {
|
|
500
505
|
readonly id = "skill-usage";
|
|
501
506
|
readonly description = "Deterministic multi-signal skill-usage analysis: flags dead skills, measurement-invisible (orchestrated) usage, discovery gaps, public-repo leaks, bloat, missing evals, and missing run-logging.";
|
|
502
507
|
readonly inputKind: 'custom';
|
|
@@ -505,9 +510,14 @@ declare class SkillUsageAnalyst implements Analyst<SkillUsageReport> {
|
|
|
505
510
|
est_usd_per_run: number;
|
|
506
511
|
};
|
|
507
512
|
readonly version = "1.0.0";
|
|
513
|
+
readonly executionConfig: {
|
|
514
|
+
readonly kind: 'skill-usage';
|
|
515
|
+
readonly bloat_line_threshold: 300;
|
|
516
|
+
readonly produced_at_source: 'tags.producedAt-or-system-clock';
|
|
517
|
+
};
|
|
508
518
|
analyze(input: SkillUsageReport, ctx: AnalystContext): Promise<AnalystFinding[]>;
|
|
509
519
|
}
|
|
510
520
|
declare const SKILL_USAGE_ANALYST: SkillUsageAnalyst;
|
|
511
521
|
//#endregion
|
|
512
522
|
export { parseFindingSubject as A, SemanticConceptJudgeInput as B, FINDING_SUBJECT_KINDS as C, FindingSubjectStringSchema as D, FindingSubjectKind as E, ConceptFinding as F, RunCritic as G, SemanticConceptJudgeResult as H, ConceptSpec as I, DEFAULT_RUN_SCORE_WEIGHTS as J, RunCriticOptions as K, ConceptWeightStrategy as L, CreateAnalystAiConfig as M, createAnalystAi as N, KIND_EXPECTED_SUBJECTS as O, ConceptComplexity as P, clamp01 as Q, DEFAULT_COMPLEXITY_WEIGHTS as R, FINDING_SUBJECT_GRAMMAR_PROMPT as S, FindingSubject as T, createSemanticConceptJudge as U, SemanticConceptJudgeOptions as V, runSemanticConceptJudge as W, RunScoreWeights as X, RunScore as Y, aggregateRunScore as Z, FindingsDiff as _, SkillUsageScanConfig as a, defaultIsMaterial as b, DEFAULT_TRACE_ANALYST_KINDS as c, IMPROVEMENT_KIND_SPEC as d, FAILURE_MODE_KIND_SPEC as f, DiffPolicy as g, emitControlIntegrityFindings as h, SkillUsageReport as i, renderFindingSubject as j, findingSubjectGrammarPromptFor as k, KNOWLEDGE_POISONING_KIND_SPEC as l, ControlIntegrityAnalyst as m, SkillUsageAnalyst as n, buildSkillUsageReport as o, CONTROL_INTEGRITY_ANALYST as p, RunTrace as q, SkillUsageRecord as r, emitSkillUsageFindings as s, SKILL_USAGE_ANALYST as t, KNOWLEDGE_GAP_KIND_SPEC as u, FindingsStore as v, FINDING_SUBJECT_SYNTAX as w, diffFindings as x, PersistedFinding as y, SEMANTIC_CONCEPT_JUDGE_VERSION as z };
|
|
513
|
-
//# sourceMappingURL=skill-usage-
|
|
523
|
+
//# sourceMappingURL=skill-usage-Bv3G4VkA.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"skill-usage-Bv3G4VkA.d.ts","names":[],"sources":["../src/run-score.ts","../src/run-critic.ts","../src/semantic-concept-judge.ts","../src/analyst/ax-service.ts","../src/analyst/finding-subject.ts","../src/analyst/findings-store.ts","../src/analyst/kinds/control-integrity.ts","../src/analyst/kinds/failure-mode.ts","../src/analyst/kinds/improvement.ts","../src/analyst/kinds/knowledge-gap.ts","../src/analyst/kinds/knowledge-poisoning.ts","../src/analyst/kinds/index.ts","../src/analyst/kinds/skill-usage.ts"],"mappings":";;;;;;;;;;;;;UAAiB;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;cAGW,2BAA2B;iBAcxB,kBAAkB,OAAO,UAAU,UAAS,QAAQ;iBAiBpD,QAAQ;;;UCxDP;EACf,KAAK;EACL,OAAO;EACP,QAAQ;EACR,WAAW;EACX,QAAQ;;UAGO;EACf,UAAU,QAAQ;EAClB,gBAAgB;;cAYL;mBACM;mBACA;EAEjB,YAAY,UAAS;EAKf,MAAM,OAAO,YAAY,gBAAgB,QAAQ;EAYvD,WAAW,OAAO,WAAW;EAmH7B,KAAK,OAAO;UAIJ;;;;;;;;;;;;;;;;;;;;;;;;;KC/GE;UAEK;EACf;;EAEA;;EAEA;;EAEA,aAAa;;UAGE;EACf;EACA;;EAEA;EACA;EACA,UAAU;;UAGK;;EAEf;;EAEA;;EAEA,aAAa;IAAQ;IAAc;;;EAEnC,kBAAkB;;EAElB;EACA;;UAGe;EACf;EACA;;EAEA;EACA;EACA;EACA,UAAU;EACV;EACA;EACA;;EAEA;EACA;;;;;;;;KASU;cAEC,4BAA4B,OAAO;UAM/B;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,MAAM;EACN,aAAa;EACb;EACA,WAAW;EACX,SAAS;;;;;;EAMT,iBAAiB;;EAEjB,oBAAoB,QAAQ,OAAO;;cAKxB;;;;;;iBAkGS,wBACpB,OAAO,2BACP,UAAS,8BACR,QAAQ;;;;;iBAsJK,2BACd,UAAS,+BACP,OAAO,8BAA8B,QAAQ;;;UC/YhC;;;EAGf;;;EAGA;;EAEA,UAAU;;EAEV;;EAEA,WAAW;;;;;;;;;;;;;;iBAeG,gBAAgB,QAAQ,wBAAwB;;;;;;;;;;KCcpD;EAEN;EAAwB;EAAc;;EACtC;EAAyB;;EACzB;EAAuB;;EACvB;EAAyB;;EAGzB;EAAuB;;EACvB;EAAe;;EACf;EAAkB;EAAc;;EAChC;EAAkB;;EAClB;EAAa;EAAgB;;EAC7B;EAAc;;EACd;EAAkB;;EAClB;EAAkB;;EAClB;EAAwB;;EACxB;EAAuB;;EACvB;EAAc;;EACd;EAAa;EAAgB;;EAC7B;EAAgB;;EAChB;EAAqB;;EACrB;EAAuB;;EAEvB;EAA4B;;EAC5B;EAA2B;;EAE3B;EAAiB;;KAEX,qBAAqB;cAEpB,uBAAuB,cAAc;;;;;;;;;;;;;;;;;iBA2ClC,oBAAoB,iCAAiC;;;;;;;iBAoHrD,qBAAqB,GAAG;;;;;;;;;;cA8D3B,wBAAwB,SAAS,OAAO;cA0DxC;;;;;;;;;cAYA,wBAAwB,eAAe,cAAc;;iBAoDlD,+BAA+B;;;;;;;;;;;;cAmBlC,4BAA0B,EAAA;;;;;;;;UCzZtB,yBAAyB;EACxC;;cAGW;WAGiB;mBAFX;EAEjB,YAA4B;EAItB,OAAO,eAAe,UAAU,mBAAmB;;EAQzD,WAAW;;EAkBX,QAAQ,gBAAgB;;UAOT;;EAEf,UAAU;;EAEV,aAAa;;EAEb,WAAW;;;;;EAKX,SAAS;IAAQ,UAAU;IAAkB,SAAS;;;UAGvC;;;;;;;;EAQf,cAAc,UAAU,gBAAgB,SAAS;;;;;;iBAOnC,kBAAkB,GAAG,gBAAgB,GAAG;;;;;iBAWxC,aACd,UAAU,oBACV,SAAS,oBACT,SAAQ,aACP;;;;iBC7Fa,6BACd,OAAO,uBAAuB,mBAC9B,qBACC;;cA4BU,mCACA,oBAAoB,uBAAuB;WAE7C;WACA;WAEA;WACA;IAAS;IAAgC;;WACzC;WACA;aACP;aACA;;EAGI,QACJ,OAAO,uBAAuB,mBAC9B,KAAK,iBACJ,QAAQ;;cAUA,2BAAyB;;;cC7CzB,wBAAwB;;;cCmBxB,uBAAuB;;;cCGvB,yBAAyB;;;cCPzB,+BAA+B;;;;;;;;cCrB/B,sCAAsC;;;KCHvC;;UAGK;EACf;EACA,MAAM;;EAEN;EACA;;EAEA;;EAEA;;;EAGA;;EAEA;;EAEA;EACA;EACA;;EAEA;;EAEA;;UAGe;EACf;EACA,SAAS;;UAGM;;EAEf;;EAEA;IAAc;IAAc,MAAM;;;EAElC;;;EAGA,kBAAkB;;EAElB;;;iBA2Ec,sBAAsB,QAAQ,uBAAuB;;iBAiHrD,uBACd,QAAQ,kBACR,qBACC;cA2HU,6BAA6B,oBAAoB;WACnD;WACA;WAEA;WACA;IAAS;IAAgC;;WACzC;WACA;aACP;aACA;aACA;;EAGI,QAAQ,OAAO,kBAAkB,KAAK,iBAAiB,QAAQ;;cAS1D,qBAAmB"}
|