@tangle-network/agent-eval 0.100.0 → 0.100.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/http.d.ts +2 -2
- package/dist/adapters/langchain.d.ts +2 -2
- package/dist/adapters/otel.d.ts +4 -4
- package/dist/analyst/index.d.ts +8 -7
- package/dist/analyst/index.js +35 -27
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DtT6F_6T.d.ts → analyze-runs-BlJRBniC.d.ts} +3 -3
- package/dist/belief-state/index.d.ts +3 -3
- package/dist/benchmarks/index.d.ts +2 -2
- package/dist/campaign/index.d.ts +37 -13
- package/dist/campaign/index.js +76 -7
- package/dist/campaign/index.js.map +1 -1
- package/dist/{chunk-WMBLMTUE.js → chunk-2KTBHICD.js} +139 -16
- package/dist/chunk-2KTBHICD.js.map +1 -0
- package/dist/{chunk-IZCEK2HR.js → chunk-2MLIEQSN.js} +3 -2
- package/dist/{chunk-IZCEK2HR.js.map → chunk-2MLIEQSN.js.map} +1 -1
- package/dist/{chunk-OKQ2LAT7.js → chunk-4LWD6GC7.js} +7 -5
- package/dist/{chunk-OKQ2LAT7.js.map → chunk-4LWD6GC7.js.map} +1 -1
- package/dist/{chunk-LO6IOIJ2.js → chunk-ABOIVNXL.js} +2 -240
- package/dist/chunk-ABOIVNXL.js.map +1 -0
- package/dist/{chunk-NZEQVRH5.js → chunk-BOETF6BU.js} +2 -2
- package/dist/{chunk-S4SYLDFX.js → chunk-FRI6RG3P.js} +3 -2
- package/dist/chunk-FRI6RG3P.js.map +1 -0
- package/dist/chunk-G6S73VA7.js +248 -0
- package/dist/chunk-G6S73VA7.js.map +1 -0
- package/dist/{chunk-GMGRBNVT.js → chunk-G7IB3GJ5.js} +2 -2
- package/dist/chunk-IN3SHQML.js +664 -0
- package/dist/chunk-IN3SHQML.js.map +1 -0
- package/dist/{chunk-SJISCGWD.js → chunk-JU6ZX3CX.js} +2 -2
- package/dist/{chunk-3NHEO6ZC.js → chunk-L5TVEZFT.js} +2 -2
- package/dist/{chunk-77T4STFI.js → chunk-PMF5WIBX.js} +3 -3
- package/dist/{chunk-OYU4D7FY.js → chunk-VWQ6PO5O.js} +2 -2
- package/dist/{code-agent-session-CPHRCb4-.d.ts → code-agent-session-B6ZcDwyA.d.ts} +1 -1
- package/dist/contract/index.d.ts +16 -16
- package/dist/contract/index.js +6 -5
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-Doncu-B_.d.ts → control-DC8TELh0.d.ts} +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +3 -2
- package/dist/{corpus-D4YW9UoJ.d.ts → corpus-ONOzGFmG.d.ts} +1 -1
- package/dist/{default-registry-GyE8X5SP.d.ts → default-registry-Dhrc__SE.d.ts} +2 -2
- package/dist/diagnose.d.ts +3 -3
- package/dist/diagnose.js +2 -1
- package/dist/diagnose.js.map +1 -1
- package/dist/{gepa-H6mlM0KN.d.ts → gepa-BRgNnmGZ.d.ts} +1 -1
- package/dist/hosted/index.d.ts +4 -4
- package/dist/{index-_Y4oNOOb.d.ts → index-W96macmS.d.ts} +1 -1
- package/dist/index.d.ts +68 -22
- package/dist/index.js +113 -43
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-BnRjTibG.d.ts → insight-report-C02J3q4T.d.ts} +1 -1
- package/dist/{kind-factory-X3eDYbKn.d.ts → kind-factory-OgqQSvLi.d.ts} +1 -1
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/policy-edit-Dccm9tyA.d.ts +103 -0
- package/dist/{pre-registration-CMm8cvrh.d.ts → pre-registration-DB8oDqZJ.d.ts} +3 -3
- package/dist/{provenance-Bg_RttR8.d.ts → provenance-B0SZw1z2.d.ts} +3 -3
- package/dist/{release-report-pidWUMZ2.d.ts → release-report-B1tA6pKu.d.ts} +2 -2
- package/dist/reporting.d.ts +4 -4
- package/dist/{researcher-Jr8ME1dZ.d.ts → researcher-Ba2y1Foi.d.ts} +2 -2
- package/dist/rl.d.ts +8 -8
- package/dist/rl.js +3 -2
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-C2hDKM8Z.d.ts → rubric-predictive-validity-w7tun-q3.d.ts} +1 -1
- package/dist/{run-record-CP2ObebC.d.ts → run-record-DEwidcqn.d.ts} +1 -1
- package/dist/{runtime-trajectory-BOUUjI0y.d.ts → runtime-trajectory-OJDaTYHN.d.ts} +1 -1
- package/dist/{semantic-concept-judge-DSBB2Cfp.d.ts → semantic-concept-judge-J8xvjdc3.d.ts} +2 -2
- package/dist/{summary-report-CInXwsza.d.ts → summary-report-C4uzRWh8.d.ts} +1 -1
- package/dist/traces.d.ts +1 -1
- package/dist/traces.js +4 -3
- package/dist/{types-B5x54y6n.d.ts → types-BEzCBMQD.d.ts} +2 -2
- package/dist/{types-BTI16iFl.d.ts → types-Cv1bo4_a.d.ts} +1 -1
- package/dist/workflow/index.d.ts +4 -4
- package/dist/workflow/index.js +2 -1
- package/dist/workflow/index.js.map +1 -1
- package/package.json +1 -1
- package/dist/chunk-BUTW4RGG.js +0 -32
- package/dist/chunk-BUTW4RGG.js.map +0 -1
- package/dist/chunk-LO6IOIJ2.js.map +0 -1
- package/dist/chunk-S4SYLDFX.js.map +0 -1
- package/dist/chunk-WMBLMTUE.js.map +0 -1
- /package/dist/{chunk-NZEQVRH5.js.map → chunk-BOETF6BU.js.map} +0 -0
- /package/dist/{chunk-GMGRBNVT.js.map → chunk-G7IB3GJ5.js.map} +0 -0
- /package/dist/{chunk-SJISCGWD.js.map → chunk-JU6ZX3CX.js.map} +0 -0
- /package/dist/{chunk-3NHEO6ZC.js.map → chunk-L5TVEZFT.js.map} +0 -0
- /package/dist/{chunk-77T4STFI.js.map → chunk-PMF5WIBX.js.map} +0 -0
- /package/dist/{chunk-OYU4D7FY.js.map → chunk-VWQ6PO5O.js.map} +0 -0
package/dist/adapters/http.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { S as Scenario, D as DispatchFn, b as DispatchContext } from '../types-
|
|
2
|
-
import '../run-record-
|
|
1
|
+
import { S as Scenario, D as DispatchFn, b as DispatchContext } from '../types-Cv1bo4_a.js';
|
|
2
|
+
import '../run-record-DEwidcqn.js';
|
|
3
3
|
import '@tangle-network/agent-interface';
|
|
4
4
|
import '../errors-CzMUYo7b.js';
|
|
5
5
|
import '../schema-m0gsnbt3.js';
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { S as Scenario, J as JudgeScore, D as DispatchFn, a as JudgeConfig } from '../types-
|
|
2
|
-
import '../run-record-
|
|
1
|
+
import { S as Scenario, J as JudgeScore, D as DispatchFn, a as JudgeConfig } from '../types-Cv1bo4_a.js';
|
|
2
|
+
import '../run-record-DEwidcqn.js';
|
|
3
3
|
import '@tangle-network/agent-interface';
|
|
4
4
|
import '../errors-CzMUYo7b.js';
|
|
5
5
|
import '../schema-m0gsnbt3.js';
|
package/dist/adapters/otel.d.ts
CHANGED
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
import { TraceSpanEvent, HostedClient } from '../hosted/index.js';
|
|
2
|
-
import '../types-
|
|
3
|
-
import '../run-record-
|
|
2
|
+
import '../types-Cv1bo4_a.js';
|
|
3
|
+
import '../run-record-DEwidcqn.js';
|
|
4
4
|
import '@tangle-network/agent-interface';
|
|
5
5
|
import '../errors-CzMUYo7b.js';
|
|
6
6
|
import '../schema-m0gsnbt3.js';
|
|
7
|
-
import '../insight-report-
|
|
8
|
-
import '../summary-report-
|
|
7
|
+
import '../insight-report-C02J3q4T.js';
|
|
8
|
+
import '../summary-report-C4uzRWh8.js';
|
|
9
9
|
import '../failure-cluster-DH9Flgcf.js';
|
|
10
10
|
import '../store-BcFXE6LG.js';
|
|
11
11
|
import '../judge-calibration-0p2QcWNE.js';
|
package/dist/analyst/index.d.ts
CHANGED
|
@@ -1,21 +1,22 @@
|
|
|
1
1
|
import { M as MultiLayerVerifier, V as VerifyOptions, S as Severity } from '../multi-layer-verifier-DUZXrPDA.js';
|
|
2
2
|
import { c as RunCritic, a as RunTrace } from '../run-critic-CmMf05uV.js';
|
|
3
|
-
import { S as SemanticConceptJudgeOptions, a as SemanticConceptJudgeInput, B as BehavioralMetrics } from '../semantic-concept-judge-
|
|
4
|
-
export { C as CreateAnalystAiConfig, D as DEFAULT_TRACE_ANALYST_KINDS, b as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, c as FINDING_SUBJECT_GRAMMAR_PROMPT, d as FINDING_SUBJECT_KINDS, e as FindingSubject, f as FindingSubjectKind, g as FindingSubjectStringSchema, h as FindingsDiff, i as FindingsStore, I as IMPROVEMENT_KIND_SPEC, K as KIND_EXPECTED_SUBJECTS, j as KNOWLEDGE_GAP_KIND_SPEC, k as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, l as SKILL_USAGE_ANALYST, m as SkillUsageAnalyst, n as SkillUsageRecord, o as SkillUsageReport, p as SkillUsageScanConfig, q as buildSkillUsageReport, r as createAnalystAi, s as defaultIsMaterial, t as diffFindings, u as emitSkillUsageFindings, v as parseFindingSubject, w as renderFindingSubject } from '../semantic-concept-judge-
|
|
3
|
+
import { S as SemanticConceptJudgeOptions, a as SemanticConceptJudgeInput, B as BehavioralMetrics } from '../semantic-concept-judge-J8xvjdc3.js';
|
|
4
|
+
export { C as CreateAnalystAiConfig, D as DEFAULT_TRACE_ANALYST_KINDS, b as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, c as FINDING_SUBJECT_GRAMMAR_PROMPT, d as FINDING_SUBJECT_KINDS, e as FindingSubject, f as FindingSubjectKind, g as FindingSubjectStringSchema, h as FindingsDiff, i as FindingsStore, I as IMPROVEMENT_KIND_SPEC, K as KIND_EXPECTED_SUBJECTS, j as KNOWLEDGE_GAP_KIND_SPEC, k as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, l as SKILL_USAGE_ANALYST, m as SkillUsageAnalyst, n as SkillUsageRecord, o as SkillUsageReport, p as SkillUsageScanConfig, q as buildSkillUsageReport, r as createAnalystAi, s as defaultIsMaterial, t as diffFindings, u as emitSkillUsageFindings, v as parseFindingSubject, w as renderFindingSubject } from '../semantic-concept-judge-J8xvjdc3.js';
|
|
5
5
|
import { b as JudgeFn, a as JudgeInput } from '../types-C7DGg5ex.js';
|
|
6
|
-
import {
|
|
7
|
-
export {
|
|
6
|
+
import { a as Analyst, h as AnalystSeverity, A as AnalystFinding } from '../types-BEzCBMQD.js';
|
|
7
|
+
export { b as AnalystContext, g as AnalystCost, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, c as AnalystRunSummary, k as ChatCallOpts, C as ChatClient, l as ChatRequest, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, p as CreateChatClientOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from '../types-BEzCBMQD.js';
|
|
8
8
|
import { TCloud } from '@tangle-network/tcloud';
|
|
9
9
|
import { T as TraceAnalysisStore } from '../store-C1YxJDEK.js';
|
|
10
|
-
export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from '../default-registry-
|
|
11
|
-
export { A as ANALYST_SEVERITIES, C as CreateTraceAnalystKindOpts, R as RAW_FINDING_SCHEMA_PROMPT, a as RawAnalystFinding, b as RawAnalystFindingSchema, c as TraceAnalystGolden, T as TraceAnalystKindSpec, d as createTraceAnalystKind, p as parseRawFinding, r as renderPriorFindings } from '../kind-factory-
|
|
10
|
+
export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from '../default-registry-Dhrc__SE.js';
|
|
11
|
+
export { A as ANALYST_SEVERITIES, C as CreateTraceAnalystKindOpts, R as RAW_FINDING_SCHEMA_PROMPT, a as RawAnalystFinding, b as RawAnalystFindingSchema, c as TraceAnalystGolden, T as TraceAnalystKindSpec, d as createTraceAnalystKind, p as parseRawFinding, r as renderPriorFindings } from '../kind-factory-OgqQSvLi.js';
|
|
12
|
+
export { F as FindingToPolicyEditOptions, P as POLICY_EDIT_AXES, a as POLICY_EDIT_TARGET_SURFACES, b as PolicyEdit, c as PolicyEditAdmission, d as PolicyEditAdmissionOptions, e as PolicyEditAxis, f as PolicyEditChange, g as PolicyEditExpectedGain, h as PolicyEditGainDirection, i as PolicyEditGainUnit, j as PolicyEditInit, k as PolicyEditRisk, l as PolicyEditSchemaVersion, m as PolicyEditSource, n as PolicyEditTarget, o as PolicyEditTargetSurface, p as PolicyEditValidationError, q as admitPolicyEdit, r as applyPolicyEditToSurface, s as computePolicyEditId, t as isPolicyEdit, u as makePolicyEdit, v as policyEditFromFinding, w as policyEditsFromFindings, x as scorePolicyEditReadiness, y as validatePolicyEdit } from '../policy-edit-Dccm9tyA.js';
|
|
12
13
|
import { L as LlmClientOptions } from '../llm-client-Bj7g0rqu.js';
|
|
13
14
|
import { AxFunction } from '@ax-llm/ax';
|
|
14
15
|
import '../verdict-C9MlYujm.js';
|
|
15
16
|
import '../schema-m0gsnbt3.js';
|
|
16
17
|
import '../store-BcFXE6LG.js';
|
|
17
18
|
import 'zod';
|
|
18
|
-
import '../run-record-
|
|
19
|
+
import '../run-record-DEwidcqn.js';
|
|
19
20
|
import '@tangle-network/agent-interface';
|
|
20
21
|
import '../errors-CzMUYo7b.js';
|
|
21
22
|
import '../raw-provider-sink-C46HDghv.js';
|
package/dist/analyst/index.js
CHANGED
|
@@ -11,14 +11,30 @@ import {
|
|
|
11
11
|
diffFindings,
|
|
12
12
|
emitSkillUsageFindings,
|
|
13
13
|
runSemanticConceptJudge
|
|
14
|
-
} from "../chunk-
|
|
14
|
+
} from "../chunk-JU6ZX3CX.js";
|
|
15
15
|
import "../chunk-DJWX3GVS.js";
|
|
16
16
|
import {
|
|
17
17
|
behavioralAnalyst,
|
|
18
18
|
buildDefaultAnalystRegistry,
|
|
19
19
|
deriveEfficiencyFindings
|
|
20
|
-
} from "../chunk-
|
|
21
|
-
import
|
|
20
|
+
} from "../chunk-L5TVEZFT.js";
|
|
21
|
+
import {
|
|
22
|
+
POLICY_EDIT_AXES,
|
|
23
|
+
POLICY_EDIT_TARGET_SURFACES,
|
|
24
|
+
PolicyEditValidationError,
|
|
25
|
+
admitPolicyEdit,
|
|
26
|
+
applyPolicyEditToSurface,
|
|
27
|
+
assertNoJudgeVerdict,
|
|
28
|
+
computePolicyEditId,
|
|
29
|
+
isJudgeVerdict,
|
|
30
|
+
isPolicyEdit,
|
|
31
|
+
isTraceObservable,
|
|
32
|
+
makePolicyEdit,
|
|
33
|
+
policyEditFromFinding,
|
|
34
|
+
policyEditsFromFindings,
|
|
35
|
+
scorePolicyEditReadiness,
|
|
36
|
+
validatePolicyEdit
|
|
37
|
+
} from "../chunk-IN3SHQML.js";
|
|
22
38
|
import {
|
|
23
39
|
ANALYST_SEVERITIES,
|
|
24
40
|
AnalystRegistry,
|
|
@@ -43,7 +59,7 @@ import {
|
|
|
43
59
|
renderPriorFindings,
|
|
44
60
|
stripCodeFences,
|
|
45
61
|
structureFindings
|
|
46
|
-
} from "../chunk-
|
|
62
|
+
} from "../chunk-FRI6RG3P.js";
|
|
47
63
|
import "../chunk-CWNP4DV4.js";
|
|
48
64
|
import {
|
|
49
65
|
computeFindingId,
|
|
@@ -51,6 +67,8 @@ import {
|
|
|
51
67
|
} from "../chunk-45EEMHTC.js";
|
|
52
68
|
import "../chunk-YBIGNSCZ.js";
|
|
53
69
|
import "../chunk-PC4UYEBM.js";
|
|
70
|
+
import "../chunk-ABOIVNXL.js";
|
|
71
|
+
import "../chunk-VSMTAMNK.js";
|
|
54
72
|
import "../chunk-3BFEG2F6.js";
|
|
55
73
|
import "../chunk-PZ5AY32C.js";
|
|
56
74
|
|
|
@@ -275,28 +293,6 @@ function createSemanticConceptJudgeAdapter(opts = {}) {
|
|
|
275
293
|
}
|
|
276
294
|
};
|
|
277
295
|
}
|
|
278
|
-
|
|
279
|
-
// src/analyst/steer-firewall.ts
|
|
280
|
-
var OBSERVABLE_KINDS = /* @__PURE__ */ new Set([
|
|
281
|
-
"span",
|
|
282
|
-
"event",
|
|
283
|
-
"artifact"
|
|
284
|
-
]);
|
|
285
|
-
function isTraceObservable(finding) {
|
|
286
|
-
return finding.evidence_refs.some((ref) => OBSERVABLE_KINDS.has(ref.kind));
|
|
287
|
-
}
|
|
288
|
-
function isJudgeVerdict(finding) {
|
|
289
|
-
return finding.derived_from_judge === true;
|
|
290
|
-
}
|
|
291
|
-
function assertNoJudgeVerdict(findings, context = "steer") {
|
|
292
|
-
const leaks = findings.filter(isJudgeVerdict);
|
|
293
|
-
if (leaks.length > 0) {
|
|
294
|
-
throw new Error(
|
|
295
|
-
`${context}: a judge verdict cannot be admitted as steering input \u2014 that is the held-out judge leaking into the loop. Offending judge-derived findings: [${leaks.map((f) => f.finding_id).join(", ")}]. Steering consumes observations of behavior, never acceptance verdicts.`
|
|
296
|
-
);
|
|
297
|
-
}
|
|
298
|
-
return findings;
|
|
299
|
-
}
|
|
300
296
|
export {
|
|
301
297
|
ANALYST_SEVERITIES,
|
|
302
298
|
AnalystRegistry,
|
|
@@ -310,10 +306,15 @@ export {
|
|
|
310
306
|
KIND_EXPECTED_SUBJECTS,
|
|
311
307
|
KNOWLEDGE_GAP_KIND_SPEC,
|
|
312
308
|
KNOWLEDGE_POISONING_KIND_SPEC,
|
|
309
|
+
POLICY_EDIT_AXES,
|
|
310
|
+
POLICY_EDIT_TARGET_SURFACES,
|
|
311
|
+
PolicyEditValidationError,
|
|
313
312
|
RAW_FINDING_SCHEMA_PROMPT,
|
|
314
313
|
RawAnalystFindingSchema,
|
|
315
314
|
SKILL_USAGE_ANALYST,
|
|
316
315
|
SkillUsageAnalyst,
|
|
316
|
+
admitPolicyEdit,
|
|
317
|
+
applyPolicyEditToSurface,
|
|
317
318
|
assertNoJudgeVerdict,
|
|
318
319
|
behavioralAnalyst,
|
|
319
320
|
buildDefaultAnalystRegistry,
|
|
@@ -322,6 +323,7 @@ export {
|
|
|
322
323
|
coerceJson,
|
|
323
324
|
coerceToFindingRows,
|
|
324
325
|
computeFindingId,
|
|
326
|
+
computePolicyEditId,
|
|
325
327
|
createAnalystAi,
|
|
326
328
|
createChatClient,
|
|
327
329
|
createJudgeAdapter,
|
|
@@ -334,14 +336,20 @@ export {
|
|
|
334
336
|
diffFindings,
|
|
335
337
|
emitSkillUsageFindings,
|
|
336
338
|
isJudgeVerdict,
|
|
339
|
+
isPolicyEdit,
|
|
337
340
|
isTraceObservable,
|
|
338
341
|
liftSeverity,
|
|
339
342
|
makeFinding,
|
|
343
|
+
makePolicyEdit,
|
|
340
344
|
parseFindingSubject,
|
|
341
345
|
parseRawFinding,
|
|
346
|
+
policyEditFromFinding,
|
|
347
|
+
policyEditsFromFindings,
|
|
342
348
|
renderFindingSubject,
|
|
343
349
|
renderPriorFindings,
|
|
350
|
+
scorePolicyEditReadiness,
|
|
344
351
|
stripCodeFences,
|
|
345
|
-
structureFindings
|
|
352
|
+
structureFindings,
|
|
353
|
+
validatePolicyEdit
|
|
346
354
|
};
|
|
347
355
|
//# sourceMappingURL=index.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"sources":["../../src/analyst/adapters.ts","../../src/analyst/steer-firewall.ts"],"sourcesContent":["/**\n * Adapter factories — lift each existing agent-eval primitive into the\n * Analyst contract without re-implementing it.\n *\n * Five primitives, five factories. Each one:\n * - Builds an Analyst with a stable id (caller chooses; defaults\n * given), a sensible default `inputKind`, a version derived from\n * the wrapped primitive's version + an adapter revision, and an\n * `analyze()` that calls the primitive and lifts its output to\n * AnalystFinding[] using `makeFinding()`.\n * - Maps severities: the existing `Severity` ('critical' | 'major' |\n * 'minor' | 'info') projects onto AnalystSeverity ('critical' |\n * 'high' | 'medium' | 'low' | 'info'); 'major' → 'high', 'minor' →\n * 'medium'. Domain analysts that want finer-grained mapping override.\n *\n * Adapters never own state. Calling the same factory twice with the\n * same primitive instance is safe.\n */\n\nimport type {\n Finding as LayerFinding,\n Severity as LayerSeverity,\n MultiLayerVerifier,\n VerifyOptions,\n} from '../multi-layer-verifier'\nimport { RunCritic, type RunTrace } from '../run-critic'\nimport {\n runSemanticConceptJudge,\n SEMANTIC_CONCEPT_JUDGE_VERSION,\n type SemanticConceptJudgeInput,\n type SemanticConceptJudgeOptions,\n} from '../semantic-concept-judge'\nimport type { JudgeFn, JudgeInput, JudgeScore, TCloud } from '../types'\nimport type { Analyst, AnalystFinding, AnalystSeverity } from './types'\nimport { makeFinding } from './types'\n\nconst ADAPTER_REV = '1'\n\n// ── Severity bridges ───────────────────────────────────────────────\n\nexport function liftSeverity(s: LayerSeverity): AnalystSeverity {\n switch (s) {\n case 'critical':\n return 'critical'\n case 'major':\n return 'high'\n case 'minor':\n return 'medium'\n case 'info':\n return 'info'\n }\n}\n\n// ── 1. MultiLayerVerifier → Analyst ─────────────────────────────────\n\nexport interface VerifierAdapterOpts<Env> {\n id?: string\n area?: string\n verifier: MultiLayerVerifier<Env>\n /**\n * The verifier expects an `env` per run. Adapters take it from\n * `AnalystRunInputs.custom[<id>]` via the registry's 'custom' routing.\n */\n options?: Omit<VerifyOptions<Env>, 'env'>\n}\n\nexport function createVerifierAdapter<Env>(opts: VerifierAdapterOpts<Env>): Analyst<Env> {\n const id = opts.id ?? 'multi-layer-verifier'\n const area = opts.area ?? 'verification'\n return {\n id,\n description:\n \"Runs a MultiLayerVerifier and lifts each layer's findings into the analyst envelope.\",\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `verifier-${ADAPTER_REV}`,\n async analyze(env, ctx) {\n const report = await opts.verifier.run({ env, ...opts.options })\n const out: AnalystFinding[] = []\n for (const layer of report.layers) {\n for (const finding of layer.findings) {\n out.push(liftLayerFinding(id, area, layer.layer, finding))\n }\n // Layer-level signal: a failed/error layer is itself a finding\n // even if it didn't emit per-finding rows.\n if (layer.status === 'fail' || layer.status === 'error' || layer.status === 'timeout') {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: layer.layer,\n claim: `layer \"${layer.layer}\" ${layer.status}: ${layer.reason ?? 'no reason given'}`,\n severity:\n layer.status === 'error' ? 'high' : layer.status === 'timeout' ? 'medium' : 'high',\n confidence: 1,\n evidence_refs: [],\n metadata: {\n layer_status: layer.status,\n duration_ms: layer.durationMs,\n score: layer.score,\n diagnostics: layer.diagnostics,\n },\n }),\n )\n }\n }\n ctx.log?.('verifier complete', {\n layers: report.layers.length,\n blended: report.blendedScore,\n all_pass: report.allPass,\n })\n return out\n },\n }\n}\n\nfunction liftLayerFinding(\n analyst_id: string,\n area: string,\n layer: string,\n f: LayerFinding,\n): AnalystFinding {\n return makeFinding({\n analyst_id,\n area,\n subject: f.layer ?? layer,\n claim: f.message,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: f.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }]\n : [],\n metadata: f.detail,\n })\n}\n\n// ── 2. RunCritic → Analyst ──────────────────────────────────────────\n\nexport interface RunCriticAdapterOpts {\n id?: string\n area?: string\n critic?: RunCritic\n /** Optional threshold below which a dimension is reported as a finding. Default 0.5. */\n threshold?: number\n}\n\nexport function createRunCriticAdapter(opts: RunCriticAdapterOpts = {}): Analyst<RunTrace> {\n const id = opts.id ?? 'run-critic'\n const area = opts.area ?? 'run-quality'\n const critic = opts.critic ?? new RunCritic()\n const threshold = opts.threshold ?? 0.5\n return {\n id,\n description:\n 'Scores a single run across success / grounding / drift / tool-quality and surfaces below-threshold dimensions.',\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `run-critic-${ADAPTER_REV}`,\n async analyze(trace) {\n const score = critic.scoreTrace(trace)\n const out: AnalystFinding[] = []\n const dims: Array<[keyof typeof score, AnalystSeverity, string]> = [\n ['success', 'critical', 'run did not complete successfully'],\n ['goalProgress', 'high', 'goal progress is low'],\n ['repoGroundedness', 'high', 'output is poorly grounded in the repository'],\n ['toolUseQuality', 'medium', 'tool use quality is low'],\n ['patchQuality', 'medium', 'no real patch/edit evidence'],\n ['testReality', 'high', 'no real test/build evidence'],\n ['finalGate', 'critical', 'final gate is blocking'],\n ]\n for (const [dim, sev, msg] of dims) {\n const value = score[dim] as number\n if (typeof value === 'number' && value < threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: dim,\n claim: msg,\n rationale: `${dim}=${value.toFixed(2)} below threshold ${threshold}`,\n severity: sev,\n confidence: 1,\n evidence_refs: [],\n metadata: { dimension: dim, value, threshold, run_id: trace.run.runId },\n }),\n )\n }\n }\n // Drift penalty is high → surface as a finding (inverse threshold).\n if (score.driftPenalty > 1 - threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: 'drift',\n claim: 'agent output drifted from repository signal',\n rationale: `driftPenalty=${score.driftPenalty.toFixed(2)}`,\n severity: 'medium',\n confidence: 0.9,\n evidence_refs: [],\n metadata: { drift_penalty: score.driftPenalty, notes: score.notes },\n }),\n )\n }\n return out\n },\n }\n}\n\n// ── 3. JudgeFn → Analyst ────────────────────────────────────────────\n\nexport interface JudgeAdapterOpts {\n id?: string\n area?: string\n judge: JudgeFn\n /** TCloud handle the JudgeFn calls. */\n tcloud: TCloud\n /** Optional cost classification — most judges call an LLM. */\n cost?: Analyst['cost']\n /** Optional threshold below which a JudgeScore becomes a finding. Default 6 (on 0-10 scale). */\n threshold?: number\n}\n\nexport function createJudgeAdapter(opts: JudgeAdapterOpts): Analyst<JudgeInput> {\n const id = opts.id ?? 'judge'\n const area = opts.area ?? 'judge'\n const threshold = opts.threshold ?? 6\n return {\n id,\n description:\n 'Wraps an agent-eval JudgeFn into an analyst; below-threshold dimensions surface as findings.',\n inputKind: 'judge-input',\n cost: opts.cost ?? { kind: 'llm' },\n version: `judge-${ADAPTER_REV}`,\n async analyze(input) {\n const scores = await opts.judge(opts.tcloud, input)\n return scores\n .filter((s) => normalize10(s.score) < threshold)\n .map((s) => liftJudgeScore(id, area, s))\n },\n }\n}\n\nfunction normalize10(s: number): number {\n // JudgeScore convention is 0-10 but some judges emit 0-1. Coerce to 0-10.\n return s <= 1 ? s * 10 : s\n}\n\nfunction liftJudgeScore(analyst_id: string, area: string, s: JudgeScore): AnalystFinding {\n const score10 = normalize10(s.score)\n const severity: AnalystSeverity =\n score10 < 3 ? 'critical' : score10 < 5 ? 'high' : score10 < 7 ? 'medium' : 'low'\n return makeFinding({\n analyst_id,\n area,\n subject: s.dimension,\n claim: `${s.judgeName}/${s.dimension} scored ${score10.toFixed(1)}/10`,\n rationale: s.reasoning,\n severity,\n confidence: 0.8,\n evidence_refs: s.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: s.evidence }]\n : [],\n // Provenance: this finding IS a judge verdict (an acceptance score), not an\n // observation of behavior. The steer firewall (assertNoJudgeVerdict) rejects\n // it from steering — even when it cites an artifact above — because letting a\n // verdict steer the next attempt is the held-out judge leaking into the loop.\n derived_from_judge: true,\n metadata: { judge_name: s.judgeName, dimension: s.dimension, score_10: score10 },\n })\n}\n\n// ── 4. SemanticConceptJudge → Analyst ──────────────────────────────\n\nexport interface SemanticConceptJudgeAdapterOpts {\n id?: string\n area?: string\n options?: SemanticConceptJudgeOptions\n}\n\nexport function createSemanticConceptJudgeAdapter(\n opts: SemanticConceptJudgeAdapterOpts = {},\n): Analyst<SemanticConceptJudgeInput> {\n const id = opts.id ?? 'semantic-concept-judge'\n const area = opts.area ?? 'concept-coverage'\n return {\n id,\n description:\n 'Runs the semantic-concept judge and surfaces missing / weak concepts as findings.',\n inputKind: 'custom',\n cost: { kind: 'llm', models: opts.options?.model ? [opts.options.model] : undefined },\n version: `${SEMANTIC_CONCEPT_JUDGE_VERSION}-adapter-${ADAPTER_REV}`,\n async analyze(input) {\n const result = await runSemanticConceptJudge(input, opts.options)\n if (!result.available) {\n return [\n makeFinding({\n analyst_id: id,\n area,\n claim: 'semantic-concept judge unavailable',\n rationale: result.error,\n severity: 'info',\n confidence: 1,\n evidence_refs: [],\n metadata: { reason: result.error },\n }),\n ]\n }\n const out: AnalystFinding[] = []\n for (const f of result.findings) {\n // Only surface gaps: missing concepts or low scores. Concepts at\n // 7+/10 with present=true are not findings — they're successes.\n if (f.present && f.score >= 7) continue\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: f.concept,\n claim: f.present\n ? `concept \"${f.concept}\" is weak (${f.score}/10)`\n : `concept \"${f.concept}\" is missing`,\n rationale: f.evidence,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }],\n metadata: {\n concept: f.concept,\n present: f.present,\n score_10: f.score,\n cost_usd: result.costUsd ?? undefined,\n },\n }),\n )\n }\n return out\n },\n }\n}\n","// The realness-oracle firewall (docs/learning-flywheel.md, \"The steer is f(trace)\").\n//\n// A realness/authenticity signal has TWO legitimate roles that must stay\n// separated by a firewall:\n// (a) anchor judge J — write-only: scores the chosen output, gates promotion,\n// NEVER seen by the worker/optimizer mid-run (else the loop games it).\n// (b) steer f(trace) — an analyst observes the agent's OWN behavior in the\n// trace (\"imported a stub\", \"used a non-crypto PRNG where encryption was\n// required\") and steers the next attempt. Legitimate, because it is derived\n// from OBSERVABLE BEHAVIOR, not from J's held-out verdict.\n//\n// The correct discriminator is PROVENANCE, not evidence presence. A judge verdict\n// lifted into a finding (createJudgeAdapter → liftJudgeScore) is a verdict even\n// when it cites an artifact; an evidence-less trace-analyst bullet is an\n// observation even though it cites nothing. So the firewall keys on\n// `AnalystFinding.derived_from_judge` (set at the judge lift site), NOT on whether\n// evidence_refs is populated. The instant a verdict steers the next attempt it is\n// a back-channel for J and the loop Goodharts realness exactly as it would\n// Goodhart pass-rate.\n\nimport type { AnalystFinding, EvidenceRef } from './types'\n\n/** Evidence grounded in the agent's OWN execution: OTLP trace elements\n * (`span`/`event`) or the artifact it produced (`artifact`). */\nconst OBSERVABLE_KINDS: ReadonlySet<EvidenceRef['kind']> = new Set<EvidenceRef['kind']>([\n 'span',\n 'event',\n 'artifact',\n])\n\n/** DESCRIPTIVE predicate: does the finding cite at least one observable\n * (span/event/artifact) evidence ref. Useful for ranking evidence quality or\n * rendering — it is NOT the steer gate. Evidence presence is the WRONG\n * discriminator for steering: a legitimate trace-analyst observation may cite\n * nothing (it would be wrongly rejected), and a judge verdict may cite an\n * artifact (it would be wrongly admitted). Use `assertNoJudgeVerdict` to gate\n * steering; use this only where \"is this grounded in observable evidence\" is the\n * literal question. */\nexport function isTraceObservable(finding: AnalystFinding): boolean {\n return finding.evidence_refs.some((ref) => OBSERVABLE_KINDS.has(ref.kind))\n}\n\n/** True iff the finding is a JUDGE VERDICT (an acceptance score lifted into a\n * finding), identified by provenance set at the lift site — independent of\n * whatever evidence it cites. */\nexport function isJudgeVerdict(finding: AnalystFinding): boolean {\n return finding.derived_from_judge === true\n}\n\n/**\n * THE steer firewall. Fail-loud guard for any path that admits analyst findings\n * as STEERING input (the `f(trace)` role): rejects — naming the offenders — any\n * finding whose provenance is a judge verdict, rather than let `J` leak into the\n * loop. Returns the findings unchanged for chaining.\n *\n * Call this at the chokepoint where a detector that ALSO scores/gates has its\n * findings turned into a steer (the judge-and-steer dual-role case). It keys on\n * provenance, so it correctly admits evidence-less trace-analyst observations and\n * correctly rejects an artifact-citing judge verdict — the cases an evidence\n * check gets backwards.\n *\n * It is necessary, not sufficient: it stops PROVENANCE-tagged verdicts. A judge\n * whose output is laundered through a hand-built finding with no provenance flag\n * is out of its reach — provenance must be honestly set at every judge→finding\n * lift (today: createJudgeAdapter). That is why the integrity rule lives at the\n * lift site, and why ProposeContext.judgeScores?: never is the complementary\n * compile-time tripwire on the obvious direct channel.\n */\nexport function assertNoJudgeVerdict(\n findings: ReadonlyArray<AnalystFinding>,\n context = 'steer',\n): ReadonlyArray<AnalystFinding> {\n const leaks = findings.filter(isJudgeVerdict)\n if (leaks.length > 0) {\n throw new Error(\n `${context}: a judge verdict cannot be admitted as steering input — that is the ` +\n `held-out judge leaking into the loop. Offending judge-derived findings: [${leaks\n .map((f) => f.finding_id)\n .join(', ')}]. Steering consumes observations of behavior, never acceptance verdicts.`,\n )\n }\n return findings\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAoCA,IAAM,cAAc;AAIb,SAAS,aAAa,GAAmC;AAC9D,UAAQ,GAAG;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,EACX;AACF;AAeO,SAAS,sBAA2B,MAA8C;AACvF,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,EAAE,MAAM,gBAAgB;AAAA,IAC9B,SAAS,YAAY,WAAW;AAAA,IAChC,MAAM,QAAQ,KAAK,KAAK;AACtB,YAAM,SAAS,MAAM,KAAK,SAAS,IAAI,EAAE,KAAK,GAAG,KAAK,QAAQ,CAAC;AAC/D,YAAM,MAAwB,CAAC;AAC/B,iBAAW,SAAS,OAAO,QAAQ;AACjC,mBAAW,WAAW,MAAM,UAAU;AACpC,cAAI,KAAK,iBAAiB,IAAI,MAAM,MAAM,OAAO,OAAO,CAAC;AAAA,QAC3D;AAGA,YAAI,MAAM,WAAW,UAAU,MAAM,WAAW,WAAW,MAAM,WAAW,WAAW;AACrF,cAAI;AAAA,YACF,YAAY;AAAA,cACV,YAAY;AAAA,cACZ;AAAA,cACA,SAAS,MAAM;AAAA,cACf,OAAO,UAAU,MAAM,KAAK,KAAK,MAAM,MAAM,KAAK,MAAM,UAAU,iBAAiB;AAAA,cACnF,UACE,MAAM,WAAW,UAAU,SAAS,MAAM,WAAW,YAAY,WAAW;AAAA,cAC9E,YAAY;AAAA,cACZ,eAAe,CAAC;AAAA,cAChB,UAAU;AAAA,gBACR,cAAc,MAAM;AAAA,gBACpB,aAAa,MAAM;AAAA,gBACnB,OAAO,MAAM;AAAA,gBACb,aAAa,MAAM;AAAA,cACrB;AAAA,YACF,CAAC;AAAA,UACH;AAAA,QACF;AAAA,MACF;AACA,UAAI,MAAM,qBAAqB;AAAA,QAC7B,QAAQ,OAAO,OAAO;AAAA,QACtB,SAAS,OAAO;AAAA,QAChB,UAAU,OAAO;AAAA,MACnB,CAAC;AACD,aAAO;AAAA,IACT;AAAA,EACF;AACF;AAEA,SAAS,iBACP,YACA,MACA,OACA,GACgB;AAChB,SAAO,YAAY;AAAA,IACjB;AAAA,IACA;AAAA,IACA,SAAS,EAAE,SAAS;AAAA,IACpB,OAAO,EAAE;AAAA,IACT,UAAU,aAAa,EAAE,QAAQ;AAAA,IACjC,YAAY;AAAA,IACZ,eAAe,EAAE,WACb,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC,IAClE,CAAC;AAAA,IACL,UAAU,EAAE;AAAA,EACd,CAAC;AACH;AAYO,SAAS,uBAAuB,OAA6B,CAAC,GAAsB;AACzF,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,QAAM,SAAS,KAAK,UAAU,IAAI,UAAU;AAC5C,QAAM,YAAY,KAAK,aAAa;AACpC,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,EAAE,MAAM,gBAAgB;AAAA,IAC9B,SAAS,cAAc,WAAW;AAAA,IAClC,MAAM,QAAQ,OAAO;AACnB,YAAM,QAAQ,OAAO,WAAW,KAAK;AACrC,YAAM,MAAwB,CAAC;AAC/B,YAAM,OAA6D;AAAA,QACjE,CAAC,WAAW,YAAY,mCAAmC;AAAA,QAC3D,CAAC,gBAAgB,QAAQ,sBAAsB;AAAA,QAC/C,CAAC,oBAAoB,QAAQ,6CAA6C;AAAA,QAC1E,CAAC,kBAAkB,UAAU,yBAAyB;AAAA,QACtD,CAAC,gBAAgB,UAAU,6BAA6B;AAAA,QACxD,CAAC,eAAe,QAAQ,6BAA6B;AAAA,QACrD,CAAC,aAAa,YAAY,wBAAwB;AAAA,MACpD;AACA,iBAAW,CAAC,KAAK,KAAK,GAAG,KAAK,MAAM;AAClC,cAAM,QAAQ,MAAM,GAAG;AACvB,YAAI,OAAO,UAAU,YAAY,QAAQ,WAAW;AAClD,cAAI;AAAA,YACF,YAAY;AAAA,cACV,YAAY;AAAA,cACZ;AAAA,cACA,SAAS;AAAA,cACT,OAAO;AAAA,cACP,WAAW,GAAG,GAAG,IAAI,MAAM,QAAQ,CAAC,CAAC,oBAAoB,SAAS;AAAA,cAClE,UAAU;AAAA,cACV,YAAY;AAAA,cACZ,eAAe,CAAC;AAAA,cAChB,UAAU,EAAE,WAAW,KAAK,OAAO,WAAW,QAAQ,MAAM,IAAI,MAAM;AAAA,YACxE,CAAC;AAAA,UACH;AAAA,QACF;AAAA,MACF;AAEA,UAAI,MAAM,eAAe,IAAI,WAAW;AACtC,YAAI;AAAA,UACF,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,SAAS;AAAA,YACT,OAAO;AAAA,YACP,WAAW,gBAAgB,MAAM,aAAa,QAAQ,CAAC,CAAC;AAAA,YACxD,UAAU;AAAA,YACV,YAAY;AAAA,YACZ,eAAe,CAAC;AAAA,YAChB,UAAU,EAAE,eAAe,MAAM,cAAc,OAAO,MAAM,MAAM;AAAA,UACpE,CAAC;AAAA,QACH;AAAA,MACF;AACA,aAAO;AAAA,IACT;AAAA,EACF;AACF;AAgBO,SAAS,mBAAmB,MAA6C;AAC9E,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,QAAM,YAAY,KAAK,aAAa;AACpC,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,KAAK,QAAQ,EAAE,MAAM,MAAM;AAAA,IACjC,SAAS,SAAS,WAAW;AAAA,IAC7B,MAAM,QAAQ,OAAO;AACnB,YAAM,SAAS,MAAM,KAAK,MAAM,KAAK,QAAQ,KAAK;AAClD,aAAO,OACJ,OAAO,CAAC,MAAM,YAAY,EAAE,KAAK,IAAI,SAAS,EAC9C,IAAI,CAAC,MAAM,eAAe,IAAI,MAAM,CAAC,CAAC;AAAA,IAC3C;AAAA,EACF;AACF;AAEA,SAAS,YAAY,GAAmB;AAEtC,SAAO,KAAK,IAAI,IAAI,KAAK;AAC3B;AAEA,SAAS,eAAe,YAAoB,MAAc,GAA+B;AACvF,QAAM,UAAU,YAAY,EAAE,KAAK;AACnC,QAAM,WACJ,UAAU,IAAI,aAAa,UAAU,IAAI,SAAS,UAAU,IAAI,WAAW;AAC7E,SAAO,YAAY;AAAA,IACjB;AAAA,IACA;AAAA,IACA,SAAS,EAAE;AAAA,IACX,OAAO,GAAG,EAAE,SAAS,IAAI,EAAE,SAAS,WAAW,QAAQ,QAAQ,CAAC,CAAC;AAAA,IACjE,WAAW,EAAE;AAAA,IACb;AAAA,IACA,YAAY;AAAA,IACZ,eAAe,EAAE,WACb,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC,IAClE,CAAC;AAAA;AAAA;AAAA;AAAA;AAAA,IAKL,oBAAoB;AAAA,IACpB,UAAU,EAAE,YAAY,EAAE,WAAW,WAAW,EAAE,WAAW,UAAU,QAAQ;AAAA,EACjF,CAAC;AACH;AAUO,SAAS,kCACd,OAAwC,CAAC,GACL;AACpC,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,EAAE,MAAM,OAAO,QAAQ,KAAK,SAAS,QAAQ,CAAC,KAAK,QAAQ,KAAK,IAAI,OAAU;AAAA,IACpF,SAAS,GAAG,8BAA8B,YAAY,WAAW;AAAA,IACjE,MAAM,QAAQ,OAAO;AACnB,YAAM,SAAS,MAAM,wBAAwB,OAAO,KAAK,OAAO;AAChE,UAAI,CAAC,OAAO,WAAW;AACrB,eAAO;AAAA,UACL,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,OAAO;AAAA,YACP,WAAW,OAAO;AAAA,YAClB,UAAU;AAAA,YACV,YAAY;AAAA,YACZ,eAAe,CAAC;AAAA,YAChB,UAAU,EAAE,QAAQ,OAAO,MAAM;AAAA,UACnC,CAAC;AAAA,QACH;AAAA,MACF;AACA,YAAM,MAAwB,CAAC;AAC/B,iBAAW,KAAK,OAAO,UAAU;AAG/B,YAAI,EAAE,WAAW,EAAE,SAAS,EAAG;AAC/B,YAAI;AAAA,UACF,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,SAAS,EAAE;AAAA,YACX,OAAO,EAAE,UACL,YAAY,EAAE,OAAO,cAAc,EAAE,KAAK,SAC1C,YAAY,EAAE,OAAO;AAAA,YACzB,WAAW,EAAE;AAAA,YACb,UAAU,aAAa,EAAE,QAAQ;AAAA,YACjC,YAAY;AAAA,YACZ,eAAe,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC;AAAA,YACjF,UAAU;AAAA,cACR,SAAS,EAAE;AAAA,cACX,SAAS,EAAE;AAAA,cACX,UAAU,EAAE;AAAA,cACZ,UAAU,OAAO,WAAW;AAAA,YAC9B;AAAA,UACF,CAAC;AAAA,QACH;AAAA,MACF;AACA,aAAO;AAAA,IACT;AAAA,EACF;AACF;;;ACzTA,IAAM,mBAAqD,oBAAI,IAAyB;AAAA,EACtF;AAAA,EACA;AAAA,EACA;AACF,CAAC;AAUM,SAAS,kBAAkB,SAAkC;AAClE,SAAO,QAAQ,cAAc,KAAK,CAAC,QAAQ,iBAAiB,IAAI,IAAI,IAAI,CAAC;AAC3E;AAKO,SAAS,eAAe,SAAkC;AAC/D,SAAO,QAAQ,uBAAuB;AACxC;AAqBO,SAAS,qBACd,UACA,UAAU,SACqB;AAC/B,QAAM,QAAQ,SAAS,OAAO,cAAc;AAC5C,MAAI,MAAM,SAAS,GAAG;AACpB,UAAM,IAAI;AAAA,MACR,GAAG,OAAO,sJACoE,MACzE,IAAI,CAAC,MAAM,EAAE,UAAU,EACvB,KAAK,IAAI,CAAC;AAAA,IACjB;AAAA,EACF;AACA,SAAO;AACT;","names":[]}
|
|
1
|
+
{"version":3,"sources":["../../src/analyst/adapters.ts"],"sourcesContent":["/**\n * Adapter factories — lift each existing agent-eval primitive into the\n * Analyst contract without re-implementing it.\n *\n * Five primitives, five factories. Each one:\n * - Builds an Analyst with a stable id (caller chooses; defaults\n * given), a sensible default `inputKind`, a version derived from\n * the wrapped primitive's version + an adapter revision, and an\n * `analyze()` that calls the primitive and lifts its output to\n * AnalystFinding[] using `makeFinding()`.\n * - Maps severities: the existing `Severity` ('critical' | 'major' |\n * 'minor' | 'info') projects onto AnalystSeverity ('critical' |\n * 'high' | 'medium' | 'low' | 'info'); 'major' → 'high', 'minor' →\n * 'medium'. Domain analysts that want finer-grained mapping override.\n *\n * Adapters never own state. Calling the same factory twice with the\n * same primitive instance is safe.\n */\n\nimport type {\n Finding as LayerFinding,\n Severity as LayerSeverity,\n MultiLayerVerifier,\n VerifyOptions,\n} from '../multi-layer-verifier'\nimport { RunCritic, type RunTrace } from '../run-critic'\nimport {\n runSemanticConceptJudge,\n SEMANTIC_CONCEPT_JUDGE_VERSION,\n type SemanticConceptJudgeInput,\n type SemanticConceptJudgeOptions,\n} from '../semantic-concept-judge'\nimport type { JudgeFn, JudgeInput, JudgeScore, TCloud } from '../types'\nimport type { Analyst, AnalystFinding, AnalystSeverity } from './types'\nimport { makeFinding } from './types'\n\nconst ADAPTER_REV = '1'\n\n// ── Severity bridges ───────────────────────────────────────────────\n\nexport function liftSeverity(s: LayerSeverity): AnalystSeverity {\n switch (s) {\n case 'critical':\n return 'critical'\n case 'major':\n return 'high'\n case 'minor':\n return 'medium'\n case 'info':\n return 'info'\n }\n}\n\n// ── 1. MultiLayerVerifier → Analyst ─────────────────────────────────\n\nexport interface VerifierAdapterOpts<Env> {\n id?: string\n area?: string\n verifier: MultiLayerVerifier<Env>\n /**\n * The verifier expects an `env` per run. Adapters take it from\n * `AnalystRunInputs.custom[<id>]` via the registry's 'custom' routing.\n */\n options?: Omit<VerifyOptions<Env>, 'env'>\n}\n\nexport function createVerifierAdapter<Env>(opts: VerifierAdapterOpts<Env>): Analyst<Env> {\n const id = opts.id ?? 'multi-layer-verifier'\n const area = opts.area ?? 'verification'\n return {\n id,\n description:\n \"Runs a MultiLayerVerifier and lifts each layer's findings into the analyst envelope.\",\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `verifier-${ADAPTER_REV}`,\n async analyze(env, ctx) {\n const report = await opts.verifier.run({ env, ...opts.options })\n const out: AnalystFinding[] = []\n for (const layer of report.layers) {\n for (const finding of layer.findings) {\n out.push(liftLayerFinding(id, area, layer.layer, finding))\n }\n // Layer-level signal: a failed/error layer is itself a finding\n // even if it didn't emit per-finding rows.\n if (layer.status === 'fail' || layer.status === 'error' || layer.status === 'timeout') {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: layer.layer,\n claim: `layer \"${layer.layer}\" ${layer.status}: ${layer.reason ?? 'no reason given'}`,\n severity:\n layer.status === 'error' ? 'high' : layer.status === 'timeout' ? 'medium' : 'high',\n confidence: 1,\n evidence_refs: [],\n metadata: {\n layer_status: layer.status,\n duration_ms: layer.durationMs,\n score: layer.score,\n diagnostics: layer.diagnostics,\n },\n }),\n )\n }\n }\n ctx.log?.('verifier complete', {\n layers: report.layers.length,\n blended: report.blendedScore,\n all_pass: report.allPass,\n })\n return out\n },\n }\n}\n\nfunction liftLayerFinding(\n analyst_id: string,\n area: string,\n layer: string,\n f: LayerFinding,\n): AnalystFinding {\n return makeFinding({\n analyst_id,\n area,\n subject: f.layer ?? layer,\n claim: f.message,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: f.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }]\n : [],\n metadata: f.detail,\n })\n}\n\n// ── 2. RunCritic → Analyst ──────────────────────────────────────────\n\nexport interface RunCriticAdapterOpts {\n id?: string\n area?: string\n critic?: RunCritic\n /** Optional threshold below which a dimension is reported as a finding. Default 0.5. */\n threshold?: number\n}\n\nexport function createRunCriticAdapter(opts: RunCriticAdapterOpts = {}): Analyst<RunTrace> {\n const id = opts.id ?? 'run-critic'\n const area = opts.area ?? 'run-quality'\n const critic = opts.critic ?? new RunCritic()\n const threshold = opts.threshold ?? 0.5\n return {\n id,\n description:\n 'Scores a single run across success / grounding / drift / tool-quality and surfaces below-threshold dimensions.',\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `run-critic-${ADAPTER_REV}`,\n async analyze(trace) {\n const score = critic.scoreTrace(trace)\n const out: AnalystFinding[] = []\n const dims: Array<[keyof typeof score, AnalystSeverity, string]> = [\n ['success', 'critical', 'run did not complete successfully'],\n ['goalProgress', 'high', 'goal progress is low'],\n ['repoGroundedness', 'high', 'output is poorly grounded in the repository'],\n ['toolUseQuality', 'medium', 'tool use quality is low'],\n ['patchQuality', 'medium', 'no real patch/edit evidence'],\n ['testReality', 'high', 'no real test/build evidence'],\n ['finalGate', 'critical', 'final gate is blocking'],\n ]\n for (const [dim, sev, msg] of dims) {\n const value = score[dim] as number\n if (typeof value === 'number' && value < threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: dim,\n claim: msg,\n rationale: `${dim}=${value.toFixed(2)} below threshold ${threshold}`,\n severity: sev,\n confidence: 1,\n evidence_refs: [],\n metadata: { dimension: dim, value, threshold, run_id: trace.run.runId },\n }),\n )\n }\n }\n // Drift penalty is high → surface as a finding (inverse threshold).\n if (score.driftPenalty > 1 - threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: 'drift',\n claim: 'agent output drifted from repository signal',\n rationale: `driftPenalty=${score.driftPenalty.toFixed(2)}`,\n severity: 'medium',\n confidence: 0.9,\n evidence_refs: [],\n metadata: { drift_penalty: score.driftPenalty, notes: score.notes },\n }),\n )\n }\n return out\n },\n }\n}\n\n// ── 3. JudgeFn → Analyst ────────────────────────────────────────────\n\nexport interface JudgeAdapterOpts {\n id?: string\n area?: string\n judge: JudgeFn\n /** TCloud handle the JudgeFn calls. */\n tcloud: TCloud\n /** Optional cost classification — most judges call an LLM. */\n cost?: Analyst['cost']\n /** Optional threshold below which a JudgeScore becomes a finding. Default 6 (on 0-10 scale). */\n threshold?: number\n}\n\nexport function createJudgeAdapter(opts: JudgeAdapterOpts): Analyst<JudgeInput> {\n const id = opts.id ?? 'judge'\n const area = opts.area ?? 'judge'\n const threshold = opts.threshold ?? 6\n return {\n id,\n description:\n 'Wraps an agent-eval JudgeFn into an analyst; below-threshold dimensions surface as findings.',\n inputKind: 'judge-input',\n cost: opts.cost ?? { kind: 'llm' },\n version: `judge-${ADAPTER_REV}`,\n async analyze(input) {\n const scores = await opts.judge(opts.tcloud, input)\n return scores\n .filter((s) => normalize10(s.score) < threshold)\n .map((s) => liftJudgeScore(id, area, s))\n },\n }\n}\n\nfunction normalize10(s: number): number {\n // JudgeScore convention is 0-10 but some judges emit 0-1. Coerce to 0-10.\n return s <= 1 ? s * 10 : s\n}\n\nfunction liftJudgeScore(analyst_id: string, area: string, s: JudgeScore): AnalystFinding {\n const score10 = normalize10(s.score)\n const severity: AnalystSeverity =\n score10 < 3 ? 'critical' : score10 < 5 ? 'high' : score10 < 7 ? 'medium' : 'low'\n return makeFinding({\n analyst_id,\n area,\n subject: s.dimension,\n claim: `${s.judgeName}/${s.dimension} scored ${score10.toFixed(1)}/10`,\n rationale: s.reasoning,\n severity,\n confidence: 0.8,\n evidence_refs: s.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: s.evidence }]\n : [],\n // Provenance: this finding IS a judge verdict (an acceptance score), not an\n // observation of behavior. The steer firewall (assertNoJudgeVerdict) rejects\n // it from steering — even when it cites an artifact above — because letting a\n // verdict steer the next attempt is the held-out judge leaking into the loop.\n derived_from_judge: true,\n metadata: { judge_name: s.judgeName, dimension: s.dimension, score_10: score10 },\n })\n}\n\n// ── 4. SemanticConceptJudge → Analyst ──────────────────────────────\n\nexport interface SemanticConceptJudgeAdapterOpts {\n id?: string\n area?: string\n options?: SemanticConceptJudgeOptions\n}\n\nexport function createSemanticConceptJudgeAdapter(\n opts: SemanticConceptJudgeAdapterOpts = {},\n): Analyst<SemanticConceptJudgeInput> {\n const id = opts.id ?? 'semantic-concept-judge'\n const area = opts.area ?? 'concept-coverage'\n return {\n id,\n description:\n 'Runs the semantic-concept judge and surfaces missing / weak concepts as findings.',\n inputKind: 'custom',\n cost: { kind: 'llm', models: opts.options?.model ? [opts.options.model] : undefined },\n version: `${SEMANTIC_CONCEPT_JUDGE_VERSION}-adapter-${ADAPTER_REV}`,\n async analyze(input) {\n const result = await runSemanticConceptJudge(input, opts.options)\n if (!result.available) {\n return [\n makeFinding({\n analyst_id: id,\n area,\n claim: 'semantic-concept judge unavailable',\n rationale: result.error,\n severity: 'info',\n confidence: 1,\n evidence_refs: [],\n metadata: { reason: result.error },\n }),\n ]\n }\n const out: AnalystFinding[] = []\n for (const f of result.findings) {\n // Only surface gaps: missing concepts or low scores. Concepts at\n // 7+/10 with present=true are not findings — they're successes.\n if (f.present && f.score >= 7) continue\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: f.concept,\n claim: f.present\n ? `concept \"${f.concept}\" is weak (${f.score}/10)`\n : `concept \"${f.concept}\" is missing`,\n rationale: f.evidence,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }],\n metadata: {\n concept: f.concept,\n present: f.present,\n score_10: f.score,\n cost_usd: result.costUsd ?? undefined,\n },\n }),\n )\n }\n return out\n },\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAoCA,IAAM,cAAc;AAIb,SAAS,aAAa,GAAmC;AAC9D,UAAQ,GAAG;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,EACX;AACF;AAeO,SAAS,sBAA2B,MAA8C;AACvF,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,EAAE,MAAM,gBAAgB;AAAA,IAC9B,SAAS,YAAY,WAAW;AAAA,IAChC,MAAM,QAAQ,KAAK,KAAK;AACtB,YAAM,SAAS,MAAM,KAAK,SAAS,IAAI,EAAE,KAAK,GAAG,KAAK,QAAQ,CAAC;AAC/D,YAAM,MAAwB,CAAC;AAC/B,iBAAW,SAAS,OAAO,QAAQ;AACjC,mBAAW,WAAW,MAAM,UAAU;AACpC,cAAI,KAAK,iBAAiB,IAAI,MAAM,MAAM,OAAO,OAAO,CAAC;AAAA,QAC3D;AAGA,YAAI,MAAM,WAAW,UAAU,MAAM,WAAW,WAAW,MAAM,WAAW,WAAW;AACrF,cAAI;AAAA,YACF,YAAY;AAAA,cACV,YAAY;AAAA,cACZ;AAAA,cACA,SAAS,MAAM;AAAA,cACf,OAAO,UAAU,MAAM,KAAK,KAAK,MAAM,MAAM,KAAK,MAAM,UAAU,iBAAiB;AAAA,cACnF,UACE,MAAM,WAAW,UAAU,SAAS,MAAM,WAAW,YAAY,WAAW;AAAA,cAC9E,YAAY;AAAA,cACZ,eAAe,CAAC;AAAA,cAChB,UAAU;AAAA,gBACR,cAAc,MAAM;AAAA,gBACpB,aAAa,MAAM;AAAA,gBACnB,OAAO,MAAM;AAAA,gBACb,aAAa,MAAM;AAAA,cACrB;AAAA,YACF,CAAC;AAAA,UACH;AAAA,QACF;AAAA,MACF;AACA,UAAI,MAAM,qBAAqB;AAAA,QAC7B,QAAQ,OAAO,OAAO;AAAA,QACtB,SAAS,OAAO;AAAA,QAChB,UAAU,OAAO;AAAA,MACnB,CAAC;AACD,aAAO;AAAA,IACT;AAAA,EACF;AACF;AAEA,SAAS,iBACP,YACA,MACA,OACA,GACgB;AAChB,SAAO,YAAY;AAAA,IACjB;AAAA,IACA;AAAA,IACA,SAAS,EAAE,SAAS;AAAA,IACpB,OAAO,EAAE;AAAA,IACT,UAAU,aAAa,EAAE,QAAQ;AAAA,IACjC,YAAY;AAAA,IACZ,eAAe,EAAE,WACb,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC,IAClE,CAAC;AAAA,IACL,UAAU,EAAE;AAAA,EACd,CAAC;AACH;AAYO,SAAS,uBAAuB,OAA6B,CAAC,GAAsB;AACzF,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,QAAM,SAAS,KAAK,UAAU,IAAI,UAAU;AAC5C,QAAM,YAAY,KAAK,aAAa;AACpC,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,EAAE,MAAM,gBAAgB;AAAA,IAC9B,SAAS,cAAc,WAAW;AAAA,IAClC,MAAM,QAAQ,OAAO;AACnB,YAAM,QAAQ,OAAO,WAAW,KAAK;AACrC,YAAM,MAAwB,CAAC;AAC/B,YAAM,OAA6D;AAAA,QACjE,CAAC,WAAW,YAAY,mCAAmC;AAAA,QAC3D,CAAC,gBAAgB,QAAQ,sBAAsB;AAAA,QAC/C,CAAC,oBAAoB,QAAQ,6CAA6C;AAAA,QAC1E,CAAC,kBAAkB,UAAU,yBAAyB;AAAA,QACtD,CAAC,gBAAgB,UAAU,6BAA6B;AAAA,QACxD,CAAC,eAAe,QAAQ,6BAA6B;AAAA,QACrD,CAAC,aAAa,YAAY,wBAAwB;AAAA,MACpD;AACA,iBAAW,CAAC,KAAK,KAAK,GAAG,KAAK,MAAM;AAClC,cAAM,QAAQ,MAAM,GAAG;AACvB,YAAI,OAAO,UAAU,YAAY,QAAQ,WAAW;AAClD,cAAI;AAAA,YACF,YAAY;AAAA,cACV,YAAY;AAAA,cACZ;AAAA,cACA,SAAS;AAAA,cACT,OAAO;AAAA,cACP,WAAW,GAAG,GAAG,IAAI,MAAM,QAAQ,CAAC,CAAC,oBAAoB,SAAS;AAAA,cAClE,UAAU;AAAA,cACV,YAAY;AAAA,cACZ,eAAe,CAAC;AAAA,cAChB,UAAU,EAAE,WAAW,KAAK,OAAO,WAAW,QAAQ,MAAM,IAAI,MAAM;AAAA,YACxE,CAAC;AAAA,UACH;AAAA,QACF;AAAA,MACF;AAEA,UAAI,MAAM,eAAe,IAAI,WAAW;AACtC,YAAI;AAAA,UACF,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,SAAS;AAAA,YACT,OAAO;AAAA,YACP,WAAW,gBAAgB,MAAM,aAAa,QAAQ,CAAC,CAAC;AAAA,YACxD,UAAU;AAAA,YACV,YAAY;AAAA,YACZ,eAAe,CAAC;AAAA,YAChB,UAAU,EAAE,eAAe,MAAM,cAAc,OAAO,MAAM,MAAM;AAAA,UACpE,CAAC;AAAA,QACH;AAAA,MACF;AACA,aAAO;AAAA,IACT;AAAA,EACF;AACF;AAgBO,SAAS,mBAAmB,MAA6C;AAC9E,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,QAAM,YAAY,KAAK,aAAa;AACpC,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,KAAK,QAAQ,EAAE,MAAM,MAAM;AAAA,IACjC,SAAS,SAAS,WAAW;AAAA,IAC7B,MAAM,QAAQ,OAAO;AACnB,YAAM,SAAS,MAAM,KAAK,MAAM,KAAK,QAAQ,KAAK;AAClD,aAAO,OACJ,OAAO,CAAC,MAAM,YAAY,EAAE,KAAK,IAAI,SAAS,EAC9C,IAAI,CAAC,MAAM,eAAe,IAAI,MAAM,CAAC,CAAC;AAAA,IAC3C;AAAA,EACF;AACF;AAEA,SAAS,YAAY,GAAmB;AAEtC,SAAO,KAAK,IAAI,IAAI,KAAK;AAC3B;AAEA,SAAS,eAAe,YAAoB,MAAc,GAA+B;AACvF,QAAM,UAAU,YAAY,EAAE,KAAK;AACnC,QAAM,WACJ,UAAU,IAAI,aAAa,UAAU,IAAI,SAAS,UAAU,IAAI,WAAW;AAC7E,SAAO,YAAY;AAAA,IACjB;AAAA,IACA;AAAA,IACA,SAAS,EAAE;AAAA,IACX,OAAO,GAAG,EAAE,SAAS,IAAI,EAAE,SAAS,WAAW,QAAQ,QAAQ,CAAC,CAAC;AAAA,IACjE,WAAW,EAAE;AAAA,IACb;AAAA,IACA,YAAY;AAAA,IACZ,eAAe,EAAE,WACb,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC,IAClE,CAAC;AAAA;AAAA;AAAA;AAAA;AAAA,IAKL,oBAAoB;AAAA,IACpB,UAAU,EAAE,YAAY,EAAE,WAAW,WAAW,EAAE,WAAW,UAAU,QAAQ;AAAA,EACjF,CAAC;AACH;AAUO,SAAS,kCACd,OAAwC,CAAC,GACL;AACpC,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,EAAE,MAAM,OAAO,QAAQ,KAAK,SAAS,QAAQ,CAAC,KAAK,QAAQ,KAAK,IAAI,OAAU;AAAA,IACpF,SAAS,GAAG,8BAA8B,YAAY,WAAW;AAAA,IACjE,MAAM,QAAQ,OAAO;AACnB,YAAM,SAAS,MAAM,wBAAwB,OAAO,KAAK,OAAO;AAChE,UAAI,CAAC,OAAO,WAAW;AACrB,eAAO;AAAA,UACL,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,OAAO;AAAA,YACP,WAAW,OAAO;AAAA,YAClB,UAAU;AAAA,YACV,YAAY;AAAA,YACZ,eAAe,CAAC;AAAA,YAChB,UAAU,EAAE,QAAQ,OAAO,MAAM;AAAA,UACnC,CAAC;AAAA,QACH;AAAA,MACF;AACA,YAAM,MAAwB,CAAC;AAC/B,iBAAW,KAAK,OAAO,UAAU;AAG/B,YAAI,EAAE,WAAW,EAAE,SAAS,EAAG;AAC/B,YAAI;AAAA,UACF,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,SAAS,EAAE;AAAA,YACX,OAAO,EAAE,UACL,YAAY,EAAE,OAAO,cAAc,EAAE,KAAK,SAC1C,YAAY,EAAE,OAAO;AAAA,YACzB,WAAW,EAAE;AAAA,YACb,UAAU,aAAa,EAAE,QAAQ;AAAA,YACjC,YAAY;AAAA,YACZ,eAAe,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC;AAAA,YACjF,UAAU;AAAA,cACR,SAAS,EAAE;AAAA,cACX,SAAS,EAAE;AAAA,cACX,UAAU,EAAE;AAAA,cACZ,UAAU,OAAO,WAAW;AAAA,YAC9B;AAAA,UACF,CAAC;AAAA,QACH;AAAA,MACF;AACA,aAAO;AAAA,IACT;AAAA,EACF;AACF;","names":[]}
|
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
import { A as AnalystRegistry } from './default-registry-
|
|
1
|
+
import { A as AnalystRegistry } from './default-registry-Dhrc__SE.js';
|
|
2
2
|
import { a as DatasetScenario } from './dataset-BbGkaN2I.js';
|
|
3
|
-
import { R as RunRecord } from './run-record-
|
|
4
|
-
import { I as InsightReport } from './insight-report-
|
|
3
|
+
import { R as RunRecord } from './run-record-DEwidcqn.js';
|
|
4
|
+
import { I as InsightReport } from './insight-report-C02J3q4T.js';
|
|
5
5
|
|
|
6
6
|
/**
|
|
7
7
|
* # `analyzeRuns()` — turn a set of agent runs into an actionable decision packet.
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
import { c as CalibrationReport } from '../calibration-BPmzuVPk.js';
|
|
2
2
|
import { O as OffPolicyEstimate, a as OffPolicyOptions, b as OffPolicyTrajectory } from '../off-policy-DiwuKKg7.js';
|
|
3
|
-
import { d as CodeAgentSessionSource, a as CodeAgentSessionIntakeOptions, c as CodeAgentSessionMetrics, C as CodeAgentSessionDiagnostic } from '../code-agent-session-
|
|
4
|
-
import { R as RunRecord,
|
|
3
|
+
import { d as CodeAgentSessionSource, a as CodeAgentSessionIntakeOptions, c as CodeAgentSessionMetrics, C as CodeAgentSessionDiagnostic } from '../code-agent-session-B6ZcDwyA.js';
|
|
4
|
+
import { R as RunRecord, b as RunSplitTag } from '../run-record-DEwidcqn.js';
|
|
5
5
|
import { T as TraceStore } from '../store-BcFXE6LG.js';
|
|
6
|
-
import { R as RuntimeTrajectoryRecord, P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection } from '../runtime-trajectory-
|
|
6
|
+
import { R as RuntimeTrajectoryRecord, P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection } from '../runtime-trajectory-OJDaTYHN.js';
|
|
7
7
|
import '../schema-m0gsnbt3.js';
|
|
8
8
|
import '../outcome-store-rnXLEqSn.js';
|
|
9
9
|
import '@tangle-network/agent-interface';
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as deterministicSplit, e as routing } from '../index-
|
|
2
|
-
import '../run-record-
|
|
1
|
+
export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as deterministicSplit, e as routing } from '../index-W96macmS.js';
|
|
2
|
+
import '../run-record-DEwidcqn.js';
|
|
3
3
|
import '@tangle-network/agent-interface';
|
|
4
4
|
import '../errors-CzMUYo7b.js';
|
|
5
5
|
import '../schema-m0gsnbt3.js';
|
package/dist/campaign/index.d.ts
CHANGED
|
@@ -1,18 +1,19 @@
|
|
|
1
|
-
import { S as SignedManifest, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, P as ProducedState, b as CorrectnessChecker } from '../pre-registration-
|
|
2
|
-
export { L as LlmJudgeDimension, c as LlmJudgeOptions, l as llmJudge } from '../pre-registration-
|
|
1
|
+
import { S as SignedManifest, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, P as ProducedState, b as CorrectnessChecker } from '../pre-registration-DB8oDqZJ.js';
|
|
2
|
+
export { L as LlmJudgeDimension, c as LlmJudgeOptions, l as llmJudge } from '../pre-registration-DB8oDqZJ.js';
|
|
3
3
|
import { A as AnalyzeTracesOptions, a as AnalyzeTracesInput, b as AnalyzeTracesResult } from '../analyst-C8HHvfJp.js';
|
|
4
|
-
import { S as Scenario, M as MutableSurface, b as DispatchContext, a as JudgeConfig, e as GenerationRecord, g as Gate, J as JudgeScore, L as LabeledScenarioStore, s as LabeledScenarioWrite, t as LabeledScenarioSampleArgs, u as LabeledScenarioRecord, v as LabelTrust, f as SurfaceProposer, w as ProposedCandidate, x as ProposeContext, y as LabeledScenarioSource, C as CampaignResult, o as CodeSurface } from '../types-
|
|
5
|
-
export { k as CampaignAggregates, l as CampaignArtifactWriter, m as CampaignCellResult, n as CampaignCostMeter, z as CampaignTokenUsage, d as CampaignTraceWriter, D as DispatchFn, h as GateContext, j as GateDecision, G as GateResult, p as GenerationCandidate, A as JudgeAggregate, c as JudgeDimension, i as Mutator, O as OptimizationProposer, q as OptimizerConfig, P as ParetoParent, R as RedactionStatus, B as ScenarioAggregate, r as SessionScript, T as TraceSpan, E as isProposedCandidate, F as labelTrustRank } from '../types-
|
|
6
|
-
import { R as RunCampaignOptions, c as RunImprovementLoopOptions, C as CampaignStorage } from '../gepa-
|
|
7
|
-
export { e as GepaProposerConstraints, G as GepaProposerOptions, O as OpenAutoPrOptions, h as OpenAutoPrResult, b as RunImprovementLoopResult, a as RunOptimizationOptions, j as RunOptimizationResult, k as countSentenceEdits, l as defaultRenderDiff, m as extractH2Sections, f as fsCampaignStorage, g as gepaProposer, i as inMemoryCampaignStorage, o as openAutoPr, r as runCampaign, d as runImprovementLoop, n as runOptimization, s as surfaceHash } from '../gepa-
|
|
8
|
-
export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, k as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, l as EmitLoopProvenanceArgs, m as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, n as LoopProvenanceBackend, o as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, P as ParetoSignificanceGateOptions, c as PromotionObjective, d as PromotionPolicy, R as RunEvalOptions, e as buildEvidenceVector, q as buildLoopProvenanceRecord, f as composeGate, g as defaultProductionGate, s as emitLoopProvenance, h as evolutionaryProposer, i as heldOutGate, t as loopProvenanceSpans, p as paretoPolicy, j as paretoSignificanceGate, u as provenanceRecordPath, v as provenanceSpansPath, r as runEval, w as surfaceContentHash } from '../provenance-
|
|
4
|
+
import { S as Scenario, M as MutableSurface, b as DispatchContext, a as JudgeConfig, e as GenerationRecord, g as Gate, J as JudgeScore, L as LabeledScenarioStore, s as LabeledScenarioWrite, t as LabeledScenarioSampleArgs, u as LabeledScenarioRecord, v as LabelTrust, f as SurfaceProposer, w as ProposedCandidate, x as ProposeContext, y as LabeledScenarioSource, C as CampaignResult, o as CodeSurface } from '../types-Cv1bo4_a.js';
|
|
5
|
+
export { k as CampaignAggregates, l as CampaignArtifactWriter, m as CampaignCellResult, n as CampaignCostMeter, z as CampaignTokenUsage, d as CampaignTraceWriter, D as DispatchFn, h as GateContext, j as GateDecision, G as GateResult, p as GenerationCandidate, A as JudgeAggregate, c as JudgeDimension, i as Mutator, O as OptimizationProposer, q as OptimizerConfig, P as ParetoParent, R as RedactionStatus, B as ScenarioAggregate, r as SessionScript, T as TraceSpan, E as isProposedCandidate, F as labelTrustRank } from '../types-Cv1bo4_a.js';
|
|
6
|
+
import { R as RunCampaignOptions, c as RunImprovementLoopOptions, C as CampaignStorage } from '../gepa-BRgNnmGZ.js';
|
|
7
|
+
export { e as GepaProposerConstraints, G as GepaProposerOptions, O as OpenAutoPrOptions, h as OpenAutoPrResult, b as RunImprovementLoopResult, a as RunOptimizationOptions, j as RunOptimizationResult, k as countSentenceEdits, l as defaultRenderDiff, m as extractH2Sections, f as fsCampaignStorage, g as gepaProposer, i as inMemoryCampaignStorage, o as openAutoPr, r as runCampaign, d as runImprovementLoop, n as runOptimization, s as surfaceHash } from '../gepa-BRgNnmGZ.js';
|
|
8
|
+
export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, k as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, l as EmitLoopProvenanceArgs, m as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, n as LoopProvenanceBackend, o as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, P as ParetoSignificanceGateOptions, c as PromotionObjective, d as PromotionPolicy, R as RunEvalOptions, e as buildEvidenceVector, q as buildLoopProvenanceRecord, f as composeGate, g as defaultProductionGate, s as emitLoopProvenance, h as evolutionaryProposer, i as heldOutGate, t as loopProvenanceSpans, p as paretoPolicy, j as paretoSignificanceGate, u as provenanceRecordPath, v as provenanceSpansPath, r as runEval, w as surfaceContentHash } from '../provenance-B0SZw1z2.js';
|
|
9
9
|
import { E as EProcessState, a as PairedBootstrapResult } from '../statistics-xP-cWc5k.js';
|
|
10
10
|
import { L as LlmClientOptions } from '../llm-client-Bj7g0rqu.js';
|
|
11
11
|
import { AgentProfile } from '@tangle-network/agent-interface';
|
|
12
12
|
import { A as AgentEvalError } from '../errors-CzMUYo7b.js';
|
|
13
|
-
import {
|
|
14
|
-
import {
|
|
15
|
-
import {
|
|
13
|
+
import { b as RunSplitTag, R as RunRecord } from '../run-record-DEwidcqn.js';
|
|
14
|
+
import { b as PolicyEdit, F as FindingToPolicyEditOptions, d as PolicyEditAdmissionOptions, c as PolicyEditAdmission } from '../policy-edit-Dccm9tyA.js';
|
|
15
|
+
import { T as TraceAnalystKindSpec } from '../kind-factory-OgqQSvLi.js';
|
|
16
|
+
import { A as AnalystFinding } from '../types-BEzCBMQD.js';
|
|
16
17
|
import '@tangle-network/tcloud';
|
|
17
18
|
import '../verdict-C9MlYujm.js';
|
|
18
19
|
import '@ax-llm/ax';
|
|
@@ -23,8 +24,8 @@ import '../store-BcFXE6LG.js';
|
|
|
23
24
|
import '../schema-m0gsnbt3.js';
|
|
24
25
|
import '../pareto-E-pembql.js';
|
|
25
26
|
import '../hosted/index.js';
|
|
26
|
-
import '../insight-report-
|
|
27
|
-
import '../summary-report-
|
|
27
|
+
import '../insight-report-C02J3q4T.js';
|
|
28
|
+
import '../summary-report-C4uzRWh8.js';
|
|
28
29
|
import '../failure-cluster-DH9Flgcf.js';
|
|
29
30
|
import '../judge-calibration-0p2QcWNE.js';
|
|
30
31
|
import '../types-C7DGg5ex.js';
|
|
@@ -1327,6 +1328,29 @@ interface MemoryCurationProposerOptions {
|
|
|
1327
1328
|
/** Build the CURATOR proposer. */
|
|
1328
1329
|
declare function memoryCurationProposer(opts?: MemoryCurationProposerOptions): SurfaceProposer;
|
|
1329
1330
|
|
|
1331
|
+
/**
|
|
1332
|
+
* `policyEditProposer` turns typed analyst policy edits into measured candidate
|
|
1333
|
+
* surfaces. It is deliberately deterministic: analysts propose the edit, this
|
|
1334
|
+
* proposer only checks admission and applies it. The campaign/holdout loop still
|
|
1335
|
+
* decides whether the candidate actually wins.
|
|
1336
|
+
*/
|
|
1337
|
+
|
|
1338
|
+
interface PolicyEditProposerOptions {
|
|
1339
|
+
/** Static edits, useful when an analyst has already emitted typed edits. When
|
|
1340
|
+
* omitted, the proposer reads `ctx.findings`. */
|
|
1341
|
+
edits?: ReadonlyArray<PolicyEdit>;
|
|
1342
|
+
/** Adapter for legacy `AnalystFinding` rows. Findings without typed expected
|
|
1343
|
+
* gain are ignored rather than inflated with fake numbers. */
|
|
1344
|
+
findingOptions?: FindingToPolicyEditOptions;
|
|
1345
|
+
/** Admission threshold. Defaults live in `admitPolicyEdit`. */
|
|
1346
|
+
admission?: PolicyEditAdmissionOptions;
|
|
1347
|
+
/** Candidate cap. Default: `ctx.populationSize`. */
|
|
1348
|
+
maxCandidates?: number;
|
|
1349
|
+
/** Optional callback for audit UIs/tests that need rejected edit reasons. */
|
|
1350
|
+
onAdmission?: (admission: PolicyEditAdmission) => void;
|
|
1351
|
+
}
|
|
1352
|
+
declare function policyEditProposer(opts?: PolicyEditProposerOptions): SurfaceProposer;
|
|
1353
|
+
|
|
1330
1354
|
/**
|
|
1331
1355
|
* `traceAnalystProposer` — wraps agent-eval's OWN trace-analyst engine
|
|
1332
1356
|
* (`AnalystRegistry` over the agentic OTLP reader) as a `SurfaceProposer`.
|
|
@@ -1451,4 +1475,4 @@ declare function gitWorktreeAdapter(opts: GitWorktreeAdapterOptions): WorktreeAd
|
|
|
1451
1475
|
* as a ref under the adapter's worktree dir. */
|
|
1452
1476
|
declare function resolveWorktreePath(surface: CodeSurface, worktreeDir?: string): string;
|
|
1453
1477
|
|
|
1454
|
-
export { type AcceptedEdit, type AceProposerOptions, type AnalystArtifact, type AnalystScenario, type ApplySkillPatchResult, type BuildAnalystSurfaceDispatchOptions, type CampaignBreakdown, CampaignResult, CampaignStorage, CodeSurface, type CompareProposersOptions, type DimensionRegression, DispatchContext, type FailureModeRecallJudgeOptions, type FapoAttributionSignals, type FapoEntryConfig, type FapoFailureCluster, type FapoOptimizationLevel, type FapoProposerOptions, type FapoReviewInput, type FapoReviewIssue, type FapoReviewResult, type FapoScopeContract, FsLabeledScenarioStore, type FsLabeledScenarioStoreOptions, Gate, GenerationRecord, type GitWorktreeAdapterOptions, type HaloProposerOptions, type HeldoutSignificance, type HeldoutSignificanceOptions, type JsonPrimitive, type JsonValue, JudgeConfig, JudgeScore, LabelTrust, LabeledScenarioRecord, LabeledScenarioSampleArgs, LabeledScenarioSource, LabeledScenarioStore, LabeledScenarioStoreError, LabeledScenarioWrite, type MemoryCurationProposerOptions, MutableSurface, type OptimizerEntryConfig, type PairedHoldout, type ParameterCandidate, type ParameterChange, type ParameterSweepProposerOptions, type PlaybackContext, type PlaybackDriver, type PlaybackStep, type ProfileDispatchFn, ProfileMatrixError, type ProfileSummary, ProposeContext, type ProposePatchesArgs, ProposedCandidate, type ProposerComparison, type ProposerEntry, type ProposerPairwise, type ProposerScore, type RejectedEdit, RunCampaignOptions, RunImprovementLoopOptions, type RunProfileMatrixOptions, type RunProfileMatrixResult, type RunSkillOptOptions, type RunSkillOptResult, Scenario, type ScenarioRollup, type ScoreboardRenderOptions, type ScoreboardRow, type ScoreboardSummary, type SequentialDecideFn, type SequentialDecideOptions, type SequentialDecision, type SequentialObservation, type SequentialPairedGate, type SequentialPairedGateOptions, type SkillOptEpochRecord, type SkillOptEvidence, type SkillOptProposer, type SkillOptProposerOptions, type SkillPatch, type SkillPatchOp, SkillPatchParseError, type SkillPatchRejection, SurfaceProposer, type TraceAnalystProposerOptions, type UserStory, type UserStoryVerdict, type Worktree, type WorktreeAdapter, WorktreeAdapterError, aceProposer, applySkillPatch, buildAnalystSurfaceDispatch, campaignBreakdown, campaignMeanComposite, compareProposers, detectScale, dimensionRegressions, extractFapoAttributionSignals, failureModeRecallJudge, fapoEscalationEntry, fapoProposer, gepaParetoEntry, gepaReflectionEntry, gitWorktreeAdapter, haloProposer, heldoutSignificance, makePlaybackDispatch, memoryCurationProposer, pairHoldout, parameterSweepProposer, parseSkillPatchResponse, patchEditCount, renderScoreboardMarkdown, resolveWorktreePath, runProfileMatrix, runSkillOpt, scoreUserStory, scoreboardSummary, sequentialDecide, sequentialPairedGate, skillOptEntry, skillOptProposer, traceAnalystProposer, userStoryScoreboard };
|
|
1478
|
+
export { type AcceptedEdit, type AceProposerOptions, type AnalystArtifact, type AnalystScenario, type ApplySkillPatchResult, type BuildAnalystSurfaceDispatchOptions, type CampaignBreakdown, CampaignResult, CampaignStorage, CodeSurface, type CompareProposersOptions, type DimensionRegression, DispatchContext, type FailureModeRecallJudgeOptions, type FapoAttributionSignals, type FapoEntryConfig, type FapoFailureCluster, type FapoOptimizationLevel, type FapoProposerOptions, type FapoReviewInput, type FapoReviewIssue, type FapoReviewResult, type FapoScopeContract, FsLabeledScenarioStore, type FsLabeledScenarioStoreOptions, Gate, GenerationRecord, type GitWorktreeAdapterOptions, type HaloProposerOptions, type HeldoutSignificance, type HeldoutSignificanceOptions, type JsonPrimitive, type JsonValue, JudgeConfig, JudgeScore, LabelTrust, LabeledScenarioRecord, LabeledScenarioSampleArgs, LabeledScenarioSource, LabeledScenarioStore, LabeledScenarioStoreError, LabeledScenarioWrite, type MemoryCurationProposerOptions, MutableSurface, type OptimizerEntryConfig, type PairedHoldout, type ParameterCandidate, type ParameterChange, type ParameterSweepProposerOptions, type PlaybackContext, type PlaybackDriver, type PlaybackStep, type PolicyEditProposerOptions, type ProfileDispatchFn, ProfileMatrixError, type ProfileSummary, ProposeContext, type ProposePatchesArgs, ProposedCandidate, type ProposerComparison, type ProposerEntry, type ProposerPairwise, type ProposerScore, type RejectedEdit, RunCampaignOptions, RunImprovementLoopOptions, type RunProfileMatrixOptions, type RunProfileMatrixResult, type RunSkillOptOptions, type RunSkillOptResult, Scenario, type ScenarioRollup, type ScoreboardRenderOptions, type ScoreboardRow, type ScoreboardSummary, type SequentialDecideFn, type SequentialDecideOptions, type SequentialDecision, type SequentialObservation, type SequentialPairedGate, type SequentialPairedGateOptions, type SkillOptEpochRecord, type SkillOptEvidence, type SkillOptProposer, type SkillOptProposerOptions, type SkillPatch, type SkillPatchOp, SkillPatchParseError, type SkillPatchRejection, SurfaceProposer, type TraceAnalystProposerOptions, type UserStory, type UserStoryVerdict, type Worktree, type WorktreeAdapter, WorktreeAdapterError, aceProposer, applySkillPatch, buildAnalystSurfaceDispatch, campaignBreakdown, campaignMeanComposite, compareProposers, detectScale, dimensionRegressions, extractFapoAttributionSignals, failureModeRecallJudge, fapoEscalationEntry, fapoProposer, gepaParetoEntry, gepaReflectionEntry, gitWorktreeAdapter, haloProposer, heldoutSignificance, makePlaybackDispatch, memoryCurationProposer, pairHoldout, parameterSweepProposer, parseSkillPatchResponse, patchEditCount, policyEditProposer, renderScoreboardMarkdown, resolveWorktreePath, runProfileMatrix, runSkillOpt, scoreUserStory, scoreboardSummary, sequentialDecide, sequentialPairedGate, skillOptEntry, skillOptProposer, traceAnalystProposer, userStoryScoreboard };
|
package/dist/campaign/index.js
CHANGED
|
@@ -10,7 +10,7 @@ import {
|
|
|
10
10
|
paretoPolicy,
|
|
11
11
|
paretoSignificanceGate,
|
|
12
12
|
runEval
|
|
13
|
-
} from "../chunk-
|
|
13
|
+
} from "../chunk-G7IB3GJ5.js";
|
|
14
14
|
import {
|
|
15
15
|
agentProfileHash,
|
|
16
16
|
agentProfileId,
|
|
@@ -18,7 +18,7 @@ import {
|
|
|
18
18
|
extractProducedState,
|
|
19
19
|
llmJudge,
|
|
20
20
|
verifyCompletion
|
|
21
|
-
} from "../chunk-
|
|
21
|
+
} from "../chunk-VWQ6PO5O.js";
|
|
22
22
|
import {
|
|
23
23
|
buildLoopProvenanceRecord,
|
|
24
24
|
campaignBreakdown,
|
|
@@ -40,7 +40,7 @@ import {
|
|
|
40
40
|
runOptimization,
|
|
41
41
|
surfaceContentHash,
|
|
42
42
|
surfaceHash
|
|
43
|
-
} from "../chunk-
|
|
43
|
+
} from "../chunk-2KTBHICD.js";
|
|
44
44
|
import {
|
|
45
45
|
assertRealBackend,
|
|
46
46
|
fsCampaignStorage,
|
|
@@ -52,12 +52,17 @@ import {
|
|
|
52
52
|
estimateCost,
|
|
53
53
|
isModelPriced
|
|
54
54
|
} from "../chunk-VI2UW6B6.js";
|
|
55
|
-
import
|
|
55
|
+
import {
|
|
56
|
+
admitPolicyEdit,
|
|
57
|
+
applyPolicyEditToSurface,
|
|
58
|
+
isPolicyEdit,
|
|
59
|
+
policyEditsFromFindings
|
|
60
|
+
} from "../chunk-IN3SHQML.js";
|
|
56
61
|
import {
|
|
57
62
|
AnalystRegistry,
|
|
58
63
|
DEFAULT_TRACE_ANALYST_KINDS,
|
|
59
64
|
createTraceAnalystKind
|
|
60
|
-
} from "../chunk-
|
|
65
|
+
} from "../chunk-FRI6RG3P.js";
|
|
61
66
|
import "../chunk-AIGWQEME.js";
|
|
62
67
|
import {
|
|
63
68
|
callLlm
|
|
@@ -70,7 +75,7 @@ import {
|
|
|
70
75
|
} from "../chunk-HRGUJTER.js";
|
|
71
76
|
import {
|
|
72
77
|
analyzeTraces
|
|
73
|
-
} from "../chunk-
|
|
78
|
+
} from "../chunk-2MLIEQSN.js";
|
|
74
79
|
import "../chunk-GGE4NNQT.js";
|
|
75
80
|
import {
|
|
76
81
|
OtlpFileTraceStore
|
|
@@ -78,7 +83,8 @@ import {
|
|
|
78
83
|
import "../chunk-PC4UYEBM.js";
|
|
79
84
|
import {
|
|
80
85
|
validateRunRecord
|
|
81
|
-
} from "../chunk-
|
|
86
|
+
} from "../chunk-G6S73VA7.js";
|
|
87
|
+
import "../chunk-ABOIVNXL.js";
|
|
82
88
|
import {
|
|
83
89
|
canonicalize
|
|
84
90
|
} from "../chunk-VSMTAMNK.js";
|
|
@@ -2343,6 +2349,68 @@ ${block}
|
|
|
2343
2349
|
};
|
|
2344
2350
|
}
|
|
2345
2351
|
|
|
2352
|
+
// src/campaign/proposers/policy-edit.ts
|
|
2353
|
+
function policyEditProposer(opts = {}) {
|
|
2354
|
+
return {
|
|
2355
|
+
kind: "policy-edit",
|
|
2356
|
+
async propose(ctx) {
|
|
2357
|
+
const edits = materializePolicyEdits(opts.edits ?? ctx.findings, opts.findingOptions);
|
|
2358
|
+
const limit = Math.max(
|
|
2359
|
+
0,
|
|
2360
|
+
Math.min(opts.maxCandidates ?? ctx.populationSize, ctx.populationSize)
|
|
2361
|
+
);
|
|
2362
|
+
const out = [];
|
|
2363
|
+
if (limit === 0) return out;
|
|
2364
|
+
for (const edit of edits) {
|
|
2365
|
+
const admission = admitPolicyEdit(edit, opts.admission);
|
|
2366
|
+
opts.onAdmission?.(admission);
|
|
2367
|
+
if (admission.decision !== "admit") continue;
|
|
2368
|
+
const surface = coerceCandidateSurface(applyPolicyEditToSurface(ctx.currentSurface, edit));
|
|
2369
|
+
if (sameSurface(ctx.currentSurface, surface)) continue;
|
|
2370
|
+
out.push({
|
|
2371
|
+
surface,
|
|
2372
|
+
label: `policy-edit:${edit.axis}`,
|
|
2373
|
+
rationale: `${edit.editId} expected ${edit.expectedGain.direction} ${edit.expectedGain.metric} by ${edit.expectedGain.amount}; source findings [${edit.source.findingIds.join(", ")}]`
|
|
2374
|
+
});
|
|
2375
|
+
if (out.length >= limit) break;
|
|
2376
|
+
}
|
|
2377
|
+
return out;
|
|
2378
|
+
}
|
|
2379
|
+
};
|
|
2380
|
+
}
|
|
2381
|
+
function materializePolicyEdits(inputs, findingOptions) {
|
|
2382
|
+
const edits = [];
|
|
2383
|
+
const findings = [];
|
|
2384
|
+
for (const input of inputs) {
|
|
2385
|
+
if (isPolicyEdit(input)) {
|
|
2386
|
+
edits.push(input);
|
|
2387
|
+
} else if (isAnalystFindingLike(input)) {
|
|
2388
|
+
findings.push(input);
|
|
2389
|
+
}
|
|
2390
|
+
}
|
|
2391
|
+
if (findings.length > 0) edits.push(...policyEditsFromFindings(findings, findingOptions));
|
|
2392
|
+
return edits;
|
|
2393
|
+
}
|
|
2394
|
+
function isAnalystFindingLike(input) {
|
|
2395
|
+
if (!input || typeof input !== "object") return false;
|
|
2396
|
+
const obj = input;
|
|
2397
|
+
return typeof obj.finding_id === "string" && typeof obj.analyst_id === "string" && typeof obj.claim === "string" && Array.isArray(obj.evidence_refs);
|
|
2398
|
+
}
|
|
2399
|
+
function coerceCandidateSurface(surface) {
|
|
2400
|
+
if (typeof surface === "string") return surface;
|
|
2401
|
+
if (surface && typeof surface === "object") {
|
|
2402
|
+
const obj = surface;
|
|
2403
|
+
if (obj.kind === "code" && typeof obj.worktreeRef === "string") {
|
|
2404
|
+
return surface;
|
|
2405
|
+
}
|
|
2406
|
+
return JSON.stringify(surface, null, 2);
|
|
2407
|
+
}
|
|
2408
|
+
throw new Error("policyEditProposer: policy edit produced an unsupported surface");
|
|
2409
|
+
}
|
|
2410
|
+
function sameSurface(a, b) {
|
|
2411
|
+
return JSON.stringify(a) === JSON.stringify(b);
|
|
2412
|
+
}
|
|
2413
|
+
|
|
2346
2414
|
// src/campaign/proposers/trace-analyst.ts
|
|
2347
2415
|
import { ai } from "@ax-llm/ax";
|
|
2348
2416
|
function renderFindings(findings) {
|
|
@@ -2511,6 +2579,7 @@ export {
|
|
|
2511
2579
|
paretoSignificanceGate,
|
|
2512
2580
|
parseSkillPatchResponse,
|
|
2513
2581
|
patchEditCount,
|
|
2582
|
+
policyEditProposer,
|
|
2514
2583
|
provenanceRecordPath,
|
|
2515
2584
|
provenanceSpansPath,
|
|
2516
2585
|
renderScoreboardMarkdown,
|