@tangle-network/agent-runtime 0.107.5 → 0.108.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +17 -4
- package/dist/{activation-Ck5ksFWg.js → activation-DRpnplEm.js} +3 -3
- package/dist/{activation-Ck5ksFWg.js.map → activation-DRpnplEm.js.map} +1 -1
- package/dist/agent.d.ts +1 -1
- package/dist/agent.js +2 -2
- package/dist/candidate-execution/index.js +4 -4
- package/dist/{candidate-execution-BUpR0mSD.js → candidate-execution-nYMa-bc-.js} +4 -4
- package/dist/{candidate-execution-BUpR0mSD.js.map → candidate-execution-nYMa-bc-.js.map} +1 -1
- package/dist/durable.d.ts +134 -0
- package/dist/durable.js +124 -0
- package/dist/durable.js.map +1 -0
- package/dist/{environment-provider-IUGU3epE.d.ts → environment-provider-D0NXc4Qz.d.ts} +73 -4
- package/dist/environment-provider.d.ts +1 -1
- package/dist/{improvement-cycle-Ulvvbg5p.js → improvement-cycle-CRnDDdX0.js} +49 -155
- package/dist/improvement-cycle-CRnDDdX0.js.map +1 -0
- package/dist/{index-B3wAV1SJ.d.ts → index-BZbJWqoZ.d.ts} +165 -25
- package/dist/{index-BbB5mmK0.d.ts → index-DQp3BPeC.d.ts} +11 -159
- package/dist/{index-D2g5qCwg.d.ts → index-cdkwHruJ.d.ts} +3 -3
- package/dist/index.d.ts +6 -6
- package/dist/index.js +14 -121
- package/dist/index.js.map +1 -1
- package/dist/intelligence.d.ts +15 -7
- package/dist/intelligence.js +4 -4
- package/dist/{knowledge-Dizh_AtL.js → knowledge-rdrpPIXs.js} +5 -5
- package/dist/{knowledge-Dizh_AtL.js.map → knowledge-rdrpPIXs.js.map} +1 -1
- package/dist/knowledge.d.ts +1 -1
- package/dist/knowledge.js +1 -1
- package/dist/{loop-runner-bin-Dw9_OZfu.d.ts → loop-runner-bin-9-JlGqN7.d.ts} +3 -3
- package/dist/{loop-runner-bin-BHELvjdO.js → loop-runner-bin-B1V-XS5A.js} +3 -3
- package/dist/{loop-runner-bin-BHELvjdO.js.map → loop-runner-bin-B1V-XS5A.js.map} +1 -1
- package/dist/loop-runner-bin.d.ts +1 -1
- package/dist/loop-runner-bin.js +1 -1
- package/dist/loops.d.ts +3 -3
- package/dist/loops.js +5 -5
- package/dist/mcp/bin.js +1 -1
- package/dist/mcp/index.d.ts +1 -1
- package/dist/mcp/index.js +4 -4
- package/dist/{openai-tools-CKLy1C7M.js → openai-tools-BjQq2TTL.js} +2 -2
- package/dist/{openai-tools-CKLy1C7M.js.map → openai-tools-BjQq2TTL.js.map} +1 -1
- package/dist/{prepare-_WTTffkz.js → prepare-Pr19X93k.js} +141 -6
- package/dist/prepare-Pr19X93k.js.map +1 -0
- package/dist/primeintellect/index.d.ts +1 -1
- package/dist/{protected-model-port-ZUwVuftA.js → protected-model-port-uthtDNqu.js} +2 -2
- package/dist/{protected-model-port-ZUwVuftA.js.map → protected-model-port-uthtDNqu.js.map} +1 -1
- package/dist/{redact-C5xOm8cu.d.ts → redact-BkasKlyd.d.ts} +17 -27
- package/dist/{runtime-M_GIzIwe.js → runtime-CDekUROm.js} +4 -4
- package/dist/{runtime-M_GIzIwe.js.map → runtime-CDekUROm.js.map} +1 -1
- package/dist/{structural-rollout-IXUEplky.js → structural-rollout-BC81Otmc.js} +10 -9
- package/dist/structural-rollout-BC81Otmc.js.map +1 -0
- package/dist/{supervise-BiRutHS9.js → supervise-JfKPwIlO.js} +270 -69
- package/dist/supervise-JfKPwIlO.js.map +1 -0
- package/dist/{supervisor-DTKhF-RV.js → supervisor-B2LzaWRb.js} +221 -16
- package/dist/supervisor-B2LzaWRb.js.map +1 -0
- package/dist/testing.js +288 -239
- package/dist/testing.js.map +1 -1
- package/dist/{workspace-archive-CIG1VNCj.js → workspace-archive-C9w_rxt6.js} +2 -2
- package/dist/{workspace-archive-CIG1VNCj.js.map → workspace-archive-C9w_rxt6.js.map} +1 -1
- package/package.json +11 -6
- package/dist/improvement-cycle-Ulvvbg5p.js.map +0 -1
- package/dist/prepare-_WTTffkz.js.map +0 -1
- package/dist/structural-rollout-IXUEplky.js.map +0 -1
- package/dist/supervise-BiRutHS9.js.map +0 -1
- package/dist/supervisor-DTKhF-RV.js.map +0 -1
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
import { i as ConfigError } from "./errors-DEAvWQPy.js";
|
|
2
|
-
import { $ as parseExactAgentProfile, G as agentCandidateProfileAsAgentProfile, J as candidateMaterializerHarness, K as applyExactAgentProfileDiff, Q as parseAgentCandidateProfileActivation, Y as createAgentCandidateProfileActivation, Z as omitUndefinedObjectFields, ct as omitTopLevelDigest, d as verifiedResourceTextByDigest, f as verifyAgentCandidateBundle, it as canonicalCandidateDocument, n as executePreparedAgentCandidate, nt as canonicalCandidateBytes, q as assertCandidateProfileBinding, rt as canonicalCandidateDigest$1, st as immutableCandidateValue, t as prepareAgentCandidateExecution, ut as verifyCanonicalCandidateDocument } from "./prepare-
|
|
3
|
-
import {
|
|
4
|
-
import { E as optimizerMethod, a as defaultStructuralRolloutPolicy } from "./structural-rollout-
|
|
2
|
+
import { $ as parseExactAgentProfile, G as agentCandidateProfileAsAgentProfile, J as candidateMaterializerHarness, K as applyExactAgentProfileDiff, Q as parseAgentCandidateProfileActivation, Y as createAgentCandidateProfileActivation, Z as omitUndefinedObjectFields, ct as omitTopLevelDigest, d as verifiedResourceTextByDigest, f as verifyAgentCandidateBundle, it as canonicalCandidateDocument, n as executePreparedAgentCandidate, nt as canonicalCandidateBytes, q as assertCandidateProfileBinding, rt as canonicalCandidateDigest$1, st as immutableCandidateValue, t as prepareAgentCandidateExecution, ut as verifyCanonicalCandidateDocument } from "./prepare-Pr19X93k.js";
|
|
3
|
+
import { A as runSettledCommand, I as harnessInvocation, R as runLocalHarness } from "./supervisor-B2LzaWRb.js";
|
|
4
|
+
import { E as optimizerMethod, a as defaultStructuralRolloutPolicy } from "./structural-rollout-BC81Otmc.js";
|
|
5
5
|
import { t as runAnalystLoop } from "./analyst-loop-C8cGThTW.js";
|
|
6
|
-
import { CostLedger, canonicalJson,
|
|
6
|
+
import { CostLedger, canonicalJson, makeProposalFinding } from "@tangle-network/agent-eval";
|
|
7
7
|
import { campaignScenarioIdentity, campaignSplitDigestFromIdentities, compareOptimizationMethods, gitWorktreeAdapter, verifyCodeSurface } from "@tangle-network/agent-eval/campaign";
|
|
8
8
|
import { AGENT_IMPROVEMENT_SOURCE_METADATA_KEY, agentCandidateMaterializationReceiptSchema, agentCandidateRunReceiptSchema, agentImprovementActivationSchema, agentImprovementProposalSchema, agentImprovementReviewSchema, agentImprovementSourceMetadata, agentImprovementSourceSchema, agentProfileDiffSchema, agentProfileImprovementArmSchema, agentProfileImprovementExecutionRefSchema, agentProfileImprovementMeasuredComparisonSchema, agentProfileModelHintsSchema, agentProfileSchema, candidateExecutionEvidenceSchema, changedProfileImprovementSurfaces, defineAgentProfileDiff, numbersApproximatelyEqual, sha256DigestSchema } from "@tangle-network/agent-interface";
|
|
9
9
|
import { createHash, randomUUID } from "node:crypto";
|
|
@@ -11,15 +11,15 @@ import { applyWorkspacePlan, materializeCandidateProfile, materializeProfile } f
|
|
|
11
11
|
import { existsSync, readFileSync, readdirSync, rmSync } from "node:fs";
|
|
12
12
|
import { basename, join, resolve, sep } from "node:path";
|
|
13
13
|
import { spawnSync } from "node:child_process";
|
|
14
|
+
import { assertProposalFindings } from "@tangle-network/agent-eval/analyst";
|
|
14
15
|
import { measuredComparisonFromAgentProfileImprovementExperiment, measuredComparisonFromCandidateExperiment, runAgentProfileImprovementExperiment, runCandidateExperiment, sealAgentProfileImprovementExperiment, sealAgentProfileImprovementSuite, sealAgentProfileImprovementTask, sealCandidateExperiment, selfImprove, verifyAgentProfileImprovementExperimentComparison, verifyCandidateExperiment, verifyCandidateExperimentComparison } from "@tangle-network/agent-eval/contract";
|
|
15
|
-
import { assertNoJudgeVerdict } from "@tangle-network/agent-eval/analyst";
|
|
16
16
|
//#region src/improvement/agentic-generator.ts
|
|
17
17
|
/**
|
|
18
18
|
*
|
|
19
19
|
* `agenticGenerator` — the full-agentic `CandidateGenerator`. It runs a real
|
|
20
20
|
* coding harness (claude / codex / opencode) inside the candidate worktree the
|
|
21
21
|
* driver already created, letting the agent read the codebase + the research
|
|
22
|
-
*
|
|
22
|
+
* proposal findings and make the change in place. The driver then commits the worktree
|
|
23
23
|
* into a `CodeSurface`.
|
|
24
24
|
*
|
|
25
25
|
* Mechanism: identical to the proven Phase-2.8 in-process executor — spawn the
|
|
@@ -67,17 +67,14 @@ function agenticGenerator(opts = {}) {
|
|
|
67
67
|
return {
|
|
68
68
|
kind: `agentic:${harness}`,
|
|
69
69
|
proposesWithoutFindings: true,
|
|
70
|
-
async generate({ worktreePath,
|
|
70
|
+
async generate({ worktreePath, findings, maxShots, signal, generation, candidateIndex, costLedger, costPhase }) {
|
|
71
71
|
signal.throwIfAborted();
|
|
72
72
|
let reproducibleCostLedger;
|
|
73
73
|
if (opts.codexReproducible) {
|
|
74
74
|
if (!costLedger) throw new Error("agenticGenerator: reproducible Codex requires the run-wide CostLedger supplied by agent-eval");
|
|
75
75
|
reproducibleCostLedger = costLedger;
|
|
76
76
|
}
|
|
77
|
-
const basePrompt = appendProfileResourcePaths(buildPrompt({
|
|
78
|
-
report,
|
|
79
|
-
findings
|
|
80
|
-
}), profileResourcePlan);
|
|
77
|
+
const basePrompt = appendProfileResourcePaths(buildPrompt({ findings }), profileResourcePlan);
|
|
81
78
|
const needsRawTraceEvidence = requiresRawTraceEvidence(findings);
|
|
82
79
|
const shots = Math.max(1, maxShots);
|
|
83
80
|
let attemptNote = "";
|
|
@@ -431,7 +428,7 @@ function costReceiptFromHarness(result, model) {
|
|
|
431
428
|
function sha256(value) {
|
|
432
429
|
return `sha256:${createHash("sha256").update(value).digest("hex")}`;
|
|
433
430
|
}
|
|
434
|
-
/** Turn
|
|
431
|
+
/** Turn proposal findings into a concrete coder task —
|
|
435
432
|
* the senior scientific-method framing shared with the tool/MCP build prompts. */
|
|
436
433
|
function defaultBuildPrompt(args) {
|
|
437
434
|
const lines = [
|
|
@@ -589,98 +586,6 @@ function worktreeChangedPaths(worktreePath) {
|
|
|
589
586
|
return result.stdout.split("\n").map((line) => line.trim()).filter((line) => line.length > 0).map((line) => line.slice(3).trim());
|
|
590
587
|
}
|
|
591
588
|
//#endregion
|
|
592
|
-
//#region src/improvement/findings.ts
|
|
593
|
-
/**
|
|
594
|
-
* Typed-findings accessor — the one place `unknown[]` findings become
|
|
595
|
-
* `AnalystFinding[]`.
|
|
596
|
-
*
|
|
597
|
-
* agent-eval's `ProposeContext.findings` is `TFindings[] = unknown[]` on the
|
|
598
|
-
* wire: the loop threads whatever the previous `analyzeGeneration` producer (or
|
|
599
|
-
* the caller's static seed) returned. Consumers that need the typed envelope
|
|
600
|
-
* (`claim`/`severity`/`recommended_action`) were down-casting with a bare
|
|
601
|
-
* `as AnalystFinding[]` — a lie at runtime whenever the seed was a raw string
|
|
602
|
-
* or an ad-hoc digest, which then rendered `undefined` into build prompts.
|
|
603
|
-
*
|
|
604
|
-
* `toAnalystFindings` replaces that cast: real findings pass through
|
|
605
|
-
* unchanged (structural guard, fail-closed), and non-conforming values are
|
|
606
|
-
* LIFTED into a real `AnalystFinding` envelope via `makeFinding` — the most
|
|
607
|
-
* actionable text becomes the claim, the original value rides in `metadata.raw`
|
|
608
|
-
* — so everything downstream of the accessor handles exactly one shape.
|
|
609
|
-
*/
|
|
610
|
-
const SEVERITIES = /* @__PURE__ */ new Set([
|
|
611
|
-
"critical",
|
|
612
|
-
"high",
|
|
613
|
-
"medium",
|
|
614
|
-
"low",
|
|
615
|
-
"info"
|
|
616
|
-
]);
|
|
617
|
-
/** Analyst id stamped on findings lifted from untyped seed values. */
|
|
618
|
-
const LIFTED_FINDING_ANALYST_ID = "lifted-seed";
|
|
619
|
-
/** Structural guard for the schema-versioned `AnalystFinding` envelope.
|
|
620
|
-
* Strict on the identity fields `makeFinding` always populates — a partial
|
|
621
|
-
* look-alike is lifted (re-enveloped), not trusted. */
|
|
622
|
-
function isAnalystFinding(value) {
|
|
623
|
-
if (!value || typeof value !== "object") return false;
|
|
624
|
-
const o = value;
|
|
625
|
-
return o.schema_version === "1.0.0" && typeof o.finding_id === "string" && typeof o.analyst_id === "string" && typeof o.severity === "string" && SEVERITIES.has(o.severity) && typeof o.area === "string" && typeof o.claim === "string" && typeof o.confidence === "number" && Array.isArray(o.evidence_refs);
|
|
626
|
-
}
|
|
627
|
-
/** The most actionable text of an untyped finding-ish value — mirrors the
|
|
628
|
-
* extraction order agent-eval's curator proposers use (`recommended_action` >
|
|
629
|
-
* `claim` > `lesson` > `notes` > `text` > `message`), so the two ends of the
|
|
630
|
-
* wire read the same field first. */
|
|
631
|
-
function liftedClaim(value) {
|
|
632
|
-
if (typeof value === "string") return value.trim() || null;
|
|
633
|
-
if (value && typeof value === "object") {
|
|
634
|
-
const o = value;
|
|
635
|
-
for (const key of [
|
|
636
|
-
"recommended_action",
|
|
637
|
-
"claim",
|
|
638
|
-
"lesson",
|
|
639
|
-
"notes",
|
|
640
|
-
"text",
|
|
641
|
-
"message"
|
|
642
|
-
]) {
|
|
643
|
-
const v = o[key];
|
|
644
|
-
if (typeof v === "string" && v.trim()) return v.trim();
|
|
645
|
-
}
|
|
646
|
-
try {
|
|
647
|
-
const json = JSON.stringify(value);
|
|
648
|
-
if (json && json !== "{}" && json !== "[]") return json.length > 400 ? `${json.slice(0, 399)}…` : json;
|
|
649
|
-
} catch {}
|
|
650
|
-
}
|
|
651
|
-
return null;
|
|
652
|
-
}
|
|
653
|
-
/**
|
|
654
|
-
* Normalize a mixed `unknown[]` findings array to `AnalystFinding[]`:
|
|
655
|
-
* conforming findings pass through by reference; strings and finding-ish
|
|
656
|
-
* objects are lifted into envelopes (claim = most actionable text, original
|
|
657
|
-
* value under `metadata.raw`); values with no extractable text are dropped.
|
|
658
|
-
* Never throws — a malformed seed must not kill a proposal round.
|
|
659
|
-
*/
|
|
660
|
-
function toAnalystFindings(findings, opts = {}) {
|
|
661
|
-
const analystId = opts.analystId ?? "lifted-seed";
|
|
662
|
-
const area = opts.area ?? "seed";
|
|
663
|
-
const out = [];
|
|
664
|
-
for (const f of findings) {
|
|
665
|
-
if (isAnalystFinding(f)) {
|
|
666
|
-
out.push(f);
|
|
667
|
-
continue;
|
|
668
|
-
}
|
|
669
|
-
const claim = liftedClaim(f);
|
|
670
|
-
if (!claim) continue;
|
|
671
|
-
out.push(makeFinding({
|
|
672
|
-
analyst_id: analystId,
|
|
673
|
-
severity: "info",
|
|
674
|
-
area,
|
|
675
|
-
confidence: .5,
|
|
676
|
-
claim,
|
|
677
|
-
evidence_refs: [],
|
|
678
|
-
...f && typeof f === "object" ? { metadata: { raw: f } } : {}
|
|
679
|
-
}));
|
|
680
|
-
}
|
|
681
|
-
return out;
|
|
682
|
-
}
|
|
683
|
-
//#endregion
|
|
684
589
|
//#region src/improvement/cleanup.ts
|
|
685
590
|
async function rethrowAfterCleanup(cause, cleanup, context) {
|
|
686
591
|
const cleanupErrors = [];
|
|
@@ -722,8 +627,8 @@ function improvementDriver(opts) {
|
|
|
722
627
|
return {
|
|
723
628
|
kind: `improvement:${opts.generator.kind}`,
|
|
724
629
|
async propose(ctx) {
|
|
725
|
-
const findings =
|
|
726
|
-
if (findings.length === 0 &&
|
|
630
|
+
const findings = ctx.findings;
|
|
631
|
+
if (findings.length === 0 && !opts.generator.proposesWithoutFindings) return [];
|
|
727
632
|
const surfaces = [];
|
|
728
633
|
const incumbent = verifiedCodeIncumbent(ctx.currentSurface);
|
|
729
634
|
const proposalBaseRef = incumbent?.baseCommit ?? baseRef;
|
|
@@ -738,9 +643,7 @@ function improvementDriver(opts) {
|
|
|
738
643
|
if (incumbent) advanceToIncumbent(wt, incumbent);
|
|
739
644
|
const { applied, summary, label, rationale } = await opts.generator.generate({
|
|
740
645
|
worktreePath: wt.path,
|
|
741
|
-
report: ctx.report,
|
|
742
646
|
findings,
|
|
743
|
-
dataset: ctx.dataset,
|
|
744
647
|
maxShots: ctx.maxImprovementShots ?? 1,
|
|
745
648
|
signal: ctx.signal,
|
|
746
649
|
generation: ctx.generation,
|
|
@@ -819,28 +722,6 @@ function advanceToIncumbent(worktree, incumbent) {
|
|
|
819
722
|
});
|
|
820
723
|
if (head.error || head.status !== 0 || head.stdout.trim() !== incumbent.candidateCommit) throw new Error("improvementDriver: candidate worktree did not reach the incumbent commit");
|
|
821
724
|
}
|
|
822
|
-
/** Phase-2 report carries `findings` when present; else fall back to the
|
|
823
|
-
* loop's `ctx.findings`. The report is opaque to the substrate, so probe it
|
|
824
|
-
* structurally. Both paths run through `toAnalystFindings` — the wire is
|
|
825
|
-
* `unknown[]` at runtime regardless of the generic (static seeds, legacy
|
|
826
|
-
* digests), and a bare cast here fed `undefined` claims into build prompts. */
|
|
827
|
-
function resolveFindings(ctx) {
|
|
828
|
-
const report = ctx.report;
|
|
829
|
-
if (report && typeof report === "object" && "findings" in report) {
|
|
830
|
-
const f = report.findings;
|
|
831
|
-
if (Array.isArray(f) && f.length > 0) {
|
|
832
|
-
const lifted = toAnalystFindings(f, {
|
|
833
|
-
analystId: "report-findings",
|
|
834
|
-
area: "report"
|
|
835
|
-
});
|
|
836
|
-
if (lifted.length > 0) return lifted;
|
|
837
|
-
}
|
|
838
|
-
}
|
|
839
|
-
return toAnalystFindings(ctx.findings ?? [], {
|
|
840
|
-
analystId: "loop-context",
|
|
841
|
-
area: "seed"
|
|
842
|
-
});
|
|
843
|
-
}
|
|
844
725
|
//#endregion
|
|
845
726
|
//#region src/improvement/rollout-policy.ts
|
|
846
727
|
/** The profile extensions namespace the policy persists under. */
|
|
@@ -1179,7 +1060,7 @@ function assertFailClosedResources(profile, surface) {
|
|
|
1179
1060
|
* instructs the agent to `grep`/`cat`/`ls` them to diagnose the failures itself
|
|
1180
1061
|
* (up to the harness's full context, ~millions of tokens, vs a ~1500-char digest).
|
|
1181
1062
|
*
|
|
1182
|
-
* It emits `
|
|
1063
|
+
* It emits `ProposalFinding[]` so it drops into the same `opts.analyzeGeneration`
|
|
1183
1064
|
* slot the default distiller uses, and renders through the same
|
|
1184
1065
|
* `agenticGenerator` prompt path (`claim` + `recommended_action`). The findings
|
|
1185
1066
|
* carry ABSOLUTE paths — the coding harness runs with `cwd` = a candidate
|
|
@@ -1232,8 +1113,9 @@ function rawTraceDistiller(options = {}) {
|
|
|
1232
1113
|
const totalFailingCells = ranked.reduce((n, c) => n + c.cells.length, 0);
|
|
1233
1114
|
if (totalFailingCells === 0) {
|
|
1234
1115
|
if (options.fallbackFindings && options.fallbackFindings.length > 0) return options.fallbackFindings;
|
|
1235
|
-
return [
|
|
1116
|
+
return [makeProposalFinding({
|
|
1236
1117
|
analyst_id: ANALYST_ID,
|
|
1118
|
+
proposal_origin: "search",
|
|
1237
1119
|
severity: "info",
|
|
1238
1120
|
area: "raw-trace-context",
|
|
1239
1121
|
confidence: 1,
|
|
@@ -1251,8 +1133,9 @@ function rawTraceDistiller(options = {}) {
|
|
|
1251
1133
|
})];
|
|
1252
1134
|
}
|
|
1253
1135
|
const findings = [];
|
|
1254
|
-
findings.push(
|
|
1136
|
+
findings.push(makeProposalFinding({
|
|
1255
1137
|
analyst_id: ANALYST_ID,
|
|
1138
|
+
proposal_origin: "search",
|
|
1256
1139
|
severity: "high",
|
|
1257
1140
|
area: "raw-trace-context",
|
|
1258
1141
|
confidence: 1,
|
|
@@ -1278,8 +1161,9 @@ function rawTraceDistiller(options = {}) {
|
|
|
1278
1161
|
const more = c.truncatedFiles ? `\n - …(ls ${c.cellDir} for the rest)` : "";
|
|
1279
1162
|
return c.files.length > 0 ? `${header}\n${files}${more}` : header;
|
|
1280
1163
|
}).join("\n");
|
|
1281
|
-
findings.push(
|
|
1164
|
+
findings.push(makeProposalFinding({
|
|
1282
1165
|
analyst_id: ANALYST_ID,
|
|
1166
|
+
proposal_origin: "search",
|
|
1283
1167
|
severity: cand.composite !== null && cand.composite < .5 ? "critical" : "high",
|
|
1284
1168
|
area: "raw-trace-context",
|
|
1285
1169
|
confidence: 1,
|
|
@@ -1436,8 +1320,9 @@ function generationFailureDistiller(staticFindings) {
|
|
|
1436
1320
|
}
|
|
1437
1321
|
if (failures.length === 0) return staticFindings;
|
|
1438
1322
|
failures.sort((left, right) => left.composite - right.composite);
|
|
1439
|
-
return failures.slice(0, CAP).map((failure) =>
|
|
1323
|
+
return failures.slice(0, CAP).map((failure) => makeProposalFinding({
|
|
1440
1324
|
analyst_id: "generation-failure-distiller",
|
|
1325
|
+
proposal_origin: "search",
|
|
1441
1326
|
severity: failure.error !== void 0 || failure.composite < .5 ? "high" : "medium",
|
|
1442
1327
|
area: "generation-failure",
|
|
1443
1328
|
confidence: 1,
|
|
@@ -1529,7 +1414,7 @@ function idempotentDispose(dispose) {
|
|
|
1529
1414
|
}
|
|
1530
1415
|
async function runCodeImprovement(opts) {
|
|
1531
1416
|
const { gate = "holdout", findings: inputFindings = [], rawTraceContext: _rawTraceContext, code, promotionGate, analyzeGeneration, surface: _surface, ...sharedOptions } = opts;
|
|
1532
|
-
const findings = [...inputFindings];
|
|
1417
|
+
const findings = immutableCandidateValue([...assertProposalFindings(inputFindings, "improve() code findings")]);
|
|
1533
1418
|
const preparedCode = await prepareCodeRun(code);
|
|
1534
1419
|
const budget = gate === "none" ? {
|
|
1535
1420
|
...sharedOptions.budget,
|
|
@@ -1749,7 +1634,7 @@ async function runMethodImprovement(profile, opts) {
|
|
|
1749
1634
|
if (!Number.isFinite(minimumLift) || minimumLift < 0) throw new ConfigError("improve(): minimumLift must be a finite number greater than or equal to 0");
|
|
1750
1635
|
if (profileComponents && surface !== "agent-profile") throw new ConfigError("improve(): profileComponents is valid only with surface 'agent-profile'");
|
|
1751
1636
|
const executionRef = validateExecutionRef(inputExecutionRef);
|
|
1752
|
-
const findings = [...inputFindings];
|
|
1637
|
+
const findings = immutableCandidateValue([...assertProposalFindings(inputFindings, "improve() method findings")]);
|
|
1753
1638
|
const preparedSurface = prepareProfileSurface(profile, surface, skills, profileComponents);
|
|
1754
1639
|
const baselineSurface = preparedSurface.surface;
|
|
1755
1640
|
const baselineValue = immutableCandidateValue(preparedSurface.value);
|
|
@@ -2478,9 +2363,7 @@ function createAgentImprovementMeasuredComparison(options) {
|
|
|
2478
2363
|
return verifyCandidateExperimentComparison(measuredComparisonFromCandidateExperiment(options));
|
|
2479
2364
|
}
|
|
2480
2365
|
async function analyzeAgentImprovement(runId, options, costLedger, signal) {
|
|
2481
|
-
|
|
2482
|
-
if (rawOptions.knowledgeProposalSource !== void 0 || rawOptions.improvementProposalSource !== void 0) throw new Error("measured agent improvement analysis must not run proposal sources");
|
|
2483
|
-
if (rawOptions.onEvent !== void 0 || rawOptions.log !== void 0) throw new Error("measured agent improvement analysis must not run callbacks");
|
|
2366
|
+
assertMeasuredAnalysisOptions(options);
|
|
2484
2367
|
const analysis = await runAnalystLoop({
|
|
2485
2368
|
...options,
|
|
2486
2369
|
runId,
|
|
@@ -2491,11 +2374,22 @@ async function analyzeAgentImprovement(runId, options, costLedger, signal) {
|
|
|
2491
2374
|
...signal ? { signal } : {}
|
|
2492
2375
|
});
|
|
2493
2376
|
if (costLedger) assertAnalysisCostRecorded(analysis, costLedger);
|
|
2377
|
+
const judgeDerived = analysis.analystResult.findings.filter((finding) => finding.derived_from_judge === true);
|
|
2378
|
+
if (judgeDerived.length > 0) throw new Error(`agent improvement analysis must not produce judge-derived findings: [${judgeDerived.map((finding) => finding.finding_id).join(", ")}]`);
|
|
2494
2379
|
return {
|
|
2495
2380
|
analysis,
|
|
2496
|
-
findings: [...
|
|
2381
|
+
findings: [...assertProposalFindings(analysis.analystResult.findings.map((finding) => ({
|
|
2382
|
+
...finding,
|
|
2383
|
+
proposal_origin: "production"
|
|
2384
|
+
})), "agent improvement findings")]
|
|
2497
2385
|
};
|
|
2498
2386
|
}
|
|
2387
|
+
function assertMeasuredAnalysisOptions(options) {
|
|
2388
|
+
const rawOptions = options;
|
|
2389
|
+
if (rawOptions.knowledgeProposalSource !== void 0 || rawOptions.improvementProposalSource !== void 0) throw new Error("measured agent improvement analysis must not run proposal sources");
|
|
2390
|
+
if (rawOptions.onEvent !== void 0 || rawOptions.log !== void 0) throw new Error("measured agent improvement analysis must not run callbacks");
|
|
2391
|
+
if (rawOptions.inputs.judgeInput !== void 0) throw new Error("measured agent improvement analysis must not receive judge input");
|
|
2392
|
+
}
|
|
2499
2393
|
function completeAnalysisAccounting(analysis) {
|
|
2500
2394
|
const analysisCost = analysis.analystResult.total_cost_provenance;
|
|
2501
2395
|
if (!analysisCost || analysisCost.kind === "uncaptured") throw new Error("agent improvement analysis cost is uncaptured");
|
|
@@ -2613,6 +2507,8 @@ function profileImprovementMetadata(metadata, source, optimizationReceipt) {
|
|
|
2613
2507
|
async function proposeAgentProfileImprovement(options) {
|
|
2614
2508
|
const source = agentImprovementSourceSchema.parse(options.source);
|
|
2615
2509
|
if (!isAgentProfileMeasuredSurface(options.improvement.surface ?? "prompt")) throw new Error("measured profile improvement supports prompt or skills; use the sealed-candidate path for this surface");
|
|
2510
|
+
assertMeasuredAnalysisOptions(options.analysis);
|
|
2511
|
+
const inputFindings = assertProposalFindings(options.improvement.findings ?? [], "profile improvement input findings");
|
|
2616
2512
|
const costLedger = createProfileImprovementCostLedger(options.budgetUsd);
|
|
2617
2513
|
const preparationStartedAt = performance.now();
|
|
2618
2514
|
const profile = parseExactAgentProfile(options.profile, "profile improvement source");
|
|
@@ -2621,13 +2517,14 @@ async function proposeAgentProfileImprovement(options) {
|
|
|
2621
2517
|
const policy = profilePolicyWithBudget(options.benchmark.policy, options.budgetUsd);
|
|
2622
2518
|
if (options.improvement.costCeiling !== void 0 && !numbersApproximatelyEqual(options.improvement.costCeiling, options.budgetUsd)) throw new Error("profile improvement costCeiling must equal the run budgetUsd");
|
|
2623
2519
|
const { analysis, findings } = await analyzeAgentImprovement(options.runId, options.analysis, costLedger, options.signal);
|
|
2520
|
+
const proposalFindings = immutableCandidateValue([...inputFindings, ...findings]);
|
|
2624
2521
|
const improvement = await improve(profile, {
|
|
2625
2522
|
...options.improvement,
|
|
2626
2523
|
executionRef: options.executor.executionRef.digest,
|
|
2627
2524
|
agent: options.executor.optimize,
|
|
2628
2525
|
costLedger,
|
|
2629
2526
|
costCeiling: options.budgetUsd,
|
|
2630
|
-
findings:
|
|
2527
|
+
findings: proposalFindings
|
|
2631
2528
|
});
|
|
2632
2529
|
try {
|
|
2633
2530
|
if (improvement.decision !== "ship") throw new Error("agent profile improvement search did not produce a promotable candidate");
|
|
@@ -2697,7 +2594,7 @@ async function proposeAgentProfileImprovement(options) {
|
|
|
2697
2594
|
}));
|
|
2698
2595
|
const proposal = createAgentImprovementProposal({
|
|
2699
2596
|
runId: options.runId,
|
|
2700
|
-
findings,
|
|
2597
|
+
findings: proposalFindings,
|
|
2701
2598
|
evaluation,
|
|
2702
2599
|
...options.now ? { now: options.now } : {}
|
|
2703
2600
|
});
|
|
@@ -2715,11 +2612,14 @@ async function proposeAgentProfileImprovement(options) {
|
|
|
2715
2612
|
/** Analyze, search, then remeasure the resulting exact candidate before proposing it. */
|
|
2716
2613
|
async function proposeAgentImprovement(options) {
|
|
2717
2614
|
assertNoCallerOptimizationReceipt(options.metadata);
|
|
2615
|
+
assertMeasuredAnalysisOptions(options.analysis);
|
|
2616
|
+
const inputFindings = assertProposalFindings(options.improvement.findings ?? [], "agent improvement input findings");
|
|
2718
2617
|
const { analysis, findings } = await analyzeAgentImprovement(options.runId, options.analysis);
|
|
2719
2618
|
const analysisAccounting = completeAnalysisAccounting(analysis);
|
|
2619
|
+
const proposalFindings = immutableCandidateValue([...inputFindings, ...findings]);
|
|
2720
2620
|
const improvementInput = {
|
|
2721
2621
|
...options.improvement,
|
|
2722
|
-
findings:
|
|
2622
|
+
findings: proposalFindings
|
|
2723
2623
|
};
|
|
2724
2624
|
const improvement = improvementInput.surface === "code" ? await improve(improvementInput) : await improve(options.profile, improvementInput);
|
|
2725
2625
|
try {
|
|
@@ -2752,7 +2652,7 @@ async function proposeAgentImprovement(options) {
|
|
|
2752
2652
|
});
|
|
2753
2653
|
const proposal = createAgentImprovementProposal({
|
|
2754
2654
|
runId: options.runId,
|
|
2755
|
-
findings,
|
|
2655
|
+
findings: proposalFindings,
|
|
2756
2656
|
evaluation: measured.evaluation,
|
|
2757
2657
|
...options.now ? { now: options.now } : {}
|
|
2758
2658
|
});
|
|
@@ -2784,7 +2684,7 @@ function assertImprovementCandidateBinding(improvement, experiment) {
|
|
|
2784
2684
|
}
|
|
2785
2685
|
/** Create the reviewable record only from a complete, recomputable experiment result. */
|
|
2786
2686
|
function createAgentImprovementProposal(options) {
|
|
2787
|
-
const findings =
|
|
2687
|
+
const findings = assertProposalFindings(options.findings, "createAgentImprovementProposal findings");
|
|
2788
2688
|
const { evaluation, changedSurfaces } = validateShippableAgentImprovementEvaluation(options.evaluation, options.runId, "agent improvement proposal");
|
|
2789
2689
|
return agentImprovementProposalSchema.parse(canonicalCandidateDocument({
|
|
2790
2690
|
kind: "agent-improvement-proposal",
|
|
@@ -2844,7 +2744,7 @@ function verifyAgentImprovementProposal(input) {
|
|
|
2844
2744
|
const proposal = verifyCanonicalCandidateDocument(agentImprovementProposalSchema.parse(input), "agent improvement proposal");
|
|
2845
2745
|
const { changedSurfaces } = validateShippableAgentImprovementEvaluation(proposal.evaluation, proposal.runId, "agent improvement proposal");
|
|
2846
2746
|
if (!(proposal.evaluation.kind === "agent-profile-improvement-measured-comparison" ? sameAgentImprovementSurfaceSet(proposal.changedSurfaces, changedSurfaces) : sameOrderedValues(proposal.changedSurfaces, changedSurfaces))) throw new Error("proposal changed surfaces do not match its exact experiment");
|
|
2847
|
-
|
|
2747
|
+
assertProposalFindings(proposal.findings, "agent improvement proposal findings");
|
|
2848
2748
|
return proposal;
|
|
2849
2749
|
}
|
|
2850
2750
|
/** Return a sealed bundle experiment; ordinary profile changes need a product-owned executor. */
|
|
@@ -2945,13 +2845,7 @@ function assertEvidenceMaterialDigest(evidence, label) {
|
|
|
2945
2845
|
function sameOrderedValues(left, right) {
|
|
2946
2846
|
return left.length === right.length && left.every((value, index) => value === right[index]);
|
|
2947
2847
|
}
|
|
2948
|
-
function assertNoJudgeDerivedProposalFindings(findings) {
|
|
2949
|
-
const leaked = findings.filter((finding) => finding.derived_from_judge === true);
|
|
2950
|
-
if (leaked.length === 0) return;
|
|
2951
|
-
const identifiers = leaked.map((finding) => typeof finding.finding_id === "string" ? finding.finding_id : "<unknown>");
|
|
2952
|
-
throw new Error(`agent improvement proposal findings: judge-derived findings cannot steer an improvement: [${identifiers.join(", ")}]`);
|
|
2953
|
-
}
|
|
2954
2848
|
//#endregion
|
|
2955
|
-
export { improve as A,
|
|
2849
|
+
export { improve as A, agenticGenerator as B, agentImprovementTargetInput as C, buildAgentImprovementActivationTargets as D, assertProfileImprovementTargetsShareIdentity as E, normalizeRolloutPolicy as F, summarizeFindings as G, defaultBuildPrompt as H, parseRolloutPolicy as I, worktreeChangedPaths as K, serializeRolloutPolicy as L, rawTraceDistiller as M, ROLLOUT_POLICY_EXTENSION as N, isAgentImprovementProfileSurface as O, applyRolloutPolicyToProfile as P, structuralRolloutPolicyFromProfile as R, agentImprovementTargetDigest as S, agentProfileImprovementStateDigest as T, rawTraceEvidenceProblem as U, commandVerifier as V, requiresRawTraceEvidence as W, AGENT_IMPROVEMENT_PROFILE_SURFACES as _, executeAgentCandidateExperimentCell as a, agentImprovementProfileSurfaceDigest as b, requireSealedCandidateExperiment as c, verifyAgentImprovementActivation as d, verifyAgentImprovementProposal as f, optimizationActivationReceiptFromMetadata as g, createOptimizationActivationReceipt as h, createAgentImprovementProposal as i, withMethodRuntimeControls as j, isAgentProfileMeasuredSurface as k, reviewAgentImprovementProposal as l, verifyCandidateExecutionEvidence as m, createAgentImprovementActivation as n, proposeAgentImprovement as o, verifyAgentImprovementReview as p, createAgentImprovementMeasuredComparison as r, proposeAgentProfileImprovement as s, AgentCandidateExperimentCellExecutionError as t, runAgentCandidateExperiment as u, AGENT_PROFILE_MEASURED_SURFACES as v, agentImprovementTargetProfileDiffs as w, agentImprovementProfileSurfaceInput as x, agentImprovementProfileDiffs as y, AGENTIC_PROFILE_RESOURCE_ROOT as z };
|
|
2956
2850
|
|
|
2957
|
-
//# sourceMappingURL=improvement-cycle-
|
|
2851
|
+
//# sourceMappingURL=improvement-cycle-CRnDDdX0.js.map
|