@tangle-network/agent-runtime 0.116.0 → 0.119.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -1
- package/dist/{activation-DuqQhee6.js → activation-DdIpwQ0k.js} +3 -3
- package/dist/{activation-DuqQhee6.js.map → activation-DdIpwQ0k.js.map} +1 -1
- package/dist/agent.d.ts +3 -64
- package/dist/agent.js +5 -207
- package/dist/agent.js.map +1 -1
- package/dist/{analyst-loop-BoNIG2hA.js → analyst-loop-DvSciOfB.js} +2 -2
- package/dist/{analyst-loop-BoNIG2hA.js.map → analyst-loop-DvSciOfB.js.map} +1 -1
- package/dist/analyst-loop.js +1 -1
- package/dist/candidate-execution/index.d.ts +3 -3
- package/dist/candidate-execution/index.js +5 -5
- package/dist/{candidate-execution-CfpJrd3o.js → candidate-execution-DDkSRPjY.js} +4 -4
- package/dist/{candidate-execution-CfpJrd3o.js.map → candidate-execution-DDkSRPjY.js.map} +1 -1
- package/dist/{environment-provider-CWsRh6Uz.d.ts → environment-provider-Bh4nX2qt.d.ts} +306 -32
- package/dist/{environment-provider-CCaEhA-l.js → environment-provider-Cr75N-rU.js} +141 -32
- package/dist/environment-provider-Cr75N-rU.js.map +1 -0
- package/dist/environment-provider.d.ts +1 -1
- package/dist/environment-provider.js +1 -1
- package/dist/{improvement-cycle-C1cmjvPD.js → improvement-cycle-zjR-MXwK.js} +163 -22
- package/dist/improvement-cycle-zjR-MXwK.js.map +1 -0
- package/dist/{index-C-FYUuFG.d.ts → index-B3bHsVLw.d.ts} +2 -2
- package/dist/{index-CYkDeM5L.d.ts → index-BLsKcxNd.d.ts} +3 -3
- package/dist/{index-COumPQka.d.ts → index-CGADWaa_.d.ts} +864 -437
- package/dist/{index-DcLMNnG5.d.ts → index-I35151Fr.d.ts} +9 -9
- package/dist/index.d.ts +9 -9
- package/dist/index.js +13 -13
- package/dist/index.js.map +1 -1
- package/dist/intelligence.d.ts +10 -9
- package/dist/intelligence.js +15 -9
- package/dist/intelligence.js.map +1 -1
- package/dist/kernel.d.ts +4 -4
- package/dist/kernel.js +9 -8
- package/dist/{knowledge-DOzbywZT.js → knowledge-I1e9zQGj.js} +7 -7
- package/dist/knowledge-I1e9zQGj.js.map +1 -0
- package/dist/knowledge.d.ts +1 -1
- package/dist/knowledge.js +1 -1
- package/dist/{local-harness-t6cDWDQ2.d.ts → local-harness-BIajef4A.d.ts} +29 -7
- package/dist/{loop-runner-bin-BFrhPLKt.d.ts → loop-runner-bin-BeG9vTdE.d.ts} +3 -3
- package/dist/{loop-runner-bin-BuQjc5DR.js → loop-runner-bin-L996YVPS.js} +4 -4
- package/dist/{loop-runner-bin-BuQjc5DR.js.map → loop-runner-bin-L996YVPS.js.map} +1 -1
- package/dist/loop-runner-bin.d.ts +1 -1
- package/dist/loop-runner-bin.js +1 -1
- package/dist/mcp/bin.js +43 -10
- package/dist/mcp/bin.js.map +1 -1
- package/dist/mcp/index.d.ts +5 -45
- package/dist/mcp/index.js +8 -212
- package/dist/mcp/index.js.map +1 -1
- package/dist/{openai-tools-_Wyp4udO.js → openai-tools-CvFEzw6f.js} +2 -2
- package/dist/{openai-tools-_Wyp4udO.js.map → openai-tools-CvFEzw6f.js.map} +1 -1
- package/dist/{prepare-BHQBb02e.js → prepare-DpV6np9e.js} +47 -34
- package/dist/prepare-DpV6np9e.js.map +1 -0
- package/dist/primeintellect/index.d.ts +1 -1
- package/dist/{protected-model-port-BP6Z4eau.d.ts → protected-model-port-CoJOiINX.d.ts} +11 -3
- package/dist/{protected-model-port-DqAH1Z2M.js → protected-model-port-vQLBRoAQ.js} +2 -2
- package/dist/{protected-model-port-DqAH1Z2M.js.map → protected-model-port-vQLBRoAQ.js.map} +1 -1
- package/dist/{redact-BEtQtvd6.d.ts → redact-PQzmE1Jn.d.ts} +4 -4
- package/dist/run-layout-C2FGmZ3v.js +158 -0
- package/dist/run-layout-C2FGmZ3v.js.map +1 -0
- package/dist/{runtime-Ut1pkd2n.js → runtime-D4e3GP5K.js} +147 -190
- package/dist/runtime-D4e3GP5K.js.map +1 -0
- package/dist/{sandbox-events-DeI5xX8P.js → sandbox-events-Yhd1GYWl.js} +4 -2
- package/dist/sandbox-events-Yhd1GYWl.js.map +1 -0
- package/dist/spawn-journal-IeXpidO2.js +857 -0
- package/dist/spawn-journal-IeXpidO2.js.map +1 -0
- package/dist/{structural-rollout-CVY_0hJp.js → structural-rollout-DQHO3b2Y.js} +4 -4
- package/dist/structural-rollout-DQHO3b2Y.js.map +1 -0
- package/dist/{supervise-BUR9ByF7.js → supervise-Cq50UT5-.js} +2874 -1892
- package/dist/supervise-Cq50UT5-.js.map +1 -0
- package/dist/{supervisor-BBbPBXpe.js → supervisor-CpT9yAxL.js} +2896 -983
- package/dist/supervisor-CpT9yAxL.js.map +1 -0
- package/dist/testing.js +98 -76
- package/dist/testing.js.map +1 -1
- package/dist/top-app-5unxqovu.js +1088 -0
- package/dist/top-app-5unxqovu.js.map +1 -0
- package/dist/tui/bin.d.ts +1 -0
- package/dist/tui/bin.js +19 -0
- package/dist/tui/bin.js.map +1 -0
- package/dist/tui/index.d.ts +209 -0
- package/dist/tui/index.js +2 -0
- package/dist/{util-Cc9g9Y-o.js → util-MVgdwuIS.js} +2 -2
- package/dist/{util-Cc9g9Y-o.js.map → util-MVgdwuIS.js.map} +1 -1
- package/dist/{workspace-archive-DXzJq7WP.js → workspace-archive-B4SkNJjw.js} +2 -2
- package/dist/{workspace-archive-DXzJq7WP.js.map → workspace-archive-B4SkNJjw.js.map} +1 -1
- package/package.json +8 -2
- package/dist/environment-provider-CCaEhA-l.js.map +0 -1
- package/dist/improvement-cycle-C1cmjvPD.js.map +0 -1
- package/dist/knowledge-DOzbywZT.js.map +0 -1
- package/dist/prepare-BHQBb02e.js.map +0 -1
- package/dist/runtime-Ut1pkd2n.js.map +0 -1
- package/dist/sandbox-events-DeI5xX8P.js.map +0 -1
- package/dist/spawn-journal-DCPbicXB.js +0 -457
- package/dist/spawn-journal-DCPbicXB.js.map +0 -1
- package/dist/structural-rollout-CVY_0hJp.js.map +0 -1
- package/dist/supervise-BUR9ByF7.js.map +0 -1
- package/dist/supervisor-BBbPBXpe.js.map +0 -1
|
@@ -1,17 +1,16 @@
|
|
|
1
1
|
import { n as AnalystError, r as BackendTransportError, s as PlannerError, u as ValidationError } from "./errors-DEAvWQPy.js";
|
|
2
2
|
import { i as normalizeBackendStreamEvent, o as newRuntimeSession, s as nowIso } from "./backends-CiOCyRHb.js";
|
|
3
|
-
import {
|
|
4
|
-
import {
|
|
3
|
+
import { i as InMemorySpawnJournal, r as InMemoryResultBlobStore, v as isTraceAnalysisStore } from "./spawn-journal-IeXpidO2.js";
|
|
4
|
+
import { a as randomSuffix, c as stringifySafe, f as zeroTokenUsage, r as isAbortError, s as sleep, t as addTokenUsage } from "./util-MVgdwuIS.js";
|
|
5
5
|
import { i as redactProtectedValue, r as redactProtectedReason } from "./protected-redaction--F3v1oo8.js";
|
|
6
|
-
import {
|
|
7
|
-
import { C as observe, O as strategyAuthorMethod, b as sample, v as refine, x as sampleThenRefine, y as runAgentic } from "./structural-rollout-
|
|
6
|
+
import { X as rollingDispatch, dt as routerBrain, gt as runBrainLoop, l as withDriverExecutor, m as settledToIteration, n as createSupervisor, pt as routerChatWithUsage } from "./supervisor-CpT9yAxL.js";
|
|
7
|
+
import { C as observe, O as strategyAuthorMethod, b as sample, v as refine, x as sampleThenRefine, y as runAgentic } from "./structural-rollout-DQHO3b2Y.js";
|
|
8
8
|
import { i as notifyRuntimeHookEvent } from "./runtime-hooks-C7iJOWm3.js";
|
|
9
|
-
import { a as notifySandboxEventObserver, i as mapSandboxToolEvent, r as mapSandboxEvent, t as createSandboxToolPartState } from "./sandbox-events-
|
|
10
|
-
import { At as
|
|
11
|
-
import { CODING_HARNESSES, InMemoryTraceStore, benjaminiHochberg, buildTrajectory, computeFindingId as computeFindingId$1, confidenceInterval, expandProfileAxes, harnessAxisOf, makeFinding as makeFinding$1, pairedBootstrap, paretoFrontier, scoreKnowledgeReadiness, wilcoxonSignedRank, wilson } from "@tangle-network/agent-eval";
|
|
9
|
+
import { a as notifySandboxEventObserver, i as mapSandboxToolEvent, r as mapSandboxEvent, t as createSandboxToolPartState } from "./sandbox-events-Yhd1GYWl.js";
|
|
10
|
+
import { At as runAgentRounds, Kt as gateOnDeliverable, Mt as createSandboxLineage, Nt as probeSandboxCapabilities, bt as createExecutorRegistry, kt as defaultSelectWinner, n as supervise, wt as createPushTraceSource, xt as createWorktreeCliExecutor, yt as createExecutor } from "./supervise-Cq50UT5-.js";
|
|
11
|
+
import { CODING_HARNESSES, InMemoryTraceStore, OUTPUT_VALUE, benjaminiHochberg, buildTrajectory, computeFindingId as computeFindingId$1, confidenceInterval, expandProfileAxes, harnessAxisOf, makeFinding as makeFinding$1, pairedBootstrap, paretoFrontier, scoreKnowledgeReadiness, wilcoxonSignedRank, wilson } from "@tangle-network/agent-eval";
|
|
12
12
|
import { heldoutSignificance, runProfileMatrix } from "@tangle-network/agent-eval/campaign";
|
|
13
13
|
import { canonicalCandidateDigest, validateAgentProfileSecurity } from "@tangle-network/agent-interface";
|
|
14
|
-
import { randomUUID } from "node:crypto";
|
|
15
14
|
import { mkdir, readFile, writeFile } from "node:fs/promises";
|
|
16
15
|
import { appendFileSync, chmodSync, constants, copyFileSync, existsSync, lstatSync, mkdirSync, mkdtempSync, readFileSync, readlinkSync, rmSync, symlinkSync, writeFileSync } from "node:fs";
|
|
17
16
|
import { dirname, isAbsolute, join, relative, resolve, sep } from "node:path";
|
|
@@ -2905,7 +2904,7 @@ async function trajectoryReport(journal, blobs, root, options = {}) {
|
|
|
2905
2904
|
const events = await journal.loadTree(root);
|
|
2906
2905
|
if (events === void 0) throw new Error(`trajectoryReport: no journaled tree for root '${root}'`);
|
|
2907
2906
|
const spawns = events.filter(isNodeCreation).sort(bySeq);
|
|
2908
|
-
const closes = events.filter((ev) => ev.kind !== "spawned" && ev.kind !== "waiting" && ev.kind !== "metered").sort(bySeq);
|
|
2907
|
+
const closes = events.filter((ev) => ev.kind !== "spawned" && ev.kind !== "waiting" && ev.kind !== "metered" && ev.kind !== "materialized" && ev.kind !== "execution-bound").sort(bySeq);
|
|
2909
2908
|
const nodes = /* @__PURE__ */ new Map();
|
|
2910
2909
|
for (const ev of spawns) nodes.set(ev.id, {
|
|
2911
2910
|
id: ev.id,
|
|
@@ -3068,6 +3067,7 @@ function addNodeSpend(a, b) {
|
|
|
3068
3067
|
input: a.tokens.input + b.tokens.input,
|
|
3069
3068
|
output: a.tokens.output + b.tokens.output
|
|
3070
3069
|
},
|
|
3070
|
+
...a.tokensKnown === false || b.tokensKnown === false ? { tokensKnown: false } : {},
|
|
3071
3071
|
usd: a.usd + b.usd,
|
|
3072
3072
|
...a.tokensKnown === false || b.tokensKnown === false ? { tokensKnown: false } : {},
|
|
3073
3073
|
...a.usdKnown === false || b.usdKnown === false ? { usdKnown: false } : {},
|
|
@@ -3081,6 +3081,7 @@ function cloneSpend(spend) {
|
|
|
3081
3081
|
input: spend.tokens.input,
|
|
3082
3082
|
output: spend.tokens.output
|
|
3083
3083
|
},
|
|
3084
|
+
...spend.tokensKnown === false ? { tokensKnown: false } : {},
|
|
3084
3085
|
usd: spend.usd,
|
|
3085
3086
|
...spend.tokensKnown === false ? { tokensKnown: false } : {},
|
|
3086
3087
|
...spend.usdKnown === false ? { usdKnown: false } : {},
|
|
@@ -3091,6 +3092,7 @@ function cloneSpend(spend) {
|
|
|
3091
3092
|
function addSpend(acc, delta) {
|
|
3092
3093
|
acc.iterations += delta.iterations;
|
|
3093
3094
|
addTokenUsage(acc.tokens, delta.tokens);
|
|
3095
|
+
if (delta.tokensKnown === false) acc.tokensKnown = false;
|
|
3094
3096
|
acc.usd += delta.usd;
|
|
3095
3097
|
if (delta.tokensKnown === false) acc.tokensKnown = false;
|
|
3096
3098
|
if (delta.usdKnown === false) acc.usdKnown = false;
|
|
@@ -4775,140 +4777,6 @@ function signalPass(value, required) {
|
|
|
4775
4777
|
return !required;
|
|
4776
4778
|
}
|
|
4777
4779
|
//#endregion
|
|
4778
|
-
//#region src/runtime/supervise/run-layout.ts
|
|
4779
|
-
/**
|
|
4780
|
-
* The on-disk supervisor-run layout: `<root>/.agent/supervisor/<id>`.
|
|
4781
|
-
*
|
|
4782
|
-
* This is the durable, cross-process face of a supervisor run — the counterpart to the in-process
|
|
4783
|
-
* `Inbox` seam in `./inbox`. A run persists its state under one directory so that any OTHER process
|
|
4784
|
-
* can find it after the fact: `@tangle-network/traces` reads exactly this layout via
|
|
4785
|
-
* `traces analyze --supervisor-run-dir`, a restarted host can rehydrate a run it no longer holds
|
|
4786
|
-
* handles to, and a human can steer a live worker by appending one NDJSON line. Until now the
|
|
4787
|
-
* layout was defined only in the unpublished `loops` repo (`src/supervisor-control.ts`) — a
|
|
4788
|
-
* published reader depending on an unpublished writer's convention — so the contract is promoted
|
|
4789
|
-
* here, names preserved.
|
|
4790
|
-
*
|
|
4791
|
-
* `.agent` is the one dot-dir for ALL agent-owned state (skills already write
|
|
4792
|
-
* `.agent/hypotheses/`, `.agent/skill-runs.jsonl`); supervisor runs live beside them rather than
|
|
4793
|
-
* under a product-branded dir. Runs written by older writers used `.loops/supervisor/<id>` —
|
|
4794
|
-
* readers that must see those keep a legacy fallback; this writer never creates `.loops` again.
|
|
4795
|
-
*
|
|
4796
|
-
* Layout, relative to `supervisorRunDir(root, id)`:
|
|
4797
|
-
*
|
|
4798
|
-
* workers/<label>.inbox.ndjson down-leg steer/answer requests for one worker (durable inbox);
|
|
4799
|
-
* each line is a {@link WorkerSteerRequest}
|
|
4800
|
-
* workers/<label>.ndjson best-effort per-worker control-event log (delivery bookkeeping)
|
|
4801
|
-
*
|
|
4802
|
-
* Reads are tolerant by contract: a partial trailing line (a writer mid-append) or a corrupt line
|
|
4803
|
-
* never poisons the rest of the file — later valid lines still matter.
|
|
4804
|
-
*
|
|
4805
|
-
* Promoted from `loops/src/supervisor-control.ts`. The one deliberate difference: the loops version
|
|
4806
|
-
* resolved a worker id to its label through the run journal before writing a steer; that resolution
|
|
4807
|
-
* stays with the caller (it is journal-format-specific), so `writeWorkerSteer` here takes the worker
|
|
4808
|
-
* LABEL directly.
|
|
4809
|
-
*
|
|
4810
|
-
* @experimental
|
|
4811
|
-
*/
|
|
4812
|
-
/** The root every supervisor run of one workspace lives under. */
|
|
4813
|
-
function supervisorRunsRoot(rootDir) {
|
|
4814
|
-
return join(resolve(rootDir), ".agent", "supervisor");
|
|
4815
|
-
}
|
|
4816
|
-
/** The run directory every artifact of one supervisor run lives under. */
|
|
4817
|
-
function supervisorRunDir(rootDir, id) {
|
|
4818
|
-
return join(supervisorRunsRoot(rootDir), id);
|
|
4819
|
-
}
|
|
4820
|
-
/**
|
|
4821
|
-
* Where a pre-rename writer put the same run (`<root>/.loops/supervisor/<id>`). Readers that must
|
|
4822
|
-
* see historical runs check {@link supervisorRunDir} first and fall back to this; nothing writes
|
|
4823
|
-
* here anymore.
|
|
4824
|
-
*/
|
|
4825
|
-
function legacySupervisorRunDir(rootDir, id) {
|
|
4826
|
-
return join(resolve(rootDir), ".loops", "supervisor", id);
|
|
4827
|
-
}
|
|
4828
|
-
/** A worker label reduced to a safe filename stem. Empty labels get a stable fallback. */
|
|
4829
|
-
function safeWorkerFile(label) {
|
|
4830
|
-
const safe = label.replace(/[^A-Za-z0-9._-]/g, "_");
|
|
4831
|
-
return safe.length > 0 ? safe : "worker";
|
|
4832
|
-
}
|
|
4833
|
-
/** The durable inbox file for one worker of one run. */
|
|
4834
|
-
function workerInboxFile(rootDir, supervisorId, worker) {
|
|
4835
|
-
return workerInboxFileFromEventDir(supervisorRunDir(rootDir, supervisorId), worker);
|
|
4836
|
-
}
|
|
4837
|
-
/** Same, addressed from an already-known run directory (the reader's usual entry point). */
|
|
4838
|
-
function workerInboxFileFromEventDir(eventDir, worker) {
|
|
4839
|
-
return join(eventDir, "workers", `${safeWorkerFile(worker)}.inbox.ndjson`);
|
|
4840
|
-
}
|
|
4841
|
-
/**
|
|
4842
|
-
* Durably append one steer request to a worker's inbox and log the delivery attempt.
|
|
4843
|
-
*
|
|
4844
|
-
* The inbox append is the durable act; the control-event log is best-effort bookkeeping and may
|
|
4845
|
-
* silently fail without voiding the steer.
|
|
4846
|
-
*/
|
|
4847
|
-
function writeWorkerSteer(rootDir, supervisorId, worker, message, source = "human") {
|
|
4848
|
-
const trimmed = message.trim();
|
|
4849
|
-
if (!trimmed) throw new Error("steer message is empty");
|
|
4850
|
-
const dir = supervisorRunDir(rootDir, supervisorId);
|
|
4851
|
-
mkdirSync(join(dir, "workers"), { recursive: true });
|
|
4852
|
-
const request = {
|
|
4853
|
-
id: randomUUID(),
|
|
4854
|
-
at: (/* @__PURE__ */ new Date()).toISOString(),
|
|
4855
|
-
source,
|
|
4856
|
-
worker,
|
|
4857
|
-
message: trimmed
|
|
4858
|
-
};
|
|
4859
|
-
const file = workerInboxFile(rootDir, supervisorId, worker);
|
|
4860
|
-
appendFileSync(file, `${JSON.stringify(request)}\n`, "utf8");
|
|
4861
|
-
appendWorkerControlEvent(dir, worker, {
|
|
4862
|
-
kind: "message",
|
|
4863
|
-
direction: "down",
|
|
4864
|
-
source,
|
|
4865
|
-
requestId: request.id,
|
|
4866
|
-
message: trimmed,
|
|
4867
|
-
queued: true,
|
|
4868
|
-
delivered: false
|
|
4869
|
-
});
|
|
4870
|
-
return {
|
|
4871
|
-
worker,
|
|
4872
|
-
file,
|
|
4873
|
-
request
|
|
4874
|
-
};
|
|
4875
|
-
}
|
|
4876
|
-
/** Read every valid steer request in a worker's inbox. Corrupt or partial lines are skipped. */
|
|
4877
|
-
function readWorkerSteerRequests(eventDir, worker) {
|
|
4878
|
-
const file = workerInboxFileFromEventDir(eventDir, worker);
|
|
4879
|
-
if (!existsSync(file)) return [];
|
|
4880
|
-
const out = [];
|
|
4881
|
-
let raw = "";
|
|
4882
|
-
try {
|
|
4883
|
-
raw = readFileSync(file, "utf8");
|
|
4884
|
-
} catch {
|
|
4885
|
-
return out;
|
|
4886
|
-
}
|
|
4887
|
-
for (const line of raw.split("\n")) {
|
|
4888
|
-
const trimmed = line.trim();
|
|
4889
|
-
if (!trimmed) continue;
|
|
4890
|
-
try {
|
|
4891
|
-
const parsed = JSON.parse(trimmed);
|
|
4892
|
-
if (isWorkerSteerRequest(parsed)) out.push(parsed);
|
|
4893
|
-
} catch {}
|
|
4894
|
-
}
|
|
4895
|
-
return out;
|
|
4896
|
-
}
|
|
4897
|
-
function appendWorkerControlEvent(eventDir, label, event) {
|
|
4898
|
-
try {
|
|
4899
|
-
const workersDir = join(eventDir, "workers");
|
|
4900
|
-
mkdirSync(workersDir, { recursive: true });
|
|
4901
|
-
appendFileSync(join(workersDir, `${safeWorkerFile(label)}.ndjson`), `${JSON.stringify({
|
|
4902
|
-
at: (/* @__PURE__ */ new Date()).toISOString(),
|
|
4903
|
-
label,
|
|
4904
|
-
...event
|
|
4905
|
-
})}\n`, "utf8");
|
|
4906
|
-
} catch {}
|
|
4907
|
-
}
|
|
4908
|
-
function isWorkerSteerRequest(value) {
|
|
4909
|
-
return typeof value.id === "string" && typeof value.at === "string" && typeof value.source === "string" && typeof value.worker === "string" && typeof value.message === "string" && value.message.trim().length > 0;
|
|
4910
|
-
}
|
|
4911
|
-
//#endregion
|
|
4912
4780
|
//#region src/runtime/supervise/trajectory-recorder.ts
|
|
4913
4781
|
/**
|
|
4914
4782
|
*
|
|
@@ -5213,24 +5081,31 @@ function worktreeFanout(options) {
|
|
|
5213
5081
|
...options.require !== void 0 ? { require: options.require } : {}
|
|
5214
5082
|
});
|
|
5215
5083
|
const itemSpec = (item) => {
|
|
5216
|
-
const
|
|
5217
|
-
|
|
5218
|
-
|
|
5219
|
-
|
|
5220
|
-
|
|
5221
|
-
|
|
5222
|
-
|
|
5223
|
-
|
|
5224
|
-
|
|
5225
|
-
|
|
5226
|
-
|
|
5227
|
-
|
|
5228
|
-
|
|
5229
|
-
|
|
5084
|
+
const executorFactory = (_spec, ctx) => {
|
|
5085
|
+
if (!ctx.node) throw new Error("worktreeFanout: supervised node context required");
|
|
5086
|
+
return gateOnDeliverable(createWorktreeCliExecutor({
|
|
5087
|
+
repoRoot: options.repoRoot,
|
|
5088
|
+
profile: item.profile,
|
|
5089
|
+
harness: item.harness,
|
|
5090
|
+
taskPrompt: options.taskPrompt,
|
|
5091
|
+
executionAttemptId: ctx.node.attemptId,
|
|
5092
|
+
...item.budgetExempt !== void 0 ? { budgetExempt: item.budgetExempt } : {},
|
|
5093
|
+
...item.codexReproducible !== void 0 ? { codexReproducible: item.codexReproducible } : {},
|
|
5094
|
+
...item.codexReadDeniedPaths !== void 0 ? { codexReadDeniedPaths: item.codexReadDeniedPaths } : {},
|
|
5095
|
+
...item.runId ? { runId: item.runId } : {},
|
|
5096
|
+
...item.baseRef ? { baseRef: item.baseRef } : {},
|
|
5097
|
+
...options.testCmd !== void 0 ? { testCmd: options.testCmd } : {},
|
|
5098
|
+
...options.typecheckCmd !== void 0 ? { typecheckCmd: options.typecheckCmd } : {},
|
|
5099
|
+
...options.harnessTimeoutMs !== void 0 ? { harnessTimeoutMs: options.harnessTimeoutMs } : {},
|
|
5100
|
+
...options.runGit ? { runGit: options.runGit } : {},
|
|
5101
|
+
...options.runHarness ? { runHarness: options.runHarness } : {},
|
|
5102
|
+
...options.runCommand ? { runCommand: options.runCommand } : {}
|
|
5103
|
+
}), deliverable);
|
|
5104
|
+
};
|
|
5230
5105
|
return {
|
|
5231
5106
|
profile: item.profile,
|
|
5232
5107
|
harness: null,
|
|
5233
|
-
|
|
5108
|
+
executorFactory
|
|
5234
5109
|
};
|
|
5235
5110
|
};
|
|
5236
5111
|
const selectWinner = selectValidWinner({
|
|
@@ -5246,29 +5121,63 @@ function worktreeFanout(options) {
|
|
|
5246
5121
|
}
|
|
5247
5122
|
//#endregion
|
|
5248
5123
|
//#region src/runtime/supervise-surface.ts
|
|
5249
|
-
/**
|
|
5250
|
-
*
|
|
5251
|
-
|
|
5124
|
+
/**
|
|
5125
|
+
* superviseSurface — drive a team of agents to solve a graded `AgenticSurface` task. ONE capability that
|
|
5126
|
+
* replaces the worker-seam + "self-improving supervisor" wrapper pair: the driver (`profile`) spawns
|
|
5127
|
+
* workers that each run `runAgentic` over the surface (`refine` by default), settle on the surface's OWN
|
|
5128
|
+
* check (settled ⟺ resolved — a worker that ran but didn't pass settles invalid, so a keep-best driver
|
|
5129
|
+
* never counts it done), and feed the driver a self-improvement lens (the still-FAILING tests, by default)
|
|
5130
|
+
* so the next spawn targets the persistently-hard cases. Returns the deployable outcome + the full
|
|
5131
|
+
* conserved spend.
|
|
5132
|
+
*
|
|
5133
|
+
* WHY this lives here and not as a `supervise()` backend: `runAgentic` depends on the supervise core
|
|
5134
|
+
* (`strategy.ts` → `supervise/`), so a surface-solving worker cannot be a supervise built-in without an
|
|
5135
|
+
* import cycle. It is therefore a COMPOSITION of `supervise()` + `runAgentic` at the layer above both —
|
|
5136
|
+
* the right home for "supervise over a graded surface". The within-run self-improvement is the analyst
|
|
5137
|
+
* (authored content, swap `analysts`); the across-run kind wraps this call in `improve()`.
|
|
5138
|
+
*/
|
|
5139
|
+
/** Instrument every real surface call with the shared push trace source. The last test report remains
|
|
5140
|
+
* available on `SurfaceWorkerOut` for compatibility, but analysts read only the persisted spans. */
|
|
5141
|
+
function traceSurfaceCalls(base) {
|
|
5252
5142
|
let lastReport = "";
|
|
5253
|
-
const
|
|
5254
|
-
name: base.name,
|
|
5255
|
-
open: (t) => base.open(t),
|
|
5256
|
-
tools: (t, h) => base.tools(t, h),
|
|
5257
|
-
async call(h, name, args) {
|
|
5258
|
-
const out = await base.call(h, name, args);
|
|
5259
|
-
if (name === "run_tests") lastReport = out;
|
|
5260
|
-
return out;
|
|
5261
|
-
},
|
|
5262
|
-
score: (t, h) => base.score(t, h),
|
|
5263
|
-
close: (h) => base.close(h)
|
|
5264
|
-
};
|
|
5265
|
-
const failing = () => {
|
|
5266
|
-
const body = /FAILING:\s*(.+)/i.exec(lastReport)?.[1];
|
|
5267
|
-
return body ? body.split(",").map((s) => s.trim()).filter(Boolean) : [];
|
|
5268
|
-
};
|
|
5143
|
+
const trace = createPushTraceSource();
|
|
5269
5144
|
return {
|
|
5270
|
-
surface
|
|
5271
|
-
|
|
5145
|
+
surface: {
|
|
5146
|
+
name: base.name,
|
|
5147
|
+
open: (t) => base.open(t),
|
|
5148
|
+
tools: (t, h) => base.tools(t, h),
|
|
5149
|
+
async call(h, name, args) {
|
|
5150
|
+
const startedAt = Date.now();
|
|
5151
|
+
const recordedArgs = structuredClone(args);
|
|
5152
|
+
try {
|
|
5153
|
+
const out = await base.call(h, name, args);
|
|
5154
|
+
trace.record({
|
|
5155
|
+
toolName: name,
|
|
5156
|
+
args: recordedArgs,
|
|
5157
|
+
result: out,
|
|
5158
|
+
status: out.startsWith("ERROR:") ? "error" : "ok",
|
|
5159
|
+
startedAt,
|
|
5160
|
+
endedAt: Date.now()
|
|
5161
|
+
});
|
|
5162
|
+
if (name === "run_tests") lastReport = out;
|
|
5163
|
+
return out;
|
|
5164
|
+
} catch (error) {
|
|
5165
|
+
trace.record({
|
|
5166
|
+
toolName: name,
|
|
5167
|
+
args: recordedArgs,
|
|
5168
|
+
result: `ERROR: ${error instanceof Error ? error.message : String(error)}`,
|
|
5169
|
+
status: "error",
|
|
5170
|
+
startedAt,
|
|
5171
|
+
endedAt: Date.now()
|
|
5172
|
+
});
|
|
5173
|
+
throw error;
|
|
5174
|
+
}
|
|
5175
|
+
},
|
|
5176
|
+
score: (t, h) => base.score(t, h),
|
|
5177
|
+
close: (h) => base.close(h)
|
|
5178
|
+
},
|
|
5179
|
+
failing: () => failingTestNames(lastReport),
|
|
5180
|
+
traceSource: trace.source
|
|
5272
5181
|
};
|
|
5273
5182
|
}
|
|
5274
5183
|
/** The default self-improvement LENS — authored content, not a code path. On each settled worker it hands
|
|
@@ -5282,20 +5191,68 @@ function failuresAnalyst() {
|
|
|
5282
5191
|
area: "progress"
|
|
5283
5192
|
}],
|
|
5284
5193
|
run: async (_kindId, trace) => {
|
|
5285
|
-
|
|
5286
|
-
|
|
5287
|
-
if (
|
|
5288
|
-
const failing =
|
|
5289
|
-
|
|
5290
|
-
return { summary: failing.length ? `${head}. STILL FAILING (${failing.length}): ${failing.slice(0, 12).join(", ")}. Spawn the next worker to fix exactly these; if a test keeps failing across workers, give it concrete guidance about that case.` : `${head}. (no failing-test list available this round)` };
|
|
5194
|
+
if (!isTraceAnalysisStore(trace)) return missingRunTestsEvidence();
|
|
5195
|
+
const report = await latestRunTestsReport(trace);
|
|
5196
|
+
if (report === void 0) return missingRunTestsEvidence();
|
|
5197
|
+
const failing = failingTestNames(report);
|
|
5198
|
+
return { summary: failing.length ? `Latest structured run_tests evidence reports STILL FAILING (${failing.length}): ${failing.join(", ")}. Spawn the next worker to fix exactly these; if a test keeps failing across workers, give it concrete guidance about that case.` : allTestsPassed(report) ? "Latest structured run_tests evidence reports every test passed; stop." : `Latest structured run_tests evidence contains no parseable failing-test names. Refusing to infer them from worker prose. run_tests output: ${report.slice(0, 300)}` };
|
|
5291
5199
|
}
|
|
5292
5200
|
};
|
|
5293
5201
|
}
|
|
5202
|
+
async function latestRunTestsReport(store) {
|
|
5203
|
+
const overview = await store.getOverview({ tool_names: ["run_tests"] });
|
|
5204
|
+
const candidates = [];
|
|
5205
|
+
let ordinal = 0;
|
|
5206
|
+
for (const traceId of overview.sample_trace_ids) {
|
|
5207
|
+
let spans = (await store.viewTrace({
|
|
5208
|
+
trace_id: traceId,
|
|
5209
|
+
per_attribute_byte_cap: 16384
|
|
5210
|
+
})).spans;
|
|
5211
|
+
if (spans === void 0) {
|
|
5212
|
+
const matches = await store.searchTrace({
|
|
5213
|
+
trace_id: traceId,
|
|
5214
|
+
regex_pattern: "run_tests",
|
|
5215
|
+
max_matches: 100
|
|
5216
|
+
});
|
|
5217
|
+
const spanIds = [...new Set(matches.hits.filter((hit) => hit.span_name === "run_tests").map((hit) => hit.span_id))];
|
|
5218
|
+
spans = spanIds.length ? (await store.viewSpans({
|
|
5219
|
+
trace_id: traceId,
|
|
5220
|
+
span_ids: spanIds,
|
|
5221
|
+
per_attribute_byte_cap: 16384
|
|
5222
|
+
})).spans : [];
|
|
5223
|
+
}
|
|
5224
|
+
for (const span of spans) {
|
|
5225
|
+
if (span.tool_name !== "run_tests") continue;
|
|
5226
|
+
const output = span.attributes[OUTPUT_VALUE];
|
|
5227
|
+
if (typeof output !== "string") continue;
|
|
5228
|
+
candidates.push({
|
|
5229
|
+
output,
|
|
5230
|
+
endedAt: span.end_time,
|
|
5231
|
+
ordinal: ordinal++
|
|
5232
|
+
});
|
|
5233
|
+
}
|
|
5234
|
+
}
|
|
5235
|
+
candidates.sort((left, right) => Date.parse(left.endedAt) - Date.parse(right.endedAt) || left.ordinal - right.ordinal);
|
|
5236
|
+
return candidates.at(-1)?.output;
|
|
5237
|
+
}
|
|
5238
|
+
function failingTestNames(report) {
|
|
5239
|
+
const body = /FAILING:\s*([^\n]+)/iu.exec(report)?.[1];
|
|
5240
|
+
if (body === void 0) return [];
|
|
5241
|
+
return body.replace(/\.\s+COLLECTION-BLOCKED:.*$/iu, "").replace(/\s*\(\+\d+\s+more\)\s*$/iu, "").split(",").map((name) => name.trim()).filter(Boolean);
|
|
5242
|
+
}
|
|
5243
|
+
function allTestsPassed(report) {
|
|
5244
|
+
const fraction = /(\d+)\s*\/\s*(\d+)\s+tests?\s+passed/iu.exec(report);
|
|
5245
|
+
return fraction !== null && Number(fraction[1]) === Number(fraction[2]);
|
|
5246
|
+
}
|
|
5247
|
+
function missingRunTestsEvidence() {
|
|
5248
|
+
return { summary: "Missing structured run_tests span evidence. Refusing to infer failing-test names from worker prose." };
|
|
5249
|
+
}
|
|
5294
5250
|
/** One spawned worker = one `runAgentic` attempt over the surface task. The driver's brief is threaded
|
|
5295
5251
|
* into the attempt (so a re-spawn can take a targeted angle, not an identical retry); `runAgentic` stamps
|
|
5296
5252
|
* real tokens/usd/ms, forwarded as `Spend`; the still-failing tests are captured for the analyst. */
|
|
5297
5253
|
function surfaceWorkerExecutor(surface, task, worker, strategy) {
|
|
5298
5254
|
let artifact;
|
|
5255
|
+
const traced = traceSurfaceCalls(surface);
|
|
5299
5256
|
return {
|
|
5300
5257
|
runtime: "surface-worker",
|
|
5301
5258
|
async execute(brief) {
|
|
@@ -5304,9 +5261,8 @@ function surfaceWorkerExecutor(surface, task, worker, strategy) {
|
|
|
5304
5261
|
...task,
|
|
5305
5262
|
systemPrompt: `${task.systemPrompt ?? ""}\n\n— Supervisor guidance for THIS attempt (incorporate it; do not just repeat a prior approach) —\n${guidance}`
|
|
5306
5263
|
} : task;
|
|
5307
|
-
const cap = captureFailures(surface);
|
|
5308
5264
|
const r = await runAgentic({
|
|
5309
|
-
surface:
|
|
5265
|
+
surface: traced.surface,
|
|
5310
5266
|
task: attemptTask,
|
|
5311
5267
|
strategy,
|
|
5312
5268
|
budget: worker.budget ?? 1,
|
|
@@ -5321,7 +5277,7 @@ function surfaceWorkerExecutor(surface, task, worker, strategy) {
|
|
|
5321
5277
|
score: r.score,
|
|
5322
5278
|
shots: r.shots,
|
|
5323
5279
|
summary: `${strategy.name} ${r.shots} shot(s) → ${(100 * r.score).toFixed(0)}% (${r.resolved ? "resolved" : "unresolved"})`,
|
|
5324
|
-
failing: r.resolved ? [] :
|
|
5280
|
+
failing: r.resolved ? [] : traced.failing()
|
|
5325
5281
|
};
|
|
5326
5282
|
const spent = {
|
|
5327
5283
|
iterations: r.completions,
|
|
@@ -5340,6 +5296,7 @@ function surfaceWorkerExecutor(surface, task, worker, strategy) {
|
|
|
5340
5296
|
};
|
|
5341
5297
|
return artifact;
|
|
5342
5298
|
},
|
|
5299
|
+
traceSource: () => traced.traceSource,
|
|
5343
5300
|
teardown: () => Promise.resolve({ destroyed: true }),
|
|
5344
5301
|
resultArtifact() {
|
|
5345
5302
|
if (!artifact) throw new Error("surfaceWorkerExecutor: resultArtifact before execute");
|
|
@@ -5820,6 +5777,6 @@ function tail(s) {
|
|
|
5820
5777
|
return s.slice(-400);
|
|
5821
5778
|
}
|
|
5822
5779
|
//#endregion
|
|
5823
|
-
export {
|
|
5780
|
+
export { verify as $, authorStrategy as A, createMcpEnvironment as At, runPersonified as B, collectAgentTurn as C, renderLeaderboardSvg as Ct, runStrategyEvolution as D, McpSpawnFault as Dt, pickChampion as E, defaultAuditorInstruction as Et, runBenchmark as F, resolveSecretEnv as Ft, InMemoryCorpus as G, createShapeRegistry as H, promotionGate as I, secretEnvOfMcpServer as It, flatWidenGate as J, renderCorpusToInstructions as K, equalKOnCost as L, SandboxRunAbortError as M, envKeyProvider as Mt, openSandboxRun as N, mcpSecretEnvMetadataKey as Nt, selectChampion as O, connectStdioMcp as Ot, printBenchmarkReport as P, resolveMcpServerLaunch as Pt, selectValidWinner as Q, trajectoryReport as R, runCoderChecks as S, renderLeaderboardMarkdown as St, discriminatingMeans as T, auditIntent as Tt, registerShape as U, builtinShapes as V, FileCorpus as W, panel as X, loopUntil as Y, pipeline as Z, settledWorkerOut as _, sentinelCompletion as _t, localShell as a, inProcessSandboxClient as at, analyzeTrace as b, pairwiseSignificance as bt, createVerifierEnvironment as c, dumbDriver as ct, worktreeFanout as d, localSandboxClient as dt, widen as et, EVIDENCE_MAX_CHARS as f, inlineSandboxClient as ft, composeWorkerEvidence as g, deterministicCompletion as gt, closingWorkerNote as h, completionAuthorizes as ht, jjWorkspace as i, registryScopeAnalyst as it, strategyAuthorContract as j, sanitizeMcpToolSchema as jt, assertStrategyContract as k, materializeLocalMcp as kt, failuresAnalyst as l, naiveDriver as lt, VERIFY_TAIL_CHARS as m, loopDispatch as mt, makeFinding$1 as n, buildSteerContext as nt, runInWorkspace as o, harvestCorpus as ot, NOTE_MAX_CHARS as p, loopCampaignDispatch as pt, fanout as q, gitWorkspace as r, createScopeAnalyst as rt, createWaterfallCollector as s, defineLeaderboard as st, computeFindingId$1 as t, assertTraceDerivedFindings as tt, superviseSurface as u, resolveSandboxClient as ut, copyUntrackedIntoClone as v, stopSentinel as vt, streamAgentTurn as w, renderPairwiseMarkdown as wt, patchDelivered as x, renderLeaderboardHtml as xt, withUntrackedArtifacts as y, leaderboard as yt, definePersona as z };
|
|
5824
5781
|
|
|
5825
|
-
//# sourceMappingURL=runtime-
|
|
5782
|
+
//# sourceMappingURL=runtime-D4e3GP5K.js.map
|