@tangle-network/agent-runtime 0.103.1 → 0.105.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +177 -39
- package/dist/agent.d.ts +1 -1
- package/dist/agent.js +5 -5
- package/dist/analyst-loop.d.ts +1 -1
- package/dist/candidate-execution/index.d.ts +2 -2
- package/dist/candidate-execution/index.js +4 -4
- package/dist/{chunk-TUSOOIDV.js → chunk-C5RTIBNZ.js} +2 -2
- package/dist/{chunk-SNSMRT6H.js → chunk-CTRA64LY.js} +3 -3
- package/dist/{chunk-ZXICDSAK.js → chunk-H5QPIZNX.js} +2 -2
- package/dist/{chunk-6WZZXQV5.js → chunk-HLKC4UYB.js} +1 -13
- package/dist/chunk-HLKC4UYB.js.map +1 -0
- package/dist/{chunk-AAN2MB2X.js → chunk-HNP72PNU.js} +3 -10
- package/dist/chunk-HNP72PNU.js.map +1 -0
- package/dist/{chunk-M6MD6JBS.js → chunk-KRBFHMV6.js} +429 -1
- package/dist/chunk-KRBFHMV6.js.map +1 -0
- package/dist/chunk-OPVWXJ2H.js +75 -0
- package/dist/chunk-OPVWXJ2H.js.map +1 -0
- package/dist/{chunk-AYU35OTU.js → chunk-PZZKQVQV.js} +1 -1
- package/dist/chunk-PZZKQVQV.js.map +1 -0
- package/dist/{chunk-B7K7V22Y.js → chunk-RDOAVVHY.js} +2 -2
- package/dist/{chunk-L5DST3QC.js → chunk-SMQXZGLZ.js} +1 -1
- package/dist/chunk-SMQXZGLZ.js.map +1 -0
- package/dist/{chunk-LFM4JBRW.js → chunk-UHEZW5BU.js} +1084 -527
- package/dist/chunk-UHEZW5BU.js.map +1 -0
- package/dist/{chunk-5AITUUHO.js → chunk-VISA6CI3.js} +3 -3
- package/dist/{chunk-ZOYN3JR5.js → chunk-WMTCUOQL.js} +5 -5
- package/dist/{chunk-3LJF5XSE.js → chunk-WRTOVNN4.js} +6 -6
- package/dist/chunk-WRTOVNN4.js.map +1 -0
- package/dist/{chunk-SBTWKPVR.js → chunk-WSTRQZYQ.js} +1 -1
- package/dist/chunk-WSTRQZYQ.js.map +1 -0
- package/dist/{chunk-EAQ5YRRY.js → chunk-XBG2W2VW.js} +50 -23
- package/dist/chunk-XBG2W2VW.js.map +1 -0
- package/dist/{chunk-QYCKIV6C.js → chunk-YJZA2BIK.js} +2 -2
- package/dist/{chunk-QYCKIV6C.js.map → chunk-YJZA2BIK.js.map} +1 -1
- package/dist/{completion-gate-DLINnrkM.d.ts → completion-gate-BMy5LGoP.d.ts} +3 -3
- package/dist/{coordination-DTehA977.d.ts → coordination-BZZSVYpZ.d.ts} +28 -28
- package/dist/environment-provider.d.ts +18 -7
- package/dist/environment-provider.js +3 -1
- package/dist/index.d.ts +106 -182
- package/dist/index.js +374 -236
- package/dist/index.js.map +1 -1
- package/dist/intelligence.d.ts +93 -47
- package/dist/intelligence.js +127 -73
- package/dist/intelligence.js.map +1 -1
- package/dist/knowledge.d.ts +6 -6
- package/dist/knowledge.js +9 -9
- package/dist/{local-harness-CtK9dSny.d.ts → local-harness-BDNcl6jI.d.ts} +1 -1
- package/dist/{loop-runner-bin-DhIXsLAd.d.ts → loop-runner-bin-68aoM9-7.d.ts} +5 -13
- package/dist/loop-runner-bin.d.ts +6 -8
- package/dist/loop-runner-bin.js +6 -6
- package/dist/loops.d.ts +67 -42
- package/dist/loops.js +10 -6
- package/dist/mcp/bin.js +4 -4
- package/dist/mcp/index.d.ts +9 -10
- package/dist/mcp/index.js +11 -11
- package/dist/mcp/index.js.map +1 -1
- package/dist/mcp/memory-bin.js +1 -1
- package/dist/primeintellect/index.d.ts +3 -4
- package/dist/primeintellect/index.js +5 -6
- package/dist/primeintellect/index.js.map +1 -1
- package/dist/profiles.d.ts +6 -6
- package/dist/profiles.js.map +1 -1
- package/dist/{protected-model-port-DzkUY3HQ.d.ts → protected-model-port-B4t-OQjL.d.ts} +1 -1
- package/dist/{improve-B40dRu0X.d.ts → redact-BuFjxBUJ.d.ts} +201 -154
- package/dist/{structural-rollout-BFPIy3iw.d.ts → structural-rollout-DEh6CEsa.d.ts} +4 -4
- package/dist/{supervise-Cp8bNcJi.d.ts → supervise-4_48wwvy.d.ts} +3 -3
- package/dist/testing.js +9 -9
- package/dist/testing.js.map +1 -1
- package/dist/{types-DK11_O4L.d.ts → types-BCoemcXU.d.ts} +4 -4
- package/dist/{types-lZTE_LBd.d.ts → types-CvNwMzQt.d.ts} +4 -4
- package/dist/{worktree-fanout-DiiJWjJO.d.ts → worktree-fanout-DxhAWr5Z.d.ts} +6 -6
- package/package.json +8 -7
- package/skills/build-with-agent-runtime/SKILL.md +15 -9
- package/skills/loop-writer/SKILL.md +1 -1
- package/dist/chunk-3LJF5XSE.js.map +0 -1
- package/dist/chunk-6WZZXQV5.js.map +0 -1
- package/dist/chunk-AAN2MB2X.js.map +0 -1
- package/dist/chunk-AYU35OTU.js.map +0 -1
- package/dist/chunk-EAQ5YRRY.js.map +0 -1
- package/dist/chunk-L5DST3QC.js.map +0 -1
- package/dist/chunk-LFM4JBRW.js.map +0 -1
- package/dist/chunk-M6MD6JBS.js.map +0 -1
- package/dist/chunk-SBTWKPVR.js.map +0 -1
- /package/dist/{chunk-TUSOOIDV.js.map → chunk-C5RTIBNZ.js.map} +0 -0
- /package/dist/{chunk-SNSMRT6H.js.map → chunk-CTRA64LY.js.map} +0 -0
- /package/dist/{chunk-ZXICDSAK.js.map → chunk-H5QPIZNX.js.map} +0 -0
- /package/dist/{chunk-B7K7V22Y.js.map → chunk-RDOAVVHY.js.map} +0 -0
- /package/dist/{chunk-5AITUUHO.js.map → chunk-VISA6CI3.js.map} +0 -0
- /package/dist/{chunk-ZOYN3JR5.js.map → chunk-WMTCUOQL.js.map} +0 -0
|
@@ -1,27 +1,14 @@
|
|
|
1
|
-
import { LabeledScenarioStore, WorktreeAdapter,
|
|
2
|
-
import {
|
|
3
|
-
import { AgentProfile, ReasoningEffort } from '@tangle-network/agent-interface';
|
|
4
|
-
import { L as LocalHarness, C as CodexTokenUsage,
|
|
1
|
+
import { LabeledScenarioStore, WorktreeAdapter, OptimizationMethod, CompareOptimizationMethodsOptions, OptimizationMethodComparison } from '@tangle-network/agent-eval/campaign';
|
|
2
|
+
import { MutableSurface, Scenario, SelfImproveResult, SelfImproveOptions, SelfImproveBudget } from '@tangle-network/agent-eval/contract';
|
|
3
|
+
import { AgentProfile, ReasoningEffort, Sha256Digest } from '@tangle-network/agent-interface';
|
|
4
|
+
import { L as LocalHarness, C as CodexTokenUsage, b as CodexExecutionEvidence, c as LocalHarnessResult, r as runLocalHarness } from './local-harness-BDNcl6jI.js';
|
|
5
5
|
import { AnalystFinding, CostLedgerHandle, MaximumCharge } from '@tangle-network/agent-eval';
|
|
6
6
|
|
|
7
7
|
/**
|
|
8
|
+
* Code-only candidate driver for Runtime-owned git worktrees.
|
|
8
9
|
*
|
|
9
|
-
* `
|
|
10
|
-
*
|
|
11
|
-
* the candidate lifecycle (worktree create → generate → finalize/discard,
|
|
12
|
-
* × populationSize); it delegates the only thing that genuinely varies — HOW
|
|
13
|
-
* a candidate change is produced — to a pluggable `CandidateGenerator`.
|
|
14
|
-
*
|
|
15
|
-
* There is no separate "analyst driver" vs "autoresearch driver": those are
|
|
16
|
-
* the SAME driver at two settings of a dial.
|
|
17
|
-
* - cheap reflective path → `reflectiveGenerator` (shots=1, no sandbox;
|
|
18
|
-
* applies pre-drafted patches)
|
|
19
|
-
* - full agentic path → `agenticGenerator` (shots=N, multi-shot
|
|
20
|
-
* verify-in-session loop; an agent reads code +
|
|
21
|
-
* report, edits, and re-tries on verifier failure)
|
|
22
|
-
* Both emit changes into a worktree the driver finalizes into a
|
|
23
|
-
* `CodeSurface{ worktreeRef }` the loop measures on the holdout. See
|
|
24
|
-
* agent-eval's `docs/design/self-improvement-engine.md`.
|
|
10
|
+
* A `CandidateGenerator` edits an isolated checkout. This driver finalizes each
|
|
11
|
+
* accepted edit as a `CodeSurface` and disposes rejected worktrees.
|
|
25
12
|
*
|
|
26
13
|
* @experimental
|
|
27
14
|
*/
|
|
@@ -56,7 +43,7 @@ interface CandidateGenerator {
|
|
|
56
43
|
* reflective generator ignores it). */
|
|
57
44
|
maxShots: number;
|
|
58
45
|
signal: AbortSignal;
|
|
59
|
-
/**
|
|
46
|
+
/** Generation coordinates supplied by Runtime's internal code candidate driver. */
|
|
60
47
|
generation?: number;
|
|
61
48
|
candidateIndex?: number;
|
|
62
49
|
/** Shared run-wide paid-call account supplied by agent-eval 0.117+. */
|
|
@@ -75,24 +62,10 @@ interface CandidateGenerator {
|
|
|
75
62
|
rationale?: string;
|
|
76
63
|
}>;
|
|
77
64
|
}
|
|
78
|
-
interface ImprovementDriverOptions {
|
|
79
|
-
worktree: WorktreeAdapter;
|
|
80
|
-
generator: CandidateGenerator;
|
|
81
|
-
/** Root ref for first-generation/direct callers. Default `main`.
|
|
82
|
-
* Later code generations retain the incumbent's original root. */
|
|
83
|
-
baseRef?: string;
|
|
84
|
-
}
|
|
85
|
-
interface ManagedImprovementDriver extends SurfaceProposer<AnalystFinding> {
|
|
86
|
-
/** Remove every owned candidate except explicitly retained finalized winners. */
|
|
87
|
-
cleanup(retainWorktreeRefs?: readonly string[]): Promise<void>;
|
|
88
|
-
}
|
|
89
|
-
/** The one reflective/agentic improvement proposer (`SurfaceProposer`): owns the candidate worktree lifecycle and delegates HOW a change is produced to a pluggable `CandidateGenerator`. */
|
|
90
|
-
declare function improvementDriver(opts: ImprovementDriverOptions): ManagedImprovementDriver;
|
|
91
65
|
|
|
92
66
|
/**
|
|
93
67
|
*
|
|
94
|
-
* `agenticGenerator` — the full-agentic `CandidateGenerator
|
|
95
|
-
* `shots=N, sandbox=on` setting of the one `improvementDriver`. It runs a real
|
|
68
|
+
* `agenticGenerator` — the full-agentic `CandidateGenerator`. It runs a real
|
|
96
69
|
* coding harness (claude / codex / opencode) inside the candidate worktree the
|
|
97
70
|
* driver already created, letting the agent read the codebase + the research
|
|
98
71
|
* report and make the change in place. The driver then commits the worktree
|
|
@@ -107,7 +80,7 @@ declare function improvementDriver(opts: ImprovementDriverOptions): ManagedImpro
|
|
|
107
80
|
* problem that does not need solving here).
|
|
108
81
|
*
|
|
109
82
|
* `maxShots` is the DEPTH dial — a multi-shot verify-in-session loop, NOT the
|
|
110
|
-
* kernel `
|
|
83
|
+
* kernel `runAgentRounds`. Each shot runs one full harness session in the (persistent)
|
|
111
84
|
* worktree; between shots the loop refines based on what the last shot produced:
|
|
112
85
|
* - empty tree → "you changed nothing, make the edits" → retry
|
|
113
86
|
* - dirty + `verify` fails → feed the verifier's failure into the next shot
|
|
@@ -270,106 +243,113 @@ declare function defaultBuildPrompt(args: {
|
|
|
270
243
|
* silent fallback). */
|
|
271
244
|
declare function commandVerifier(command: string, args?: string[], timeoutMs?: number): Verifier;
|
|
272
245
|
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
* loop). It removes the two things a caller otherwise has to know to drive the
|
|
279
|
-
* loop by hand: WHICH `MutableSurface` of the profile is being optimized, and
|
|
280
|
-
* WHICH `SurfaceProposer` mutates that surface. You name a `surface`; the
|
|
281
|
-
* facade picks the matching default proposer, extracts the baseline surface from
|
|
282
|
-
* the profile, and runs `selfImprove`. It returns a frozen candidate and never
|
|
283
|
-
* changes the input profile or caller-owned state.
|
|
284
|
-
*
|
|
285
|
-
* - `surface: 'prompt'` → `gepaProposer` mutates `profile.prompt.systemPrompt`.
|
|
286
|
-
* - `surface: 'skills'` → `skillOptProposer` mutates one named inline skill.
|
|
287
|
-
* - `surface: 'memory'` → `memoryCurationProposer` curates the profile's
|
|
288
|
-
* additional instructions as bounded durable lessons.
|
|
289
|
-
* - `surface: 'rollout-policy'` → `rolloutPolicyProposer` mutates the
|
|
290
|
-
* inference-time `StructuralRolloutPolicy` dials ({ k, repairRounds, testgen })
|
|
291
|
-
* persisted in `profile.extensions['structural-rollout']` — deterministic
|
|
292
|
-
* bounded neighbor enumeration; the held-out gate does the deciding. No-op
|
|
293
|
-
* (nothing proposed, nothing shipped) when the profile has no such extension.
|
|
294
|
-
* - `surface: 'agent-profile'` → caller-supplied proposer mutates the complete
|
|
295
|
-
* canonical AgentProfile JSON in one candidate.
|
|
296
|
-
* - `surface` ∈ {`tools`, `mcp`, `hooks`, `subagents`, `agent-profile`} → no zero-config default
|
|
297
|
-
* proposer exists (a code/config proposer needs caller-supplied wiring — a
|
|
298
|
-
* worktree repo root, a candidate generator, a serializer). The facade
|
|
299
|
-
* requires an explicit `opts.generator` for these and throws a `ConfigError`
|
|
300
|
-
* otherwise. This is a designed boundary, not a missing default: there is
|
|
301
|
-
* no safe value the facade could invent for those surfaces. Code instead
|
|
302
|
-
* requires `opts.code.repoRoot` and accepts only the runtime-owned
|
|
303
|
-
* `opts.code.generator` path so every isolated checkout can be released.
|
|
304
|
-
*
|
|
305
|
-
* Everything else (`scenarios`, `judge`, `agent`, `budget`, `llm`) passes
|
|
306
|
-
* straight through to `selfImprove`.
|
|
307
|
-
*
|
|
308
|
-
* @experimental
|
|
309
|
-
*/
|
|
246
|
+
type DeepReadonly<T> = T extends (...args: never[]) => unknown ? T : T extends readonly (infer TItem)[] ? readonly DeepReadonly<TItem>[] : T extends object ? {
|
|
247
|
+
readonly [TKey in keyof T]: DeepReadonly<T[TKey]>;
|
|
248
|
+
} : T;
|
|
249
|
+
/** Complete immutable profile value used during measured execution. */
|
|
250
|
+
type ReadonlyAgentProfile = DeepReadonly<AgentProfile>;
|
|
310
251
|
|
|
311
252
|
/** The executable agent lever `improve` optimizes. Profile fields remain
|
|
312
|
-
*
|
|
313
|
-
*
|
|
314
|
-
*
|
|
315
|
-
*
|
|
253
|
+
* portable AgentProfile coordinates; implementation and orchestration files
|
|
254
|
+
* use the code surface so a winner can be sealed into an exact candidate.
|
|
255
|
+
* `rollout-policy` is the inference-time structuralRollout dials
|
|
256
|
+
* (`profile.extensions['structural-rollout']`). */
|
|
316
257
|
type ImproveSurface = 'prompt' | 'skills' | 'tools' | 'mcp' | 'hooks' | 'subagents' | 'agent-profile' | 'memory' | 'code' | 'rollout-policy';
|
|
317
|
-
type
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
/**
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
258
|
+
type ImproveProfileSurface = Exclude<ImproveSurface, 'code'>;
|
|
259
|
+
interface ImproveMethodContext {
|
|
260
|
+
/** Validated baseline profile. */
|
|
261
|
+
readonly profile: ReadonlyAgentProfile;
|
|
262
|
+
/** Runtime-derived identity for upstream optimizer resume state. */
|
|
263
|
+
readonly evaluationRef: Sha256Digest;
|
|
264
|
+
/** Exact profile coordinate being optimized. */
|
|
265
|
+
readonly surface: ImproveProfileSurface;
|
|
266
|
+
/** Exact bytes supplied to the optimization method. */
|
|
267
|
+
readonly baselineSurface: MutableSurface;
|
|
268
|
+
/** Structured value represented by `baselineSurface`, before serialization. */
|
|
269
|
+
readonly baselineValue: unknown;
|
|
270
|
+
/** Findings produced before this search, if any. */
|
|
271
|
+
readonly findings: readonly unknown[];
|
|
272
|
+
}
|
|
273
|
+
/** Build a complete method after trace findings are available. */
|
|
274
|
+
type ImproveMethodFactory<TScenario extends Scenario, TArtifact> = (context: ImproveMethodContext) => OptimizationMethod<TScenario, TArtifact>;
|
|
275
|
+
type ImproveMethodSource<TScenario extends Scenario, TArtifact> = OptimizationMethod<TScenario, TArtifact> | ImproveMethodFactory<TScenario, TArtifact>;
|
|
276
|
+
/** Runs one exact materialized profile on one scenario. */
|
|
277
|
+
type ImproveProfileAgent<TScenario extends Scenario, TArtifact> = (profile: ReadonlyAgentProfile, scenario: TScenario, ctx: Parameters<CompareOptimizationMethodsOptions<TScenario, TArtifact>['dispatchWithSurface']>[2]) => Promise<TArtifact>;
|
|
278
|
+
/** Exact materialized profile presented for validation before any candidate run. */
|
|
279
|
+
interface ImproveCandidateValidationInput {
|
|
280
|
+
profile: ReadonlyAgentProfile;
|
|
281
|
+
surface: ImproveProfileSurface;
|
|
282
|
+
candidateSurface: MutableSurface;
|
|
283
|
+
value: unknown;
|
|
284
|
+
isBaseline: boolean;
|
|
285
|
+
}
|
|
286
|
+
type ImproveCandidateValidator = (input: ImproveCandidateValidationInput) => void;
|
|
287
|
+
type ImproveOptimizationRunOptions<TScenario extends Scenario, TArtifact> = Omit<NonNullable<CompareOptimizationMethodsOptions<TScenario, TArtifact>['optimizationRunOptions']>, 'dispatchRef'>;
|
|
288
|
+
/** Complete-method configuration for every non-code profile surface. */
|
|
289
|
+
type ImproveMethodOptions<TScenario extends Scenario, TArtifact> = Omit<CompareOptimizationMethodsOptions<TScenario, TArtifact>, 'baselineSurface' | 'dispatchRef' | 'dispatchWithSurface' | 'methods' | 'optimizationConcurrency' | 'optimizationRunOptions'> & {
|
|
290
|
+
/** Exact profile coordinate optimized by `method`. Default `'prompt'`. */
|
|
291
|
+
surface?: ImproveProfileSurface;
|
|
292
|
+
/**
|
|
293
|
+
* Immutable digest of `agent`, profile component mapping, models, tools, and
|
|
294
|
+
* every closure or external setting that can change measured behavior.
|
|
295
|
+
*/
|
|
296
|
+
executionRef: Sha256Digest;
|
|
297
|
+
/** A complete optimizer or a factory that can incorporate current findings. */
|
|
298
|
+
method: ImproveMethodSource<TScenario, TArtifact>;
|
|
299
|
+
/** Runs the exact complete profile materialized from one candidate surface. */
|
|
300
|
+
agent: ImproveProfileAgent<TScenario, TArtifact>;
|
|
301
|
+
/** Reject a materialized profile before it reaches the agent callback. */
|
|
302
|
+
validateCandidate?: ImproveCandidateValidator;
|
|
303
|
+
/** Trace or analyst findings available to a method factory. */
|
|
304
|
+
findings?: readonly unknown[];
|
|
305
|
+
/** Select the exact inline skill document for `surface: 'skills'`. */
|
|
306
|
+
skills?: ImproveSkillsOptions;
|
|
307
|
+
/**
|
|
308
|
+
* Map a profile to named text components and apply the winning components.
|
|
309
|
+
* Valid only with `surface: 'agent-profile'`.
|
|
310
|
+
*/
|
|
311
|
+
profileComponents?: ImproveProfileComponents;
|
|
312
|
+
/** Shared settings for method train and selection calls. */
|
|
313
|
+
optimizationRunOptions?: ImproveOptimizationRunOptions<TScenario, TArtifact>;
|
|
314
|
+
/** Ship only when the paired final-test interval is entirely above this lift. Default `0`. */
|
|
315
|
+
minimumLift?: number;
|
|
316
|
+
};
|
|
317
|
+
/** Runtime-owned code search in isolated git worktrees. */
|
|
318
|
+
type ImproveCodeRunOptions<TScenario extends Scenario, TArtifact> = Omit<SelfImproveOptions<TScenario, TArtifact>, 'analyzeGeneration' | 'baselineSurface' | 'budget' | 'findings' | 'gate' | 'llm' | 'method' | 'mutationPrimitives' | 'proposer' | 'proposerTarget' | 'selectionScenarios'> & {
|
|
319
|
+
surface: 'code';
|
|
320
|
+
/** Local code-search budget. Method-only selection controls do not apply. */
|
|
321
|
+
budget?: Omit<SelfImproveBudget, 'selectionFraction'>;
|
|
322
|
+
/** Findings supplied to Runtime's code candidate driver. */
|
|
323
|
+
findings?: readonly unknown[];
|
|
326
324
|
/** Gate mode. `'holdout'` (default) runs the held-out promotion gate;
|
|
327
|
-
*
|
|
325
|
+
* `'none'` is a baseline-only run (`budget.generations = 0`). */
|
|
328
326
|
gate?: 'holdout' | 'none';
|
|
329
|
-
/**
|
|
330
|
-
*
|
|
331
|
-
*
|
|
332
|
-
allowedModels?: readonly string[];
|
|
333
|
-
/** Per-generation findings producer passthrough (see selfImprove.analyzeGeneration).
|
|
334
|
-
* DEFAULT: with a real (non-`mem://`) `runDir`, the raw-trace distiller
|
|
335
|
-
* (`rawTraceDistiller`) — typed `AnalystFinding`s pointing the proposer at the
|
|
336
|
-
* prior generation's actual on-disk traces; for in-memory runs (no traces on
|
|
337
|
-
* disk to point at), the built-in failure distiller — the worst-scoring/errored
|
|
338
|
-
* cells distilled into typed `AnalystFinding`s for the NEXT proposal round.
|
|
339
|
-
* Pass your own producer to replace either; pass `null` to disable and keep the
|
|
340
|
-
* static `findings` all the way through. */
|
|
327
|
+
/** Per-generation findings producer for Runtime's code search.
|
|
328
|
+
* Pass your own producer to replace the code-trace distiller; pass `null`
|
|
329
|
+
* to keep the static findings for every generation. */
|
|
341
330
|
analyzeGeneration?: SelfImproveOptions<TScenario, TArtifact>['analyzeGeneration'] | null;
|
|
342
|
-
/**
|
|
343
|
-
*
|
|
344
|
-
* real run traces under `runDir` (per-cell `spans.jsonl` event logs +
|
|
345
|
-
* `cached-result.json` scores + artifacts) plus a `grep`/`cat`-to-diagnose
|
|
346
|
-
* instruction — so the coding agent reads the actual failures itself rather than
|
|
347
|
-
* a pre-summary. Unset (default): raw-trace findings whenever the run is durable
|
|
348
|
-
* (a real `runDir` — that is where the traces live), the distilled failure digest
|
|
349
|
-
* otherwise; the `memory` surface always defaults to its curation distiller.
|
|
350
|
-
* `true` forces `rawTraceDistiller()` even for an in-memory run (it emits a loud
|
|
351
|
-
* warning finding instead of paths); `false` forces the digest distiller even
|
|
352
|
-
* with a real `runDir`. Ignored when `analyzeGeneration` is set explicitly
|
|
353
|
-
* (that wins) or is `null` (disabled). */
|
|
331
|
+
/** Feed code candidates paths to prior raw traces instead of a failure digest.
|
|
332
|
+
* Defaults to true for durable runs and false for in-memory runs. */
|
|
354
333
|
rawTraceContext?: boolean;
|
|
355
|
-
/**
|
|
356
|
-
|
|
357
|
-
* (`gitWorktreeAdapter`) driven by `improvementDriver` with the full agentic
|
|
358
|
-
* generator (a real coding harness edits each candidate worktree; a `verify`
|
|
359
|
-
* hook gates candidates before they are ever measured). Ignored when
|
|
360
|
-
* `opts.generator` is supplied. Required for every code run because a real
|
|
361
|
-
* repository and base ref are necessary to measure the incumbent. */
|
|
362
|
-
code?: ImproveCodeOptions;
|
|
363
|
-
/** Select the exact inline skill document to optimize. */
|
|
364
|
-
skills?: ImproveSkillsOptions;
|
|
334
|
+
/** Isolated repository and candidate generator settings. */
|
|
335
|
+
code: ImproveCodeOptions;
|
|
365
336
|
/** Custom held-back-exam decision. The string `gate` above controls whether
|
|
366
|
-
*
|
|
337
|
+
* the exam runs; this callback controls how its evidence decides promotion. */
|
|
367
338
|
promotionGate?: SelfImproveOptions<TScenario, TArtifact>['gate'];
|
|
368
339
|
};
|
|
340
|
+
/** The canonical improvement API: complete methods for profiles, worktrees for code. */
|
|
341
|
+
type ImproveOptions<TScenario extends Scenario, TArtifact> = ImproveMethodOptions<TScenario, TArtifact> | ImproveCodeRunOptions<TScenario, TArtifact>;
|
|
369
342
|
interface ImproveSkillsOptions {
|
|
370
343
|
/** `name` of one inline entry in `profile.resources.skills`. */
|
|
371
344
|
resourceName: string;
|
|
372
345
|
}
|
|
346
|
+
/** Caller-owned mapping for optimizing several profile fields as one candidate. */
|
|
347
|
+
interface ImproveProfileComponents {
|
|
348
|
+
/** Extract the exact named text components optimized together. */
|
|
349
|
+
read(profile: ReadonlyAgentProfile): Readonly<Record<string, string>>;
|
|
350
|
+
/** Apply a complete winning component map to a detached profile. */
|
|
351
|
+
apply(profile: ReadonlyAgentProfile, components: Readonly<Record<string, string>>): ReadonlyAgentProfile;
|
|
352
|
+
}
|
|
373
353
|
interface ImproveCodeOptions {
|
|
374
354
|
/** Repo root candidate worktrees fork from. */
|
|
375
355
|
repoRoot: string;
|
|
@@ -378,57 +358,124 @@ interface ImproveCodeOptions {
|
|
|
378
358
|
/** Directory worktrees are created under. Default `<repoRoot>/.worktrees`. */
|
|
379
359
|
worktreeDir?: string;
|
|
380
360
|
/** Git-compatible adapter override, primarily for tests. Candidate advancement
|
|
381
|
-
*
|
|
361
|
+
* still requires normal Git worktree and commit semantics. */
|
|
382
362
|
worktree?: WorktreeAdapter;
|
|
383
363
|
/** Coding harness the agentic generator runs in each worktree. Default `claude`. */
|
|
384
364
|
harness?: LocalHarness;
|
|
385
365
|
/** Verify a candidate worktree before it becomes a measurable surface; failures
|
|
386
|
-
*
|
|
366
|
+
* feed the next shot (see `agenticGenerator.verify` / `commandVerifier`). */
|
|
387
367
|
verify?: Verifier;
|
|
388
368
|
/** Per-shot wall-clock timeout for the harness (ms). */
|
|
389
369
|
timeoutMs?: number;
|
|
390
|
-
/** Byte-producer override
|
|
391
|
-
*
|
|
370
|
+
/** Byte-producer override, used for tests and custom candidate production.
|
|
371
|
+
* When set, `harness`, `verify`, and `timeoutMs` are unused. */
|
|
392
372
|
generator?: CandidateGenerator;
|
|
393
373
|
}
|
|
394
|
-
interface
|
|
374
|
+
interface ImprovementProfileCandidate {
|
|
395
375
|
/** Surface searched by this run. */
|
|
396
|
-
surface:
|
|
376
|
+
surface: ImproveProfileSurface;
|
|
397
377
|
/** Exact winning value returned by agent-eval. */
|
|
398
378
|
value: MutableSurface;
|
|
399
|
-
/**
|
|
400
|
-
profile
|
|
379
|
+
/** Exact complete profile instance measured on the final cases. */
|
|
380
|
+
profile: ReadonlyAgentProfile;
|
|
381
|
+
}
|
|
382
|
+
interface ImprovementCodeCandidate {
|
|
383
|
+
surface: 'code';
|
|
384
|
+
value: MutableSurface;
|
|
385
|
+
profile?: never;
|
|
386
|
+
}
|
|
387
|
+
type ImprovementCandidate = ImprovementProfileCandidate | ImprovementCodeCandidate;
|
|
388
|
+
/** Normalized spend reported for one Runtime improvement run. */
|
|
389
|
+
interface ImproveCost {
|
|
390
|
+
totalCostUsd: number;
|
|
391
|
+
accountingComplete: boolean;
|
|
392
|
+
incompleteReasons: string[];
|
|
393
|
+
}
|
|
394
|
+
/** Optimizer ancestry sealed into downstream candidate experiments. */
|
|
395
|
+
interface ImproveLineage {
|
|
396
|
+
/** Unique Runtime invocation used to isolate this run's cost receipts. */
|
|
397
|
+
invocationId: string;
|
|
398
|
+
/** Upstream optimizer run when reported, otherwise this Runtime optimization invocation. */
|
|
399
|
+
runId: string;
|
|
400
|
+
/** Exact train-plus-selection scenario payloads exposed to candidate selection. */
|
|
401
|
+
developmentSplitDigest: Sha256Digest;
|
|
402
|
+
/** Complete callback, materializer, model, tool, and closure identity for a profile run. */
|
|
403
|
+
executionRef?: Sha256Digest;
|
|
404
|
+
/** Complete baseline profile identity for a profile run. */
|
|
405
|
+
baselineProfileDigest?: Sha256Digest;
|
|
401
406
|
}
|
|
402
|
-
interface
|
|
407
|
+
interface ImproveResultBase<TCandidate extends ImprovementCandidate> {
|
|
403
408
|
/** Frozen candidate only. Live state is changed through an approved activation. */
|
|
404
|
-
candidate:
|
|
405
|
-
/**
|
|
406
|
-
decision: SelfImproveResult<
|
|
407
|
-
/**
|
|
408
|
-
* `budget.holdout === 'deferred'` — no held-out measurement ran, so there
|
|
409
|
-
* is no lift to report (never a fabricated 0). */
|
|
409
|
+
candidate: TCandidate;
|
|
410
|
+
/** Final-test decision for this search result. */
|
|
411
|
+
decision: SelfImproveResult<Scenario, unknown>['gateDecision'];
|
|
412
|
+
/** Final-test lift when one was measured. */
|
|
410
413
|
lift?: number;
|
|
411
|
-
/**
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
414
|
+
/** Paired final-test confidence interval for method-based profile runs. */
|
|
415
|
+
liftInterval?: {
|
|
416
|
+
low: number;
|
|
417
|
+
high: number;
|
|
418
|
+
};
|
|
419
|
+
/** Full search and final-test spend. */
|
|
420
|
+
cost: ImproveCost;
|
|
421
|
+
/** Full wall-clock duration. */
|
|
422
|
+
durationMs: number;
|
|
423
|
+
/** Optimizer ancestry used when sealing a candidate experiment. */
|
|
424
|
+
lineage: ImproveLineage;
|
|
425
|
+
/** Number of generations explored by Runtime's code path. */
|
|
426
|
+
generationsExplored?: number;
|
|
415
427
|
/** Release resources owned by this result. Idempotent; currently disposes
|
|
416
|
-
*
|
|
428
|
+
* the returned code worktree and is a no-op for profile-only surfaces. */
|
|
417
429
|
dispose(): Promise<void>;
|
|
418
430
|
}
|
|
431
|
+
interface ImproveMethodResult extends ImproveResultBase<ImprovementProfileCandidate> {
|
|
432
|
+
mode: 'method';
|
|
433
|
+
method: string;
|
|
434
|
+
/** External optimizer package and resumable run identity, when reported. */
|
|
435
|
+
provenance?: OptimizationMethodComparison['best']['provenance'];
|
|
436
|
+
decision: 'ship' | 'hold';
|
|
437
|
+
lift: number;
|
|
438
|
+
liftInterval: {
|
|
439
|
+
low: number;
|
|
440
|
+
high: number;
|
|
441
|
+
};
|
|
442
|
+
raw: OptimizationMethodComparison;
|
|
443
|
+
}
|
|
444
|
+
interface ImproveCodeResult<TScenario extends Scenario, TArtifact> extends ImproveResultBase<ImprovementCodeCandidate> {
|
|
445
|
+
mode: 'code';
|
|
446
|
+
raw: SelfImproveResult<TScenario, TArtifact>;
|
|
447
|
+
}
|
|
448
|
+
type ImproveResult<TScenario extends Scenario, TArtifact> = ImproveMethodResult | ImproveCodeResult<TScenario, TArtifact>;
|
|
449
|
+
|
|
419
450
|
/**
|
|
420
|
-
* Run the held-out-gated self-improvement loop on ONE profile surface.
|
|
421
451
|
*
|
|
422
|
-
*
|
|
452
|
+
* Redaction for values that may leave the Runtime process. The default scrubs
|
|
453
|
+
* common leak classes (API keys, bearer tokens, emails, private keys) from
|
|
454
|
+
* strings and walks nested objects and arrays. A customer with domain-specific
|
|
455
|
+
* PII supplies their own `redact` hook.
|
|
423
456
|
*
|
|
424
|
-
*
|
|
425
|
-
*
|
|
426
|
-
*
|
|
427
|
-
*
|
|
428
|
-
*
|
|
429
|
-
|
|
430
|
-
|
|
457
|
+
* This is intentionally narrower than `src/sanitize.ts` (which redacts the
|
|
458
|
+
* runtime's *event envelope* field-by-field): here the value is opaque
|
|
459
|
+
* customer payload, so the scrub is value-shaped, not schema-shaped.
|
|
460
|
+
*
|
|
461
|
+
* @experimental
|
|
462
|
+
*/
|
|
463
|
+
/** A redactor maps an arbitrary trace value to a safe-to-export value. Pure;
|
|
464
|
+
* must not throw on cyclic input (the default tolerates cycles). */
|
|
465
|
+
type Redactor = (value: unknown) => unknown;
|
|
466
|
+
/**
|
|
467
|
+
* The built-in redactor. Walks objects and arrays; replaces values under
|
|
468
|
+
* secret-bearing keys wholesale; scrubs in-value patterns from every string.
|
|
469
|
+
* Cycle-safe (a seen-set short-circuits self-referential payloads to
|
|
470
|
+
* `'[circular]'`), depth-bounded, and total — never throws on customer input.
|
|
471
|
+
*/
|
|
472
|
+
declare function defaultRedactor(value: unknown): unknown;
|
|
473
|
+
/**
|
|
474
|
+
* Resolve the redactor a client uses. A caller-supplied hook handles
|
|
475
|
+
* domain-specific values first, then the built-in scrubber still removes
|
|
476
|
+
* common credentials and email addresses. Returning `false` is the explicit
|
|
477
|
+
* opt-out for already-reviewed public values.
|
|
431
478
|
*/
|
|
432
|
-
declare function
|
|
479
|
+
declare function resolveRedactor(redact: Redactor | false | undefined): Redactor;
|
|
433
480
|
|
|
434
|
-
export { AGENTIC_PROFILE_RESOURCE_ROOT as A, type CandidateGenerator as C, type ImproveOptions as I, type
|
|
481
|
+
export { AGENTIC_PROFILE_RESOURCE_ROOT as A, type ImprovementCandidate as B, type CandidateGenerator as C, type DeepReadonly as D, type ImprovementCodeCandidate as E, type ImprovementProfileCandidate as F, type VerifyResult as G, agenticGenerator as H, type ImproveOptions as I, commandVerifier as J, defaultBuildPrompt as K, type Redactor as R, type Verifier as V, type ImproveResult as a, type ImproveMethodResult as b, type ImproveMethodOptions as c, defaultRedactor as d, type ImproveCodeRunOptions as e, type ImproveCodeResult as f, type ImproveCandidateValidationInput as g, type ImproveMethodFactory as h, type ReadonlyAgentProfile as i, type AgenticGeneratorOptions as j, type AgenticGeneratorShotDisposition as k, type AgenticGeneratorShotExecution as l, type AgenticGeneratorShotReceipt as m, type ImproveCandidateValidator as n, type ImproveCodeOptions as o, type ImproveCost as p, type ImproveLineage as q, resolveRedactor as r, type ImproveMethodContext as s, type ImproveMethodSource as t, type ImproveOptimizationRunOptions as u, type ImproveProfileAgent as v, type ImproveProfileComponents as w, type ImproveProfileSurface as x, type ImproveSkillsOptions as y, type ImproveSurface as z };
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import { R as RuntimeHooks,
|
|
2
|
-
import { C as Corpus, O as Outcome } from './worktree-fanout-
|
|
3
|
-
import { A as Agent, B as Budget, S as Scope } from './types-
|
|
1
|
+
import { R as RuntimeHooks, S as SelectionReceipt } from './types-BCoemcXU.js';
|
|
2
|
+
import { C as Corpus, O as Outcome } from './worktree-fanout-DxhAWr5Z.js';
|
|
3
|
+
import { A as Agent, B as Budget, S as Scope } from './types-CvNwMzQt.js';
|
|
4
4
|
|
|
5
5
|
/**
|
|
6
6
|
* The general agentic primitive — sequential (depth) and parallel (breadth) over a shared,
|
|
@@ -266,7 +266,7 @@ declare function runAgentic(opts: RunAgenticOptions): Promise<AgenticRunResult>;
|
|
|
266
266
|
* poison repair at saturation — the glm /47,/116 regressions).
|
|
267
267
|
*
|
|
268
268
|
* Placement rule: this is an INFERENCE-TIME capability (it wraps the model call via the
|
|
269
|
-
* strategy seam). It does not belong in improve()
|
|
269
|
+
* strategy seam). It does not belong in `improve()` (training-time); `improve()`
|
|
270
270
|
* may later tune `StructuralRolloutPolicy` as an optimizable surface.
|
|
271
271
|
*/
|
|
272
272
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
import { W as WorkerProgress, S as Scope, a as Settled, b as ResultBlobStore, B as Budget, A as Agent, c as SpawnJournal, d as WaitProbeRegistry, e as SupervisedResult } from './types-
|
|
2
|
-
import { M as MakeWorkerAgent, A as AnalystRegistry, W as WorkerWatchOptions, E as ExecutorConfig } from './coordination-
|
|
1
|
+
import { W as WorkerProgress, S as Scope, a as Settled, b as ResultBlobStore, B as Budget, A as Agent, c as SpawnJournal, d as WaitProbeRegistry, e as SupervisedResult } from './types-CvNwMzQt.js';
|
|
2
|
+
import { M as MakeWorkerAgent, A as AnalystRegistry, W as WorkerWatchOptions, E as ExecutorConfig } from './coordination-BZZSVYpZ.js';
|
|
3
3
|
import { R as RouterConfig, T as ToolLoopChat, a as ToolLoopCompactionOptions } from './sanitize-DEbPNtyI.js';
|
|
4
|
-
import { D as DeliverableSpec } from './completion-gate-
|
|
4
|
+
import { D as DeliverableSpec } from './completion-gate-BMy5LGoP.js';
|
|
5
5
|
|
|
6
6
|
/**
|
|
7
7
|
*
|
package/dist/testing.js
CHANGED
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
import {
|
|
2
2
|
verifyAgentImprovementProposal
|
|
3
|
-
} from "./chunk-
|
|
4
|
-
import "./chunk-
|
|
5
|
-
import "./chunk-
|
|
3
|
+
} from "./chunk-UHEZW5BU.js";
|
|
4
|
+
import "./chunk-WSTRQZYQ.js";
|
|
5
|
+
import "./chunk-YJZA2BIK.js";
|
|
6
6
|
import "./chunk-UPWGXKXB.js";
|
|
7
|
-
import "./chunk-
|
|
7
|
+
import "./chunk-HLKC4UYB.js";
|
|
8
8
|
import "./chunk-ISPWRSEC.js";
|
|
9
9
|
import "./chunk-3MDZX7YU.js";
|
|
10
10
|
import "./chunk-FD2MBMOH.js";
|
|
@@ -14,7 +14,7 @@ import "./chunk-YEJR7IXO.js";
|
|
|
14
14
|
// src/testing/fixtures/agent-improvement-proposal.json
|
|
15
15
|
var agent_improvement_proposal_default = {
|
|
16
16
|
changedSurfaces: ["prompt"],
|
|
17
|
-
digest: "sha256:
|
|
17
|
+
digest: "sha256:1f223831184b2b53fe1be2b950a63856acb5abe887e5022a6114b5165a4f1802",
|
|
18
18
|
evaluation: {
|
|
19
19
|
decision: {
|
|
20
20
|
contributingChecks: [
|
|
@@ -2486,7 +2486,7 @@ var agent_improvement_proposal_default = {
|
|
|
2486
2486
|
],
|
|
2487
2487
|
metadata: {
|
|
2488
2488
|
fixture: "agent-improvement-proposal",
|
|
2489
|
-
runtimeVersion: "0.
|
|
2489
|
+
runtimeVersion: "0.105.0"
|
|
2490
2490
|
},
|
|
2491
2491
|
objectives: [
|
|
2492
2492
|
{
|
|
@@ -2597,8 +2597,8 @@ var agent_improvement_proposal_default = {
|
|
|
2597
2597
|
baselineContentHash: "sha256:5c21ee53e513fc604cb09754e21c392b24a424da0ef37dbf8f1ee4a8a0b08f09",
|
|
2598
2598
|
candidateContentHash: "sha256:60fcbb1c728194bd51d7d19cb732d1c3f1881dce7e0a6266b41c8b98cfd65693",
|
|
2599
2599
|
kind: "agent-eval-loop",
|
|
2600
|
-
recordDigest: "sha256:
|
|
2601
|
-
runId: "agent-runtime-0.
|
|
2600
|
+
recordDigest: "sha256:0e732c488490a383c61f092698759f8a32979f739b3e3689bf6466fb8fb10c3b",
|
|
2601
|
+
runId: "agent-runtime-0.105.0-proposal-fixture",
|
|
2602
2602
|
schema: "agent-candidate-experiment"
|
|
2603
2603
|
}
|
|
2604
2604
|
},
|
|
@@ -2624,7 +2624,7 @@ var agent_improvement_proposal_default = {
|
|
|
2624
2624
|
],
|
|
2625
2625
|
kind: "agent-improvement-proposal",
|
|
2626
2626
|
proposedAt: "2026-07-10T01:00:00.000Z",
|
|
2627
|
-
runId: "agent-runtime-0.
|
|
2627
|
+
runId: "agent-runtime-0.105.0-proposal-fixture"
|
|
2628
2628
|
};
|
|
2629
2629
|
|
|
2630
2630
|
// src/testing/index.ts
|