@tangle-network/agent-runtime 0.90.1 → 0.92.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -2
- package/dist/agent.d.ts +3 -3
- package/dist/agent.js +88 -9
- package/dist/agent.js.map +1 -1
- package/dist/{mcp-serve-verifier-XsX8rkB9.d.ts → agentic-generator-B8oeE2Yv.d.ts} +6 -33
- package/dist/analyst-loop.d.ts +1 -1
- package/dist/candidate-execution/index.d.ts +104 -0
- package/dist/candidate-execution/index.js +34 -0
- package/dist/candidate-execution/index.js.map +1 -0
- package/dist/chunk-3BE7KTMU.js +1229 -0
- package/dist/chunk-3BE7KTMU.js.map +1 -0
- package/dist/chunk-3D2RHC4K.js +73 -0
- package/dist/chunk-3D2RHC4K.js.map +1 -0
- package/dist/chunk-3MDZX7YU.js +125 -0
- package/dist/chunk-3MDZX7YU.js.map +1 -0
- package/dist/{chunk-RYBVU4M3.js → chunk-6O5USWVH.js} +32 -1413
- package/dist/chunk-6O5USWVH.js.map +1 -0
- package/dist/chunk-6O73TRHW.js +142 -0
- package/dist/chunk-6O73TRHW.js.map +1 -0
- package/dist/{chunk-R2VAJGR3.js → chunk-7VJJJ2T2.js} +2 -2
- package/dist/chunk-A62TP7SK.js +4784 -0
- package/dist/chunk-A62TP7SK.js.map +1 -0
- package/dist/chunk-APVPRF4Y.js +2166 -0
- package/dist/chunk-APVPRF4Y.js.map +1 -0
- package/dist/{chunk-QK4DV5PR.js → chunk-AUEIDTR3.js} +2 -2
- package/dist/{chunk-OOL3675H.js → chunk-FRBHUNQ7.js} +2 -139
- package/dist/chunk-FRBHUNQ7.js.map +1 -0
- package/dist/{chunk-7ON74BQO.js → chunk-GDAQUFG6.js} +2 -2
- package/dist/{chunk-ZV4LXYCJ.js → chunk-I7WVPJBZ.js} +23 -1231
- package/dist/chunk-I7WVPJBZ.js.map +1 -0
- package/dist/{chunk-WRUSWK4F.js → chunk-IGGZGKJD.js} +3 -3
- package/dist/chunk-PH65PR4F.js +860 -0
- package/dist/chunk-PH65PR4F.js.map +1 -0
- package/dist/chunk-RSWM2ZKM.js +659 -0
- package/dist/chunk-RSWM2ZKM.js.map +1 -0
- package/dist/{chunk-BZF3KQ6G.js → chunk-VSWBYWFK.js} +4 -122
- package/dist/chunk-VSWBYWFK.js.map +1 -0
- package/dist/{completion-gate-DkAnUmpb.d.ts → completion-gate-BLaiN0-X.d.ts} +1 -1
- package/dist/{coordination-rRj5hjJK.d.ts → coordination-DxJ83oZA.d.ts} +12 -5
- package/dist/environment-provider.d.ts +2 -2
- package/dist/environment-provider.js +2 -1
- package/dist/improve-CUVCq7xg.d.ts +152 -0
- package/dist/{improvement-adapter-CDR8QNVM.d.ts → improvement-adapter-BieWeK5J.d.ts} +16 -0
- package/dist/index.d.ts +27 -291
- package/dist/index.js +70 -889
- package/dist/index.js.map +1 -1
- package/dist/intelligence.d.ts +160 -13
- package/dist/intelligence.js +535 -59
- package/dist/intelligence.js.map +1 -1
- package/dist/knowledge.d.ts +6 -6
- package/dist/knowledge.js +6 -4
- package/dist/lifecycle.d.ts +2 -1
- package/dist/lifecycle.js +5 -3
- package/dist/lifecycle.js.map +1 -1
- package/dist/{loop-runner-bin-DTbZVGfM.d.ts → loop-runner-bin-kKUNGLyV.d.ts} +2 -2
- package/dist/loop-runner-bin.d.ts +5 -5
- package/dist/loop-runner-bin.js +8 -5
- package/dist/loops.d.ts +16 -16
- package/dist/loops.js +47 -41
- package/dist/mcp/bin.js +6 -4
- package/dist/mcp/bin.js.map +1 -1
- package/dist/mcp/index.d.ts +8 -8
- package/dist/mcp/index.js +9 -6
- package/dist/mcp/index.js.map +1 -1
- package/dist/mcp-serve-verifier-Bg4C3p5S.d.ts +34 -0
- package/dist/{openai-tools-C4ZfUD4L.d.ts → openai-tools-E3woykz9.d.ts} +1 -1
- package/dist/prepare-Z08a4heC.d.ts +713 -0
- package/dist/profiles.d.ts +1 -1
- package/dist/{router-client-DJImUDlm.d.ts → sanitize-C9go6tXj.d.ts} +113 -1
- package/dist/{structural-rollout-MwlpgQ-6.d.ts → structural-rollout-DHGDbhvR.d.ts} +3 -3
- package/dist/{supervise-DPmYPk0j.d.ts → supervise-T2pazU3G.d.ts} +4 -4
- package/dist/{types-SyuwunY_.d.ts → types-B00NtbCs.d.ts} +1 -1
- package/dist/{types-eMNgWgFi.d.ts → types-DAdIm4AC.d.ts} +1 -1
- package/dist/{worktree-fanout-BDFQIO-Y.d.ts → worktree-fanout-BUb2Ag02.d.ts} +3 -3
- package/package.json +26 -36
- package/skills/build-with-agent-runtime/SKILL.md +1 -1
- package/dist/chunk-BZF3KQ6G.js.map +0 -1
- package/dist/chunk-IVGYLCFH.js +0 -381
- package/dist/chunk-IVGYLCFH.js.map +0 -1
- package/dist/chunk-OOL3675H.js.map +0 -1
- package/dist/chunk-RYBVU4M3.js.map +0 -1
- package/dist/chunk-ZV4LXYCJ.js.map +0 -1
- /package/dist/{chunk-R2VAJGR3.js.map → chunk-7VJJJ2T2.js.map} +0 -0
- /package/dist/{chunk-QK4DV5PR.js.map → chunk-AUEIDTR3.js.map} +0 -0
- /package/dist/{chunk-7ON74BQO.js.map → chunk-GDAQUFG6.js.map} +0 -0
- /package/dist/{chunk-WRUSWK4F.js.map → chunk-IGGZGKJD.js.map} +0 -0
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
import { Scenario, SelfImproveOptions, SurfaceProposer, SelfImproveResult } from '@tangle-network/agent-eval/contract';
|
|
2
|
+
import { AgentProfile } from '@tangle-network/agent-interface';
|
|
3
|
+
import { L as LocalHarness } from './local-harness-dcD5WTTr.js';
|
|
4
|
+
import { V as Verifier, C as CandidateGenerator } from './agentic-generator-B8oeE2Yv.js';
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
*
|
|
8
|
+
* `improve` — the ONE public, surface-pluggable RSI verb.
|
|
9
|
+
*
|
|
10
|
+
* A thin facade over agent-eval's `selfImprove` (the held-out-gated closed
|
|
11
|
+
* loop). It removes the two things a caller otherwise has to know to drive the
|
|
12
|
+
* loop by hand: WHICH `MutableSurface` of the profile is being optimized, and
|
|
13
|
+
* WHICH `SurfaceProposer` mutates that surface. You name a `surface`; the
|
|
14
|
+
* facade picks the matching default proposer, extracts the baseline surface from
|
|
15
|
+
* the profile, runs `selfImprove`, and (on a ship verdict) writes the promoted
|
|
16
|
+
* winner back into the corresponding profile field.
|
|
17
|
+
*
|
|
18
|
+
* - `surface: 'prompt'` → `gepaProposer` mutates `profile.prompt.systemPrompt`.
|
|
19
|
+
* - `surface: 'skills'` → `skillOptProposer` mutates a skills document string.
|
|
20
|
+
* - `surface: 'agent-profile'` → caller-supplied proposer mutates the complete
|
|
21
|
+
* canonical AgentProfile JSON in one candidate.
|
|
22
|
+
* - `surface: 'rollout-policy'` → `rolloutPolicyProposer` mutates the
|
|
23
|
+
* inference-time `StructuralRolloutPolicy` dials ({ k, repairRounds, testgen })
|
|
24
|
+
* persisted in `profile.extensions['structural-rollout']` — deterministic
|
|
25
|
+
* bounded neighbor enumeration; the held-out gate does the deciding. No-op
|
|
26
|
+
* (nothing proposed, nothing shipped) when the profile has no such extension.
|
|
27
|
+
* - `surface` ∈ {`tools`, `mcp`, `hooks`, `subagents`, `workflow`, `agent-profile`, `code`} → no zero-config default
|
|
28
|
+
* proposer exists (a code/config proposer needs caller-supplied wiring — a
|
|
29
|
+
* worktree repo root, a candidate generator, a serializer). The facade
|
|
30
|
+
* requires an explicit `opts.generator` for these and throws a `ConfigError`
|
|
31
|
+
* otherwise. This is a designed boundary, not a missing default: there is
|
|
32
|
+
* no safe value the facade could invent for those surfaces. Code also
|
|
33
|
+
* requires `opts.code.repoRoot` so its incumbent is a real isolated checkout.
|
|
34
|
+
*
|
|
35
|
+
* Everything else (`scenarios`, `judge`, `agent`, `budget`, `llm`) passes
|
|
36
|
+
* straight through to `selfImprove`.
|
|
37
|
+
*
|
|
38
|
+
* @experimental
|
|
39
|
+
*/
|
|
40
|
+
|
|
41
|
+
/** The agent-profile lever `improve` optimizes. Mirrors the AgentProfile-law
|
|
42
|
+
* profile levers; `code` is the implementation-tier surface, `rollout-policy`
|
|
43
|
+
* the inference-time structuralRollout dials
|
|
44
|
+
* (`profile.extensions['structural-rollout']`). */
|
|
45
|
+
type ImproveSurface = 'prompt' | 'skills' | 'tools' | 'mcp' | 'hooks' | 'subagents' | 'workflow' | 'agent-profile' | 'code' | 'rollout-policy';
|
|
46
|
+
type ImproveOptions<TScenario extends Scenario, TArtifact> = Omit<SelfImproveOptions<TScenario, TArtifact>, 'analyzeGeneration' | 'baselineSurface' | 'findings' | 'gate' | 'proposer'> & {
|
|
47
|
+
/** Which profile lever to optimize. Default `'prompt'`. Selects the default
|
|
48
|
+
* generator + the baseline-surface extraction shape. */
|
|
49
|
+
surface?: ImproveSurface;
|
|
50
|
+
/** The `SurfaceProposer` that mutates the surface. When unset, the facade
|
|
51
|
+
* picks the default for `surface` (`gepaProposer` for prompt, `skillOptProposer`
|
|
52
|
+
* for skills); surfaces with no default REQUIRE this (fail-loud otherwise). */
|
|
53
|
+
generator?: SurfaceProposer;
|
|
54
|
+
/** Gate mode. `'holdout'` (default) runs the held-out promotion gate;
|
|
55
|
+
* `'none'` is a baseline-only run (`budget.generations = 0`). */
|
|
56
|
+
gate?: 'holdout' | 'none';
|
|
57
|
+
/** Restrict the run to this subset of models. When set, the reflection model
|
|
58
|
+
* (`llm.model`, or the default when unset) must be a member, or `improve()` throws
|
|
59
|
+
* a `ConfigError` before the generator is built. Unset = unrestricted. */
|
|
60
|
+
allowedModels?: readonly string[];
|
|
61
|
+
/** Per-generation findings producer passthrough (see selfImprove.analyzeGeneration).
|
|
62
|
+
* DEFAULT: the built-in failure distiller — after each generation it turns the
|
|
63
|
+
* worst-scoring/errored cells into structured findings ({ scenario, composite,
|
|
64
|
+
* notes, error }) for the NEXT proposal round, so the proposer reasons over what
|
|
65
|
+
* actually failed instead of a static seed. Pass your own producer (e.g. a
|
|
66
|
+
* trace-analyst over the runDir's traces) to replace it; pass `null` to disable
|
|
67
|
+
* and keep the static `findings` all the way through. */
|
|
68
|
+
analyzeGeneration?: SelfImproveOptions<TScenario, TArtifact>['analyzeGeneration'] | null;
|
|
69
|
+
/** META-HARNESS mode: instead of the ~400-char distilled findings, feed the
|
|
70
|
+
* proposer RAW-TRACE FILESYSTEM CONTEXT — the PATHS into the prior generation's
|
|
71
|
+
* real run traces under `runDir` (per-cell `spans.jsonl` event logs +
|
|
72
|
+
* `cached-result.json` scores + artifacts) plus a `grep`/`cat`-to-diagnose
|
|
73
|
+
* instruction — so the coding agent reads the actual failures itself rather than
|
|
74
|
+
* a pre-summary. Requires a REAL `runDir` (that is where the traces live).
|
|
75
|
+
* Ignored when `analyzeGeneration` is set explicitly (that wins) or is `null`
|
|
76
|
+
* (disabled). Equivalent to `analyzeGeneration: rawTraceDistiller()`; this flag
|
|
77
|
+
* is the one-line enable. Default `false` (the distiller stays the default). */
|
|
78
|
+
rawTraceContext?: boolean;
|
|
79
|
+
/** CODE-surface wiring: name `surface: 'code'`, point at a repo, and the
|
|
80
|
+
* facade assembles the whole candidate pipeline — an isolated incumbent plus git worktrees
|
|
81
|
+
* (`gitWorktreeAdapter`) driven by `improvementDriver` with the full agentic
|
|
82
|
+
* generator (a real coding harness edits each candidate worktree; a `verify`
|
|
83
|
+
* hook gates candidates before they are ever measured). Ignored when
|
|
84
|
+
* `opts.generator` is supplied. Required for every code run because a real
|
|
85
|
+
* repository and base ref are necessary to measure the incumbent. */
|
|
86
|
+
code?: ImproveCodeOptions;
|
|
87
|
+
/** SKILLS-surface wiring for real skill-DOCUMENT optimization. Without this,
|
|
88
|
+
* `surface: 'skills'` optimizes the profile's skills REFS array (file pointers)
|
|
89
|
+
* — which `skillOptProposer` (a document patcher) cannot meaningfully edit.
|
|
90
|
+
* Provide the document CONTENT to optimize + a `writeBack` to persist the
|
|
91
|
+
* shipped winner (the profile ref points at a file the caller owns). This is
|
|
92
|
+
* what makes skillOpt reachable through improve(). */
|
|
93
|
+
skills?: ImproveSkillsOptions;
|
|
94
|
+
/** Custom held-back-exam decision. The string `gate` above controls whether
|
|
95
|
+
* the exam runs; this callback controls how its evidence decides promotion. */
|
|
96
|
+
promotionGate?: SelfImproveOptions<TScenario, TArtifact>['gate'];
|
|
97
|
+
};
|
|
98
|
+
interface ImproveSkillsOptions {
|
|
99
|
+
/** The skill document's current text — the baseline `skillOptProposer` patches. */
|
|
100
|
+
document: string;
|
|
101
|
+
/** Persist the shipped winner document (write the file the profile ref points at).
|
|
102
|
+
* Called only on a ship verdict. When omitted, the winner is still returned in
|
|
103
|
+
* `result.raw.winner.surface` for the caller to materialize. */
|
|
104
|
+
writeBack?: (winnerDocument: string) => void;
|
|
105
|
+
}
|
|
106
|
+
interface ImproveCodeOptions {
|
|
107
|
+
/** Repo root candidate worktrees fork from. */
|
|
108
|
+
repoRoot: string;
|
|
109
|
+
/** Base ref candidates fork from. Default `main`. */
|
|
110
|
+
baseRef?: string;
|
|
111
|
+
/** Directory worktrees are created under. Default `<repoRoot>/.worktrees`. */
|
|
112
|
+
worktreeDir?: string;
|
|
113
|
+
/** Coding harness the agentic generator runs in each worktree. Default `claude`. */
|
|
114
|
+
harness?: LocalHarness;
|
|
115
|
+
/** Verify a candidate worktree before it becomes a measurable surface; failures
|
|
116
|
+
* feed the next shot (see `agenticGenerator.verify` / `commandVerifier`). */
|
|
117
|
+
verify?: Verifier;
|
|
118
|
+
/** Per-shot wall-clock timeout for the harness (ms). */
|
|
119
|
+
timeoutMs?: number;
|
|
120
|
+
/** Byte-producer override — the test seam and the escape hatch for custom
|
|
121
|
+
* candidate production. When set, `harness`/`verify`/`timeoutMs` are unused. */
|
|
122
|
+
generator?: CandidateGenerator;
|
|
123
|
+
}
|
|
124
|
+
interface ImproveResult<TScenario extends Scenario, TArtifact> {
|
|
125
|
+
/** The profile after improvement: the winner surface applied back into the
|
|
126
|
+
* matching field when the gate shipped, else the input profile unchanged. */
|
|
127
|
+
profile: AgentProfile;
|
|
128
|
+
/** True when `gateDecision === 'ship'`. */
|
|
129
|
+
shipped: boolean;
|
|
130
|
+
/** Held-out lift (`winner − baseline` composite). */
|
|
131
|
+
lift: number;
|
|
132
|
+
/** The five-valued gate verdict from `selfImprove`. */
|
|
133
|
+
gateDecision: SelfImproveResult<TScenario, TArtifact>['gateDecision'];
|
|
134
|
+
/** Full `selfImprove` result for advanced inspection. */
|
|
135
|
+
raw: SelfImproveResult<TScenario, TArtifact>;
|
|
136
|
+
}
|
|
137
|
+
/**
|
|
138
|
+
* Run the held-out-gated self-improvement loop on ONE profile surface.
|
|
139
|
+
*
|
|
140
|
+
* @example Optimize the system prompt, default holdout gate:
|
|
141
|
+
*
|
|
142
|
+
* const out = await improve(profile, findings, {
|
|
143
|
+
* surface: 'prompt',
|
|
144
|
+
* scenarios,
|
|
145
|
+
* judge,
|
|
146
|
+
* agent: (surface, scenario, ctx) => runAgent(surface, scenario, ctx.signal),
|
|
147
|
+
* })
|
|
148
|
+
* if (out.shipped) deploy(out.profile)
|
|
149
|
+
*/
|
|
150
|
+
declare function improve<TScenario extends Scenario, TArtifact>(profile: AgentProfile, findings: unknown[], opts: ImproveOptions<TScenario, TArtifact>): Promise<ImproveResult<TScenario, TArtifact>>;
|
|
151
|
+
|
|
152
|
+
export { type ImproveSurface as I, type ImproveOptions as a, type ImproveResult as b, type ImproveCodeOptions as c, type ImproveSkillsOptions as d, improve as i };
|
|
@@ -52,6 +52,22 @@ interface AgentSurfaces {
|
|
|
52
52
|
rag?: string;
|
|
53
53
|
/** Optional: single file defining the output schema (Zod / JSON Schema). */
|
|
54
54
|
outputSchema?: string;
|
|
55
|
+
/** Optional: directory containing Agent Skill packages. */
|
|
56
|
+
skills?: string;
|
|
57
|
+
/** Optional: directory containing MCP server/tool configuration. */
|
|
58
|
+
mcp?: string;
|
|
59
|
+
/** Optional: directory containing hook definitions. */
|
|
60
|
+
hooks?: string;
|
|
61
|
+
/** Optional: directory containing subagent definitions. */
|
|
62
|
+
subagents?: string;
|
|
63
|
+
/** Optional: directory containing orchestration/workflow policies. */
|
|
64
|
+
workflows?: string;
|
|
65
|
+
/** Optional: single file containing rollout-policy settings. */
|
|
66
|
+
rolloutPolicy?: string;
|
|
67
|
+
/** Optional: single canonical AgentProfile file. */
|
|
68
|
+
agentProfile?: string;
|
|
69
|
+
/** Optional: source root for code findings. */
|
|
70
|
+
code?: string;
|
|
55
71
|
}
|
|
56
72
|
interface ResolvedSurface {
|
|
57
73
|
/** Absolute filesystem path the operator can `cat` / `vim`. */
|
package/dist/index.d.ts
CHANGED
|
@@ -1,28 +1,33 @@
|
|
|
1
|
-
import { AgentProfile, AgentEvalError, AnalystFinding, KnowledgeReadinessReport, RunRecord, ControlEvalResult
|
|
1
|
+
import { AgentProfile, AgentEvalError, AnalystFinding, KnowledgeReadinessReport, RunRecord, ControlEvalResult } from '@tangle-network/agent-eval';
|
|
2
2
|
export { AgentEvalError, AgentEvalErrorCode, ConfigError, ControlBudget, ControlDecision, ControlEvalResult, ControlRunResult, ControlStep, DataAcquisitionPlan, JudgeError, KnowledgeReadinessReport, KnowledgeRequirement, NotFoundError, RunRecord, ValidationError } from '@tangle-network/agent-eval';
|
|
3
|
-
import {
|
|
4
|
-
export { s as AgentAdapter, t as AgentKnowledgeProvider, u as AgentRuntimeEventSink, v as AgentTaskContext, w as AgentTaskSpec, B as BackendErrorDetail, x as RuntimeDecisionEvidenceRef, y as RuntimeDecisionKind, z as RuntimeDecisionPoint, C as RuntimeHookContext, F as RuntimeHookErrorContext, G as RuntimeHookEvent, H as RuntimeHookPhase, J as RuntimeHookTarget, M as RuntimeRunHandle, N as RuntimeRunPersistenceAdapter, P as RuntimeRunRow, Q as composeRuntimeHooks, T as defineRuntimeHooks, U as notifyRuntimeDecisionPoint, W as notifyRuntimeHookEvent, X as startRuntimeRun } from './types-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
import {
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
3
|
+
import { i as AgentBackendInput, O as OpenAIChatTool, j as OpenAIChatToolChoice, k as OpenAIChatResponseFormat, l as AgentExecutionBackend, m as AgentBackendContext, b as RuntimeStreamEvent, K as KnowledgeReadinessDecision, n as RunAgentTaskOptions, o as AgentTaskRunResult, p as RunAgentTaskStreamOptions, q as RuntimeSessionStore, r as RuntimeSession, R as RuntimeHooks } from './types-B00NtbCs.js';
|
|
4
|
+
export { s as AgentAdapter, t as AgentKnowledgeProvider, A as AgentRuntimeEvent, u as AgentRuntimeEventSink, v as AgentTaskContext, w as AgentTaskSpec, c as AgentTaskStatus, B as BackendErrorDetail, x as RuntimeDecisionEvidenceRef, y as RuntimeDecisionKind, z as RuntimeDecisionPoint, C as RuntimeHookContext, F as RuntimeHookErrorContext, G as RuntimeHookEvent, H as RuntimeHookPhase, J as RuntimeHookTarget, M as RuntimeRunHandle, N as RuntimeRunPersistenceAdapter, P as RuntimeRunRow, Q as composeRuntimeHooks, T as defineRuntimeHooks, U as notifyRuntimeDecisionPoint, W as notifyRuntimeHookEvent, X as startRuntimeRun } from './types-B00NtbCs.js';
|
|
5
|
+
export { c as AgentCandidateArtifactPort, d as AgentCandidateBenchmarkGraderIdentity, e as AgentCandidateBenchmarkGraderPort, f as AgentCandidateContainerPort, g as AgentCandidateExecutionAttemptRecord, h as AgentCandidateExecutionAttemptRef, i as AgentCandidateExecutionClaim, j as AgentCandidateExecutionClaimResult, k as AgentCandidateExecutionClaimStore, l as AgentCandidateExecutionCleanupHandles, m as AgentCandidateExecutionFailureClass, n as AgentCandidateExecutionFinishResult, o as AgentCandidateExecutionLease, p as AgentCandidateExecutionPhase, q as AgentCandidateExecutionPhaseResult, a as AgentCandidateExecutionPorts, r as AgentCandidateExecutionRecoveryEvidence, s as AgentCandidateExecutionStageResult, t as AgentCandidateExecutionTerminalRecord, u as AgentCandidateExecutionTerminalResult, v as AgentCandidateExecutionUsage, w as AgentCandidateExecutorFinalCapture, x as AgentCandidateExecutorMemoryCapture, y as AgentCandidateExecutorPort, z as AgentCandidateExecutorProfileFile, B as AgentCandidateExecutorRequest, C as AgentCandidateExecutorStopRequest, D as AgentCandidateExecutorTaskOutcomeCapture, F as AgentCandidateExecutorWorkspaceFile, G as AgentCandidateExecutorWorkspaceInput, H as AgentCandidateMemoryPort, I as AgentCandidateMemoryResetResult, J as AgentCandidateModelLimits, K as AgentCandidateModelPort, L as AgentCandidateOutputArtifactPort, M as AgentCandidateOutputPurpose, N as AgentCandidateProtectedModelActivation, O as AgentCandidateProtectedModelCall, Q as AgentCandidateProtectedModelReservation, R as AgentCandidateProtectedModelSettlement, S as AgentCandidateProtectedRunCapture, T as AgentCandidateRepositoryPort, U as AgentCandidateRetryRejection, b as AgentCandidateRunFinalization, A as AgentCandidateTaskExecution, V as AgentCandidateVerificationPorts, W as AgentCandidateWorkspacePort, X as CANDIDATE_TRACE_ENV, Y as CANDIDATE_TRACE_TAGS, Z as CanonicalCandidateDocument, E as ExecutePreparedAgentCandidateOptions, _ as InMemoryAgentCandidateExecutionClaimStore, P as PrepareAgentCandidateExecutionOptions, $ as PreparedAgentCandidateExecution, a0 as PreparedAgentCandidateInstruction, a1 as PreparedAgentCandidateLaunch, a2 as PreparedAgentCandidateTrace, a3 as ResolvedAgentCandidateContainer, a4 as VerifiedAgentCandidate, a5 as VerifiedAgentCandidateTaskOutcome, a6 as executePreparedAgentCandidate, a7 as prepareAgentCandidateExecution } from './prepare-Z08a4heC.js';
|
|
6
|
+
export { AgentCandidateModelGrantActivateInput, AgentCandidateModelGrantClient, AgentCandidateModelGrantReservation, AgentCandidateModelGrantReserveInput, AgentCandidateModelGrantSettleInput, CreateProtectedAgentCandidateModelPortOptions, DisposePreparedAgentCandidateOptions, FileAgentCandidateExecutionClaimStore, FileAgentCandidateExecutionClaimStoreOptions, RecoverExpiredAgentCandidateOptions, candidateExecutionClaim, createProtectedAgentCandidateModelPort, disposePreparedAgentCandidateExecution, persistCandidateOutputArtifact, recoverExpiredAgentCandidateExecution, verifyAgentCandidateBundle } from './candidate-execution/index.js';
|
|
7
|
+
import { Scenario, ProfileDispatchFn, MutableSurface, SurfaceProposer } from '@tangle-network/agent-eval/campaign';
|
|
8
|
+
import { C as CandidateGenerator } from './agentic-generator-B8oeE2Yv.js';
|
|
9
|
+
export { A as AgenticGeneratorOptions, I as ImprovementDriverOptions, M as ManagedImprovementDriver, V as Verifier, a as VerifyResult, b as agenticGenerator, c as commandVerifier, i as improvementDriver } from './agentic-generator-B8oeE2Yv.js';
|
|
10
|
+
export { c as ImproveCodeOptions, a as ImproveOptions, b as ImproveResult, d as ImproveSkillsOptions, I as ImproveSurface, i as improve } from './improve-CUVCq7xg.js';
|
|
11
|
+
export { M as McpServeSpec, m as mcpServeVerifier } from './mcp-serve-verifier-Bg4C3p5S.js';
|
|
12
|
+
import { Scenario as Scenario$1, SelfImproveOptions } from '@tangle-network/agent-eval/contract';
|
|
13
|
+
import { S as SurfaceImprovementEdit } from './improvement-adapter-BieWeK5J.js';
|
|
12
14
|
import { I as ImprovementAdapter } from './types-BC3bZpH0.js';
|
|
13
|
-
import {
|
|
15
|
+
import { AgentProfile as AgentProfile$1 } from '@tangle-network/agent-interface';
|
|
16
|
+
import { S as StructuralRolloutPolicy } from './structural-rollout-DHGDbhvR.js';
|
|
14
17
|
export { AgentKnowledgeReadinessCheckOptions, KnowledgeImprovementJobMeasurement, KnowledgeImprovementJobResult, KnowledgeReadinessCheck, KnowledgeReadinessCheckInput, KnowledgeReadinessCheckResult, RESEARCH_SUPERVISOR_SYSTEM_PROMPT, RunKnowledgeImprovementJobOptions, SupervisedKnowledgeUpdateInput, SupervisedKnowledgeUpdateOptions, SupervisedKnowledgeUpdateResult, SupervisedKnowledgeUpdater, createAgentKnowledgeReadinessCheck, createSupervisedKnowledgeUpdater, formatSupervisedKnowledgeTask, knowledgeReadinessDeliverable, runKnowledgeImprovementJob, runSupervisedKnowledgeUpdate } from './knowledge.js';
|
|
15
|
-
export { D as DELEGATED_LOOP_MODES, a as DelegatedLoopMode, b as DelegatedLoopRegistry, c as DelegatedLoopResult, d as DelegatedLoopRunner, L as LoopRunnerCliArgs, e as LoopRunnerCliResult, R as ResearchLoopResult, f as ResearchLoopRunnerOptions, g as RunDelegatedLoopOptions, V as VetoedFact, W as WorktreeLoopRunnerOptions, h as auditLoopRunner, i as isDelegatedLoopMode, p as parseLoopRunnerArgv, r as researchLoopRunner, j as runDelegatedLoop, k as runLoopRunnerCli, s as selfImproveLoopRunner, w as worktreeLoopRunner } from './loop-runner-bin-
|
|
16
|
-
export { m as mcpToolsForRuntimeMcp, a as mcpToolsForRuntimeMcpSubset } from './openai-tools-
|
|
17
|
-
export { b0 as EvalRunEvent, b1 as EvalRunGeneration, b2 as EvalRunsExportConfig, b3 as EvalRunsExportResult, b4 as INTELLIGENCE_WIRE_VERSION, b5 as LoopSpanNode, b6 as OtelAttribute, b7 as OtelExportConfig, b8 as OtelExporter, b9 as OtelSpan, ba as
|
|
18
|
+
export { D as DELEGATED_LOOP_MODES, a as DelegatedLoopMode, b as DelegatedLoopRegistry, c as DelegatedLoopResult, d as DelegatedLoopRunner, L as LoopRunnerCliArgs, e as LoopRunnerCliResult, R as ResearchLoopResult, f as ResearchLoopRunnerOptions, g as RunDelegatedLoopOptions, V as VetoedFact, W as WorktreeLoopRunnerOptions, h as auditLoopRunner, i as isDelegatedLoopMode, p as parseLoopRunnerArgv, r as researchLoopRunner, j as runDelegatedLoop, k as runLoopRunnerCli, s as selfImproveLoopRunner, w as worktreeLoopRunner } from './loop-runner-bin-kKUNGLyV.js';
|
|
19
|
+
export { m as mcpToolsForRuntimeMcp, a as mcpToolsForRuntimeMcpSubset } from './openai-tools-E3woykz9.js';
|
|
20
|
+
export { b0 as EvalRunEvent, b1 as EvalRunGeneration, b2 as EvalRunsExportConfig, b3 as EvalRunsExportResult, b4 as INTELLIGENCE_WIRE_VERSION, b5 as LoopSpanNode, b6 as OtelAttribute, b7 as OtelExportConfig, b8 as OtelExporter, b9 as OtelSpan, ba as RuntimeEventOtelOptions, bb as buildLoopOtelSpans, bc as buildLoopSpanNodes, bd as buildRuntimeEventOtelSpans, be as createOtelExporter, bf as exportEvalRuns, bg as loopEventToOtelSpan } from './coordination-DxJ83oZA.js';
|
|
21
|
+
import { c as RuntimeTelemetryOptions } from './sanitize-C9go6tXj.js';
|
|
22
|
+
export { d as RuntimeEventCollector, e as RuntimeStreamEventCollector, S as SanitizedKnowledgeReadinessReport, f as createRuntimeEventCollector, g as createRuntimeStreamEventCollector, s as sanitizeAgentRuntimeEvent, h as sanitizeKnowledgeReadinessReport, i as sanitizeRuntimeStreamEvent } from './sanitize-C9go6tXj.js';
|
|
18
23
|
import '@tangle-network/sandbox';
|
|
24
|
+
import './local-harness-dcD5WTTr.js';
|
|
19
25
|
import 'node:child_process';
|
|
20
|
-
import './worktree-fanout-
|
|
21
|
-
import './types-
|
|
22
|
-
import './completion-gate-
|
|
26
|
+
import './worktree-fanout-BUb2Ag02.js';
|
|
27
|
+
import './types-DAdIm4AC.js';
|
|
28
|
+
import './completion-gate-BLaiN0-X.js';
|
|
23
29
|
import '@tangle-network/agent-knowledge';
|
|
24
|
-
import './supervise-
|
|
25
|
-
import './router-client-DJImUDlm.js';
|
|
30
|
+
import './supervise-T2pazU3G.js';
|
|
26
31
|
import './kb-gate-CwHO0vz6.js';
|
|
27
32
|
import './substrate-DO2GHNg2.js';
|
|
28
33
|
import './environment-provider.js';
|
|
@@ -1169,166 +1174,6 @@ declare function toolBuildPrompt(args: FindingsArg): string;
|
|
|
1169
1174
|
/** Build the starting instruction for a coder agent tasked with implementing a new MCP server. */
|
|
1170
1175
|
declare function mcpBuildPrompt(args: FindingsArg): string;
|
|
1171
1176
|
|
|
1172
|
-
/**
|
|
1173
|
-
*
|
|
1174
|
-
* `improve` — the ONE public, surface-pluggable RSI verb.
|
|
1175
|
-
*
|
|
1176
|
-
* A thin facade over agent-eval's `selfImprove` (the held-out-gated closed
|
|
1177
|
-
* loop). It removes the two things a caller otherwise has to know to drive the
|
|
1178
|
-
* loop by hand: WHICH `MutableSurface` of the profile is being optimized, and
|
|
1179
|
-
* WHICH `SurfaceProposer` mutates that surface. You name a `surface`; the
|
|
1180
|
-
* facade picks the matching default proposer, extracts the baseline surface from
|
|
1181
|
-
* the profile, runs `selfImprove`, and (on a ship verdict) writes the promoted
|
|
1182
|
-
* winner back into the corresponding profile field.
|
|
1183
|
-
*
|
|
1184
|
-
* - `surface: 'prompt'` → `gepaProposer` mutates `profile.prompt.systemPrompt`.
|
|
1185
|
-
* - `surface: 'skills'` → `skillOptProposer` mutates a skills document string.
|
|
1186
|
-
* - `surface: 'rollout-policy'` → `rolloutPolicyProposer` mutates the
|
|
1187
|
-
* inference-time `StructuralRolloutPolicy` dials ({ k, repairRounds, testgen })
|
|
1188
|
-
* persisted in `profile.extensions['structural-rollout']` — deterministic
|
|
1189
|
-
* bounded neighbor enumeration; the held-out gate does the deciding. No-op
|
|
1190
|
-
* (nothing proposed, nothing shipped) when the profile has no such extension.
|
|
1191
|
-
* - `surface` ∈ {`tools`, `mcp`, `hooks`, `code`} → no zero-config default
|
|
1192
|
-
* proposer exists (a code/config proposer needs caller-supplied wiring — a
|
|
1193
|
-
* worktree repo root, a candidate generator, a serializer). The facade
|
|
1194
|
-
* requires an explicit `opts.generator` for these and throws a `ConfigError`
|
|
1195
|
-
* otherwise. This is a designed boundary, not a missing default: there is
|
|
1196
|
-
* no safe value the facade could invent for those seams.
|
|
1197
|
-
*
|
|
1198
|
-
* Everything else (`scenarios`, `judge`, `agent`, `budget`, `llm`) passes
|
|
1199
|
-
* straight through to `selfImprove`.
|
|
1200
|
-
*
|
|
1201
|
-
* @experimental
|
|
1202
|
-
*/
|
|
1203
|
-
|
|
1204
|
-
/** The agent-profile lever `improve` optimizes. Mirrors the AgentProfile-law
|
|
1205
|
-
* profile levers; `code` is the implementation-tier surface, `rollout-policy`
|
|
1206
|
-
* the inference-time structuralRollout dials
|
|
1207
|
-
* (`profile.extensions['structural-rollout']`). */
|
|
1208
|
-
type ImproveSurface = 'prompt' | 'skills' | 'tools' | 'mcp' | 'hooks' | 'code' | 'rollout-policy';
|
|
1209
|
-
interface ImproveOptions<TScenario extends Scenario$1, TArtifact> {
|
|
1210
|
-
/** Which profile lever to optimize. Default `'prompt'`. Selects the default
|
|
1211
|
-
* generator + the baseline-surface extraction shape. */
|
|
1212
|
-
surface?: ImproveSurface;
|
|
1213
|
-
/** The `SurfaceProposer` that mutates the surface. When unset, the facade
|
|
1214
|
-
* picks the default for `surface` (`gepaProposer` for prompt, `skillOptProposer`
|
|
1215
|
-
* for skills); surfaces with no default REQUIRE this (fail-loud otherwise). */
|
|
1216
|
-
generator?: SurfaceProposer;
|
|
1217
|
-
/** Gate mode. `'holdout'` (default) runs the held-out promotion gate;
|
|
1218
|
-
* `'none'` is a baseline-only run (`budget.generations = 0`). */
|
|
1219
|
-
gate?: 'holdout' | 'none';
|
|
1220
|
-
/** Scenarios to evaluate against. Passthrough to `selfImprove`. */
|
|
1221
|
-
scenarios: TScenario[];
|
|
1222
|
-
/** Judge that scores artifacts. Passthrough to `selfImprove`. */
|
|
1223
|
-
judge: JudgeConfig<TArtifact, TScenario>;
|
|
1224
|
-
/** The agent under improvement — same shape as `selfImprove.agent`: it takes
|
|
1225
|
-
* the current surface + scenario + ctx and returns the artifact to judge. */
|
|
1226
|
-
agent: (surface: MutableSurface, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
|
|
1227
|
-
/** Budget + loop shape. Passthrough; `gate: 'none'` forces `generations = 0`. */
|
|
1228
|
-
budget?: SelfImproveBudget;
|
|
1229
|
-
/** LLM config. Passthrough to `selfImprove` AND used to construct the default
|
|
1230
|
-
* reflective proposer (`gepaProposer`/`skillOptProposer`) when `generator` is unset. */
|
|
1231
|
-
llm?: SelfImproveLlm;
|
|
1232
|
-
/** Restrict the run to this subset of models. When set, the reflection model
|
|
1233
|
-
* (`llm.model`, or the default when unset) must be a member, or `improve()` throws
|
|
1234
|
-
* a `ConfigError` before the generator is built. Unset = unrestricted. */
|
|
1235
|
-
allowedModels?: readonly string[];
|
|
1236
|
-
/** Run directory passthrough to `selfImprove`. Pass a REAL path to make the loop
|
|
1237
|
-
* durable: campaign cells + the loop provenance record land on the filesystem as
|
|
1238
|
-
* they complete, so a multi-hour search survives a process/infra death instead of
|
|
1239
|
-
* losing every generation with it (the default `mem://` run keeps everything
|
|
1240
|
-
* in-process). */
|
|
1241
|
-
runDir?: string;
|
|
1242
|
-
/** Per-generation findings producer passthrough (see selfImprove.analyzeGeneration).
|
|
1243
|
-
* DEFAULT: the built-in failure distiller — after each generation it turns the
|
|
1244
|
-
* worst-scoring/errored cells into structured findings ({ scenario, composite,
|
|
1245
|
-
* notes, error }) for the NEXT proposal round, so the proposer reasons over what
|
|
1246
|
-
* actually failed instead of a static seed. Pass your own producer (e.g. a
|
|
1247
|
-
* trace-analyst over the runDir's traces) to replace it; pass `null` to disable
|
|
1248
|
-
* and keep the static `findings` all the way through. */
|
|
1249
|
-
analyzeGeneration?: SelfImproveOptions<TScenario, TArtifact>['analyzeGeneration'] | null;
|
|
1250
|
-
/** META-HARNESS mode: instead of the ~400-char distilled findings, feed the
|
|
1251
|
-
* proposer RAW-TRACE FILESYSTEM CONTEXT — the PATHS into the prior generation's
|
|
1252
|
-
* real run traces under `runDir` (per-cell `spans.jsonl` event logs +
|
|
1253
|
-
* `cached-result.json` scores + artifacts) plus a `grep`/`cat`-to-diagnose
|
|
1254
|
-
* instruction — so the coding agent reads the actual failures itself rather than
|
|
1255
|
-
* a pre-summary. Requires a REAL `runDir` (that is where the traces live).
|
|
1256
|
-
* Ignored when `analyzeGeneration` is set explicitly (that wins) or is `null`
|
|
1257
|
-
* (disabled). Equivalent to `analyzeGeneration: rawTraceDistiller()`; this flag
|
|
1258
|
-
* is the one-line enable. Default `false` (the distiller stays the default). */
|
|
1259
|
-
rawTraceContext?: boolean;
|
|
1260
|
-
/** CODE-surface wiring with prompt-parity DX: name `surface: 'code'`, point at a
|
|
1261
|
-
* repo, and the facade assembles the whole candidate pipeline — git worktrees
|
|
1262
|
-
* (`gitWorktreeAdapter`) driven by `improvementDriver` with the full agentic
|
|
1263
|
-
* generator (a real coding harness edits each candidate worktree; a `verify`
|
|
1264
|
-
* hook gates candidates before they are ever measured). Ignored when
|
|
1265
|
-
* `opts.generator` is supplied. Without either, `surface: 'code'` still fails
|
|
1266
|
-
* loud — there is no safe zero-config repo to invent. */
|
|
1267
|
-
code?: ImproveCodeOptions;
|
|
1268
|
-
/** SKILLS-surface wiring for real skill-DOCUMENT optimization. Without this,
|
|
1269
|
-
* `surface: 'skills'` optimizes the profile's skills REFS array (file pointers)
|
|
1270
|
-
* — which `skillOptProposer` (a document patcher) cannot meaningfully edit.
|
|
1271
|
-
* Provide the document CONTENT to optimize + a `writeBack` to persist the
|
|
1272
|
-
* shipped winner (the profile ref points at a file the caller owns). This is
|
|
1273
|
-
* what makes skillOpt reachable through improve(). */
|
|
1274
|
-
skills?: ImproveSkillsOptions;
|
|
1275
|
-
/** Storage passthrough to `selfImprove`; overrides the default chosen from `runDir`. */
|
|
1276
|
-
storage?: SelfImproveOptions<TScenario, TArtifact>['storage'];
|
|
1277
|
-
}
|
|
1278
|
-
interface ImproveSkillsOptions {
|
|
1279
|
-
/** The skill document's current text — the baseline `skillOptProposer` patches. */
|
|
1280
|
-
document: string;
|
|
1281
|
-
/** Persist the shipped winner document (write the file the profile ref points at).
|
|
1282
|
-
* Called only on a ship verdict. When omitted, the winner is still returned in
|
|
1283
|
-
* `result.raw.winner.surface` for the caller to materialize. */
|
|
1284
|
-
writeBack?: (winnerDocument: string) => void;
|
|
1285
|
-
}
|
|
1286
|
-
interface ImproveCodeOptions {
|
|
1287
|
-
/** Repo root candidate worktrees fork from. */
|
|
1288
|
-
repoRoot: string;
|
|
1289
|
-
/** Base ref candidates fork from. Default `main`. */
|
|
1290
|
-
baseRef?: string;
|
|
1291
|
-
/** Directory worktrees are created under. Default `<repoRoot>/.worktrees`. */
|
|
1292
|
-
worktreeDir?: string;
|
|
1293
|
-
/** Coding harness the agentic generator runs in each worktree. Default `claude`. */
|
|
1294
|
-
harness?: LocalHarness;
|
|
1295
|
-
/** Verify a candidate worktree before it becomes a measurable surface; failures
|
|
1296
|
-
* feed the next shot (see `agenticGenerator.verify` / `commandVerifier`). */
|
|
1297
|
-
verify?: Verifier;
|
|
1298
|
-
/** Per-shot wall-clock timeout for the harness (ms). */
|
|
1299
|
-
timeoutMs?: number;
|
|
1300
|
-
/** Byte-producer override — the test seam and the escape hatch for custom
|
|
1301
|
-
* candidate production. When set, `harness`/`verify`/`timeoutMs` are unused. */
|
|
1302
|
-
generator?: CandidateGenerator;
|
|
1303
|
-
}
|
|
1304
|
-
interface ImproveResult<TScenario extends Scenario$1, TArtifact> {
|
|
1305
|
-
/** The profile after improvement: the winner surface applied back into the
|
|
1306
|
-
* matching field when the gate shipped, else the input profile unchanged. */
|
|
1307
|
-
profile: AgentProfile$1;
|
|
1308
|
-
/** True when `gateDecision === 'ship'`. */
|
|
1309
|
-
shipped: boolean;
|
|
1310
|
-
/** Held-out lift (`winner − baseline` composite). */
|
|
1311
|
-
lift: number;
|
|
1312
|
-
/** The five-valued gate verdict from `selfImprove`. */
|
|
1313
|
-
gateDecision: SelfImproveResult<TScenario, TArtifact>['gateDecision'];
|
|
1314
|
-
/** Full `selfImprove` result for advanced inspection. */
|
|
1315
|
-
raw: SelfImproveResult<TScenario, TArtifact>;
|
|
1316
|
-
}
|
|
1317
|
-
/**
|
|
1318
|
-
* Run the held-out-gated self-improvement loop on ONE profile surface.
|
|
1319
|
-
*
|
|
1320
|
-
* @example Optimize the system prompt, default holdout gate:
|
|
1321
|
-
*
|
|
1322
|
-
* const out = await improve(profile, findings, {
|
|
1323
|
-
* surface: 'prompt',
|
|
1324
|
-
* scenarios,
|
|
1325
|
-
* judge,
|
|
1326
|
-
* agent: (surface, scenario, ctx) => runAgent(surface, scenario, ctx.signal),
|
|
1327
|
-
* })
|
|
1328
|
-
* if (out.shipped) deploy(out.profile)
|
|
1329
|
-
*/
|
|
1330
|
-
declare function improve<TScenario extends Scenario$1, TArtifact>(profile: AgentProfile$1, findings: unknown[], opts: ImproveOptions<TScenario, TArtifact>): Promise<ImproveResult<TScenario, TArtifact>>;
|
|
1331
|
-
|
|
1332
1177
|
/**
|
|
1333
1178
|
*
|
|
1334
1179
|
* `rawTraceDistiller` — the meta-harness `analyzeGeneration` producer.
|
|
@@ -1473,7 +1318,7 @@ declare const ROLLOUT_POLICY_BOUNDS: {
|
|
|
1473
1318
|
* violates the policy's own invariants: the no-op signal. Unknown dials are
|
|
1474
1319
|
* dropped; `diverse`/`temperature` ride through untouched (the proposer never
|
|
1475
1320
|
* mutates them — `diverse` is a measured paired null). */
|
|
1476
|
-
declare function parseRolloutPolicy(surface: MutableSurface
|
|
1321
|
+
declare function parseRolloutPolicy(surface: MutableSurface): StructuralRolloutPolicy | undefined;
|
|
1477
1322
|
/** Normalize an untyped policy bag (a parsed surface or a profile extension) into
|
|
1478
1323
|
* a full `StructuralRolloutPolicy`, defaults merged. Returns `undefined` when any
|
|
1479
1324
|
* present dial violates the policy invariants (mirrors `resolvePolicy`: integer
|
|
@@ -1505,7 +1350,7 @@ declare function enumerateNeighborPolicies(policy: StructuralRolloutPolicy): Str
|
|
|
1505
1350
|
* carries no policy (the profile never opted in) — an empty proposal is the
|
|
1506
1351
|
* loop-native no-op, mirroring `improvementDriver`'s no-findings behavior.
|
|
1507
1352
|
*/
|
|
1508
|
-
declare function rolloutPolicyProposer(): SurfaceProposer
|
|
1353
|
+
declare function rolloutPolicyProposer(): SurfaceProposer;
|
|
1509
1354
|
|
|
1510
1355
|
/**
|
|
1511
1356
|
*
|
|
@@ -1726,115 +1571,6 @@ declare function runAgentTask<TState, TAction, TActionResult, TEval extends Cont
|
|
|
1726
1571
|
*/
|
|
1727
1572
|
declare function runAgentTaskStream<TInput extends AgentBackendInput = AgentBackendInput>(options: RunAgentTaskStreamOptions<TInput>): AsyncIterable<RuntimeStreamEvent>;
|
|
1728
1573
|
|
|
1729
|
-
/**
|
|
1730
|
-
*
|
|
1731
|
-
* Sanitization for runtime telemetry. The rule: nothing user-controlled leaks
|
|
1732
|
-
* unless the caller opts in with a `RuntimeTelemetryOptions` flag. This is the
|
|
1733
|
-
* envelope that ends up in `agent_run.metadata.runtimeEvents` on every
|
|
1734
|
-
* consumer, so the default must be safe.
|
|
1735
|
-
*
|
|
1736
|
-
* @stable
|
|
1737
|
-
*/
|
|
1738
|
-
|
|
1739
|
-
/** @stable */
|
|
1740
|
-
interface RuntimeTelemetryOptions {
|
|
1741
|
-
/**
|
|
1742
|
-
* Include raw task inputs. Off by default because task inputs often contain
|
|
1743
|
-
* customer facts, credentials, source text, or internal IDs.
|
|
1744
|
-
*/
|
|
1745
|
-
includeInputs?: boolean;
|
|
1746
|
-
/** Include requirement descriptions. Secret requirements are always redacted. */
|
|
1747
|
-
includeRequirementDescriptions?: boolean;
|
|
1748
|
-
/** Include evidence IDs. Off by default; counts are safer for shared reports. */
|
|
1749
|
-
includeEvidenceIds?: boolean;
|
|
1750
|
-
/** Include user answers from question preflight. Off by default. */
|
|
1751
|
-
includeUserAnswers?: boolean;
|
|
1752
|
-
/** Include action payloads and action results for control steps. Off by default. */
|
|
1753
|
-
includeControlPayloads?: boolean;
|
|
1754
|
-
/** Include task metadata. Off by default because metadata may carry IDs or policy internals. */
|
|
1755
|
-
includeMetadata?: boolean;
|
|
1756
|
-
/** Include eval detail/evidence strings. Off by default because validators may echo private input. */
|
|
1757
|
-
includeEvalDetails?: boolean;
|
|
1758
|
-
}
|
|
1759
|
-
/** @stable */
|
|
1760
|
-
interface SanitizedKnowledgeRequirement {
|
|
1761
|
-
id: string;
|
|
1762
|
-
description?: string;
|
|
1763
|
-
requiredFor: string[];
|
|
1764
|
-
category: KnowledgeRequirement['category'];
|
|
1765
|
-
acquisitionMode: KnowledgeRequirement['acquisitionMode'];
|
|
1766
|
-
importance: KnowledgeRequirement['importance'];
|
|
1767
|
-
freshness: KnowledgeRequirement['freshness'];
|
|
1768
|
-
sensitivity: KnowledgeRequirement['sensitivity'];
|
|
1769
|
-
confidenceNeeded: number;
|
|
1770
|
-
currentConfidence: number;
|
|
1771
|
-
evidenceCount: number;
|
|
1772
|
-
evidenceIds?: string[];
|
|
1773
|
-
fallbackPolicy: KnowledgeRequirement['fallbackPolicy'];
|
|
1774
|
-
}
|
|
1775
|
-
/** @stable */
|
|
1776
|
-
interface SanitizedKnowledgeReadinessReport {
|
|
1777
|
-
taskId: string;
|
|
1778
|
-
readinessScore: number;
|
|
1779
|
-
recommendedAction: KnowledgeReadinessReport['recommendedAction'];
|
|
1780
|
-
severity: KnowledgeReadinessReport['severity'];
|
|
1781
|
-
reason: string;
|
|
1782
|
-
blockingMissingRequirements: SanitizedKnowledgeRequirement[];
|
|
1783
|
-
nonBlockingGaps: SanitizedKnowledgeRequirement[];
|
|
1784
|
-
evidenceCount: number;
|
|
1785
|
-
evidenceIds?: string[];
|
|
1786
|
-
missingRequirementIds: string[];
|
|
1787
|
-
}
|
|
1788
|
-
/** Strip PII and large blobs from a `KnowledgeReadinessReport` for safe telemetry emission. @stable */
|
|
1789
|
-
declare function sanitizeKnowledgeReadinessReport(report: KnowledgeReadinessReport, options?: RuntimeTelemetryOptions): SanitizedKnowledgeReadinessReport;
|
|
1790
|
-
/** Reduce an `AgentRuntimeEvent` to a PII-safe, serializable plain object for telemetry. @stable */
|
|
1791
|
-
declare function sanitizeAgentRuntimeEvent<TState, TAction, TActionResult, TEval extends ControlEvalResult>(event: AgentRuntimeEvent<TState, TAction, TActionResult, TEval>, options?: RuntimeTelemetryOptions): Record<string, unknown>;
|
|
1792
|
-
/** Reduce a `RuntimeStreamEvent` to a PII-safe, serializable plain object for telemetry. @stable */
|
|
1793
|
-
declare function sanitizeRuntimeStreamEvent(event: RuntimeStreamEvent, options?: RuntimeTelemetryOptions): Record<string, unknown>;
|
|
1794
|
-
/** @stable */
|
|
1795
|
-
interface RuntimeEventCollector<TState = unknown, TAction = unknown, TActionResult = unknown, TEval extends ControlEvalResult = ControlEvalResult> {
|
|
1796
|
-
onEvent: (event: AgentRuntimeEvent<TState, TAction, TActionResult, TEval>) => void;
|
|
1797
|
-
events: Array<Record<string, unknown>>;
|
|
1798
|
-
}
|
|
1799
|
-
/** @stable */
|
|
1800
|
-
type RuntimeStreamEventSink = (event: RuntimeStreamEvent) => void;
|
|
1801
|
-
/** @stable */
|
|
1802
|
-
interface RuntimeStreamEventSummary {
|
|
1803
|
-
/** Total count of sanitized events collected. */
|
|
1804
|
-
eventCount: number;
|
|
1805
|
-
/** Count of events per `type`. Useful for log-line summaries. */
|
|
1806
|
-
eventCountsByType: Record<string, number>;
|
|
1807
|
-
/** First session id observed in a `session_created` / `session_resumed` event, if any. */
|
|
1808
|
-
firstSessionId?: string;
|
|
1809
|
-
/** Last `final` event's status, if a final event was observed. */
|
|
1810
|
-
finalStatus?: AgentTaskStatus;
|
|
1811
|
-
/** Last `final` event's reason, if a final event was observed. */
|
|
1812
|
-
finalReason?: string;
|
|
1813
|
-
/** Concatenated `text_delta.text` across the stream, even when payloads are redacted. */
|
|
1814
|
-
finalText: string;
|
|
1815
|
-
}
|
|
1816
|
-
/** @stable */
|
|
1817
|
-
interface RuntimeStreamEventCollector {
|
|
1818
|
-
onEvent: RuntimeStreamEventSink;
|
|
1819
|
-
events: Array<Record<string, unknown>>;
|
|
1820
|
-
/** Snapshot of a small streaming-flavored summary derived from collected events. */
|
|
1821
|
-
summary(): RuntimeStreamEventSummary;
|
|
1822
|
-
}
|
|
1823
|
-
/** Build an in-memory collector that sanitizes and accumulates `AgentRuntimeEvent`s for inspection. @stable */
|
|
1824
|
-
declare function createRuntimeEventCollector<TState = unknown, TAction = unknown, TActionResult = unknown, TEval extends ControlEvalResult = ControlEvalResult>(options?: RuntimeTelemetryOptions): RuntimeEventCollector<TState, TAction, TActionResult, TEval>;
|
|
1825
|
-
/**
|
|
1826
|
-
*
|
|
1827
|
-
* Streaming-event counterpart of `createRuntimeEventCollector`. Pass each
|
|
1828
|
-
* event yielded by `runAgentTaskStream` through `onEvent` and read the
|
|
1829
|
-
* sanitized copies off `events`; the same `RuntimeTelemetryOptions` redaction
|
|
1830
|
-
* flags apply. Kept distinct from `createRuntimeEventCollector` because the
|
|
1831
|
-
* stream and non-stream event shapes overlap on `type` literals — dispatching
|
|
1832
|
-
* on `type` alone would misroute events.
|
|
1833
|
-
*
|
|
1834
|
-
* @stable
|
|
1835
|
-
*/
|
|
1836
|
-
declare function createRuntimeStreamEventCollector(options?: RuntimeTelemetryOptions): RuntimeStreamEventCollector;
|
|
1837
|
-
|
|
1838
1574
|
/**
|
|
1839
1575
|
*
|
|
1840
1576
|
* Session helpers + an in-memory `RuntimeSessionStore` implementation suitable
|
|
@@ -2041,4 +1777,4 @@ interface StreamToolLoopOptions<Raw> {
|
|
|
2041
1777
|
* `capped` if it stops for any non-completed reason with calls still pending. */
|
|
2042
1778
|
declare function streamToolLoop<Raw>(opts: StreamToolLoopOptions<Raw>): AsyncGenerator<StreamToolLoopYield<Raw>, void, unknown>;
|
|
2043
1779
|
|
|
2044
|
-
export { AgentBackendContext, AgentBackendInput, type AgentBackendKind, AgentExecutionBackend,
|
|
1780
|
+
export { AgentBackendContext, AgentBackendInput, type AgentBackendKind, AgentExecutionBackend, AgentTaskRunResult, type AuthSource, type BackendCallPolicy, BackendTransportError, CandidateGenerator, type ChatStreamEvent, type ChatTurnHooks, type ChatTurnIdentity, type ChatTurnProducer, type ChatTurnResult, type CircuitBreakerConfig, CircuitBreakerState, CircuitOpenError, type Conversation, type ConversationDriveState, type ConversationJournal, type ConversationJournalEntry, type ConversationParticipant, type ConversationPolicy, type ConversationResult, type ConversationStreamEvent, type ConversationTurn, type D1DatabaseLike, type D1StmtLike, DEFAULT_MAX_DEPTH, DEFAULT_ROUTER_BASE_URL, DeadlineExceededError, FORWARD_HEADERS, FileConversationJournal, type ForwardHeaderName, type HaltContext, type HaltPredicate, type HaltReason, type HaltSignal, InMemoryConversationJournal, InMemoryRuntimeSessionStore, type ModelInfo, OpenAIChatResponseFormat, OpenAIChatTool, OpenAIChatToolChoice, type PersonaConversationResult, type PersonaDriver, PlannerError, type PropagatedHeaders, ROLLOUT_POLICY_BOUNDS, ROLLOUT_POLICY_EXTENSION, type RawTraceDistillerOptions, type ReflectiveGeneratorOptions, type ResolveAgentBackendOptions, type ResolvedChatModel, type RetryBackoff, type RetryableErrorPredicate, type RouterEnv, type RunChatTurnInput, type RunConversationOptions, type RunPersonaConfig, type RunPersonaConversationOptions, type RunToolLoopOptions, RuntimeHooks, RuntimeRunStateError, RuntimeSessionStore, RuntimeStreamEvent, RuntimeTelemetryOptions, type SqlAdapter, SqlConversationJournal, type StreamToolLoopOptions, type StreamToolLoopYield, type ToolCallOutcome, type ToolLoopAssistantToolCall, type ToolLoopCall, type ToolLoopEvent, type ToolLoopMessage, type ToolLoopResult, type ToolLoopStopReason, type TurnOrder, applyRolloutPolicyToProfile, applyRunRecordDefaults, buildForwardHeaders, cleanModelId, computeBackoff, createConversationBackend, createIterableBackend, createOpenAICompatibleBackend, createSandboxPromptBackend, d1ToSqlAdapter, decideKnowledgeReadiness, defaultIsRetryable, defineConversation, deriveExecutionId, enumerateNeighborPolicies, getModels, handleChatTurn, isDepthExceeded, makePerAttemptSignal, mcpBuildPrompt, normalizeRolloutPolicy, parseRolloutPolicy, rawTraceDistiller, readDepth, readinessServerSentEvent, reflectiveGenerator, resolveAgentBackend, resolveChatModel, resolveRouterBaseUrl, rolloutPolicyProposer, runAgentTask, runAgentTaskStream, runConversation, runConversationStream, runPersonaConversation, runPersonaDispatch, runToolLoop, runtimeStreamServerSentEvent, serializeRolloutPolicy, sleep, slugifySpeaker, streamToolLoop, structuralRolloutPolicyFromProfile, toolBuildPrompt, turnId, validateChatModelId };
|