@tangle-network/agent-runtime 0.95.0 → 0.96.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +64 -15
- package/dist/activation-B0ZD7nfX.d.ts +63 -0
- package/dist/agent.d.ts +5 -169
- package/dist/agent.js +8 -229
- package/dist/agent.js.map +1 -1
- package/dist/analyst-loop.d.ts +6 -9
- package/dist/analyst-loop.js +1 -2
- package/dist/candidate-execution/index.js +6 -7
- package/dist/chunk-3XKSBI2U.js +474 -0
- package/dist/chunk-3XKSBI2U.js.map +1 -0
- package/dist/{chunk-6YBA64Z2.js → chunk-6XKPVJAZ.js} +5 -18
- package/dist/chunk-6XKPVJAZ.js.map +1 -0
- package/dist/{chunk-MKGRLDWB.js → chunk-BLQIYRVR.js} +17 -2
- package/dist/chunk-BLQIYRVR.js.map +1 -0
- package/dist/{chunk-QDSOD7RC.js → chunk-FD2MBMOH.js} +13 -101
- package/dist/chunk-FD2MBMOH.js.map +1 -0
- package/dist/{chunk-YLUOTX6U.js → chunk-FXF2OL34.js} +7 -7
- package/dist/{chunk-EP6RVHMX.js → chunk-HZDEXTSL.js} +848 -2
- package/dist/chunk-HZDEXTSL.js.map +1 -0
- package/dist/chunk-PSOCBNM3.js +2069 -0
- package/dist/chunk-PSOCBNM3.js.map +1 -0
- package/dist/{chunk-BPGXIKK7.js → chunk-SGQ4YIQW.js} +4 -4
- package/dist/{chunk-IADLKE7I.js → chunk-UQ6PNNXM.js} +5 -7
- package/dist/{chunk-IADLKE7I.js.map → chunk-UQ6PNNXM.js.map} +1 -1
- package/dist/{chunk-Z5I642SY.js → chunk-WYC2XJF2.js} +2 -2
- package/dist/{chunk-ZEYAT33L.js → chunk-Y3SRWZMP.js} +2 -2
- package/dist/{chunk-WTZ37EQY.js → chunk-YOLKCWRV.js} +197 -90
- package/dist/chunk-YOLKCWRV.js.map +1 -0
- package/dist/conversation.js +0 -1
- package/dist/environment-provider.js +0 -1
- package/dist/{agentic-generator-hCaQRAes.d.ts → improve-g75IE2Cx.d.ts} +152 -3
- package/dist/{improvement-adapter-BieWeK5J.d.ts → improvement-adapter-HAZz-7vK.d.ts} +8 -31
- package/dist/index.d.ts +41 -11
- package/dist/index.js +180 -51
- package/dist/index.js.map +1 -1
- package/dist/intelligence.d.ts +16 -9
- package/dist/intelligence.js +13 -8
- package/dist/intelligence.js.map +1 -1
- package/dist/knowledge.d.ts +30 -13
- package/dist/knowledge.js +11 -10
- package/dist/{loop-runner-bin-BIQldFS8.d.ts → loop-runner-bin-Cn1N2rRo.d.ts} +1 -1
- package/dist/loop-runner-bin.d.ts +2 -2
- package/dist/loop-runner-bin.js +6 -8
- package/dist/loops.d.ts +1 -1
- package/dist/loops.js +4 -6
- package/dist/mcp/bin.js +3 -5
- package/dist/mcp/bin.js.map +1 -1
- package/dist/mcp/index.js +10 -12
- package/dist/mcp/index.js.map +1 -1
- package/dist/platform.js +0 -2
- package/dist/platform.js.map +1 -1
- package/dist/primeintellect/index.js +0 -1
- package/dist/primeintellect/index.js.map +1 -1
- package/dist/profiles.js +0 -1
- package/dist/profiles.js.map +1 -1
- package/dist/{types-BC3bZpH0.d.ts → types-CmYCMbFT.d.ts} +12 -54
- package/package.json +7 -12
- package/skills/build-with-agent-runtime/SKILL.md +122 -213
- package/dist/chunk-6O73TRHW.js +0 -142
- package/dist/chunk-6O73TRHW.js.map +0 -1
- package/dist/chunk-6YBA64Z2.js.map +0 -1
- package/dist/chunk-AP7CPGMZ.js +0 -334
- package/dist/chunk-AP7CPGMZ.js.map +0 -1
- package/dist/chunk-DGUM43GV.js +0 -11
- package/dist/chunk-DGUM43GV.js.map +0 -1
- package/dist/chunk-DHCHL6OG.js +0 -625
- package/dist/chunk-DHCHL6OG.js.map +0 -1
- package/dist/chunk-EP6RVHMX.js.map +0 -1
- package/dist/chunk-G55QE4IQ.js +0 -1137
- package/dist/chunk-G55QE4IQ.js.map +0 -1
- package/dist/chunk-ISTDY47H.js +0 -849
- package/dist/chunk-ISTDY47H.js.map +0 -1
- package/dist/chunk-MKGRLDWB.js.map +0 -1
- package/dist/chunk-QDSOD7RC.js.map +0 -1
- package/dist/chunk-WTZ37EQY.js.map +0 -1
- package/dist/generator-YkAQrOoD.d.ts +0 -382
- package/dist/improve-B-UYaEH5.d.ts +0 -172
- package/dist/lifecycle.d.ts +0 -870
- package/dist/lifecycle.js +0 -981
- package/dist/lifecycle.js.map +0 -1
- package/dist/mcp-serve-verifier-Bs_n0xPc.d.ts +0 -34
- package/skills/agent-runtime-adoption/SKILL.md +0 -246
- /package/dist/{chunk-YLUOTX6U.js.map → chunk-FXF2OL34.js.map} +0 -0
- /package/dist/{chunk-BPGXIKK7.js.map → chunk-SGQ4YIQW.js.map} +0 -0
- /package/dist/{chunk-Z5I642SY.js.map → chunk-WYC2XJF2.js.map} +0 -0
- /package/dist/{chunk-ZEYAT33L.js.map → chunk-Y3SRWZMP.js.map} +0 -0
package/README.md
CHANGED
|
@@ -33,7 +33,7 @@ pnpm tsx examples/driver-loop/driver-loop.ts
|
|
|
33
33
|
| Run a **chat turn** for a production product agent | `handleChatTurn(...)` |
|
|
34
34
|
| Have one agent **supervise a team of agents** toward a goal | `supervise(profile, task, opts)` |
|
|
35
35
|
| **Improve** an agent and prove the gain on fresh tasks | `improve(profile, findings, opts)` |
|
|
36
|
-
|
|
|
36
|
+
| Produce a measured knowledge-base candidate with agents and checks | `runKnowledgeImprovementJob(...)` |
|
|
37
37
|
| Evaluate or train the same agent on **PrimeIntellect** | `createPrimeIntellectPackage(...)` |
|
|
38
38
|
|
|
39
39
|
### Run a chat turn
|
|
@@ -70,7 +70,7 @@ const result = await supervise(
|
|
|
70
70
|
|
|
71
71
|
### Improve an agent
|
|
72
72
|
|
|
73
|
-
`improve` optimizes one part of an agent and
|
|
73
|
+
`improve` optimizes one part of an agent and returns a detached winner plus a decision. The decision is `ship` only when the candidate beats the current agent on tasks it never practiced on.
|
|
74
74
|
It accepts prompt, skill document, curated memory, tool, MCP, hook, subagent, whole-profile, and code surfaces through one call.
|
|
75
75
|
Prompt, skill-document, and memory optimization have built-in generators; structured profile surfaces take an explicit generator, and code runs from isolated incumbent and candidate checkouts.
|
|
76
76
|
Workflow and rollout-policy files use the code surface so the measured winner is an exact patch that can be sealed and executed; JSON parameter sweeps use agent-eval's `parameterSweepProposer` instead of a runtime-specific optimizer.
|
|
@@ -78,26 +78,74 @@ Workflow and rollout-policy files use the code surface so the measured winner is
|
|
|
78
78
|
```ts
|
|
79
79
|
import { improve } from '@tangle-network/agent-runtime'
|
|
80
80
|
|
|
81
|
-
const {
|
|
82
|
-
surface: 'prompt',
|
|
83
|
-
gate: 'holdout',
|
|
84
|
-
scenarios,
|
|
81
|
+
const { candidate, decision, lift } = await improve(baseProfile, findings, {
|
|
82
|
+
surface: 'prompt',
|
|
83
|
+
gate: 'holdout',
|
|
84
|
+
scenarios,
|
|
85
|
+
judge,
|
|
86
|
+
agent,
|
|
85
87
|
})
|
|
88
|
+
|
|
89
|
+
if (decision === 'ship') console.log({ candidate, lift })
|
|
86
90
|
```
|
|
87
91
|
|
|
88
|
-
|
|
92
|
+
Skill and curated-memory candidates are exact profile changes, not free-floating text.
|
|
93
|
+
Name one inline skill through `skills.resourceName`; curated memory uses `profile.resources.instructions`.
|
|
94
|
+
Both require `profile.resources.failOnError: true` so an unsupported resource cannot silently disappear.
|
|
89
95
|
|
|
90
96
|
```ts
|
|
91
|
-
await improve(baseProfile, findings, {
|
|
92
|
-
surface: '
|
|
93
|
-
|
|
97
|
+
const skillResult = await improve(baseProfile, findings, {
|
|
98
|
+
surface: 'skills',
|
|
99
|
+
skills: { resourceName: 'incident-response' },
|
|
94
100
|
scenarios, judge, agent,
|
|
95
101
|
})
|
|
96
102
|
```
|
|
97
103
|
|
|
104
|
+
`improve` is the search call.
|
|
105
|
+
For production, `proposeAgentImprovement` adds trace analysis and reruns the exact frozen baseline and winner before creating a reviewable proposal.
|
|
106
|
+
Runtime rejects a candidate bundle that differs from the search winner.
|
|
107
|
+
|
|
108
|
+
```ts
|
|
109
|
+
import {
|
|
110
|
+
createAgentImprovementActivation,
|
|
111
|
+
executeAgentImprovementActivation,
|
|
112
|
+
proposeAgentImprovement,
|
|
113
|
+
reviewAgentImprovementProposal,
|
|
114
|
+
} from '@tangle-network/agent-runtime/intelligence'
|
|
115
|
+
|
|
116
|
+
const result = await proposeAgentImprovement({
|
|
117
|
+
runId,
|
|
118
|
+
profile: liveProfile,
|
|
119
|
+
analysis,
|
|
120
|
+
improvement: { surface: 'prompt', scenarios, judge, agent },
|
|
121
|
+
buildExperiment: ({ improvement }) => freezeExperiment(liveProfile, improvement.candidate),
|
|
122
|
+
placeCell,
|
|
123
|
+
})
|
|
124
|
+
|
|
125
|
+
const review = reviewAgentImprovementProposal(result.proposal, {
|
|
126
|
+
decision: 'approve',
|
|
127
|
+
reviewedBy: user.id,
|
|
128
|
+
reason: 'The measured gain is worth the cost.',
|
|
129
|
+
})
|
|
130
|
+
const activation = createAgentImprovementActivation(result.proposal, review, {
|
|
131
|
+
intent: 'activate-candidate',
|
|
132
|
+
targets: [{ surface: 'prompt', identity: profileId }],
|
|
133
|
+
fundingOwner: tenantId,
|
|
134
|
+
authorizedBy: user.id,
|
|
135
|
+
expiresAt,
|
|
136
|
+
})
|
|
137
|
+
const outcome = await executeAgentImprovementActivation(
|
|
138
|
+
{ proposal: result.proposal, review, activation },
|
|
139
|
+
{ transition: commitProfileTransaction, reconcile: readCommittedResult },
|
|
140
|
+
)
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
`freezeExperiment`, `placeCell`, and the transaction functions are application ports because storage and compute differ by product.
|
|
144
|
+
Runtime owns candidate identity, measurement, review binding, expiry, retry identity, and result validation; the application owns its atomic write.
|
|
145
|
+
|
|
98
146
|
### Improve a knowledge base
|
|
99
147
|
|
|
100
|
-
`runKnowledgeImprovementJob` is the runtime-owned front door for KB, wiki, memory-backed, and RAG improvement jobs. It creates a candidate copy, runs supervised agents against it, checks readiness through `@tangle-network/agent-knowledge`,
|
|
148
|
+
`runKnowledgeImprovementJob` is the runtime-owned front door for KB, wiki, memory-backed, and RAG improvement jobs. It creates a candidate copy, runs supervised agents against it, checks readiness through `@tangle-network/agent-knowledge`, and returns frozen baseline and candidate snapshots with spend and timing. It never changes the live knowledge base. Use `improve(..., { surface: 'memory' })` for the agent's curated lesson document; use this job for source, retrieval, and knowledge-store changes.
|
|
101
149
|
|
|
102
150
|
```ts
|
|
103
151
|
import { runKnowledgeImprovementJob } from '@tangle-network/agent-runtime/knowledge'
|
|
@@ -110,10 +158,11 @@ const result = await runKnowledgeImprovementJob({
|
|
|
110
158
|
backend,
|
|
111
159
|
})
|
|
112
160
|
|
|
113
|
-
console.log(result.
|
|
161
|
+
console.log(result.knowledge?.reference.candidateHash, result.measurement.supervisedSpent)
|
|
114
162
|
```
|
|
115
163
|
|
|
116
|
-
Use it when the product needs one knob for "make this knowledge base better" instead of wiring `improveKnowledgeBase`, a runtime supervisor, candidate workspaces, readiness checks
|
|
164
|
+
Use it when the product needs one knob for "make this knowledge base better" instead of wiring `improveKnowledgeBase`, a runtime supervisor, candidate workspaces, and readiness checks by hand.
|
|
165
|
+
Measure the returned bundle pair, record the review, then activate through `executeAgentImprovementActivation`; activation is the only write path.
|
|
117
166
|
|
|
118
167
|
### Run on PrimeIntellect
|
|
119
168
|
|
|
@@ -200,14 +249,14 @@ Runnable, grouped by what they show. Copy the one nearest your task:
|
|
|
200
249
|
| Evaluate or train a runtime program on PrimeIntellect | `@tangle-network/agent-runtime/primeintellect` |
|
|
201
250
|
| Study coordination vs raw compute | [`ablation-suite`](./examples/ablation-suite) |
|
|
202
251
|
|
|
203
|
-
All
|
|
252
|
+
All 28 live in [`examples/`](./examples).
|
|
204
253
|
|
|
205
254
|
## Where to go next
|
|
206
255
|
|
|
207
256
|
- New here? [`docs/concepts.md`](./docs/concepts.md), the mental model in plain terms.
|
|
208
257
|
- [`docs/canonical-api.md`](./docs/canonical-api.md), find the primitive: "I want to ___ → use ___".
|
|
209
258
|
- [`docs/api/primitive-catalog.md`](./docs/api/primitive-catalog.md), every export in one generated, never-stale list with its import path. Check it before building anything new.
|
|
210
|
-
- Import subpaths: the root export is the product surface (`handleChatTurn`, `improve`); deeper capabilities ship as subpaths: `/loops` (multi-agent + the loop kernel), `/conversation` (multi-turn conversations), `/knowledge` (KB improvement), `/primeintellect` (Prime task, runtime, and trace adapter), `/mcp` (tool servers), `/intelligence` (observability drop-in), `/
|
|
259
|
+
- Import subpaths: the root export is the product surface (`handleChatTurn`, `improve`); deeper capabilities ship as subpaths: `/loops` (multi-agent + the loop kernel), `/conversation` (multi-turn conversations), `/knowledge` (KB improvement), `/primeintellect` (Prime task, runtime, and trace adapter), `/mcp` (tool servers), `/intelligence` (observability drop-in), `/agent`, `/profiles`, `/platform`, `/analyst-loop`, `/environment-provider`.
|
|
211
260
|
- [`docs/architecture.md`](./docs/architecture.md), the design, end to end.
|
|
212
261
|
- [`bench/HARNESS.md`](./bench/HARNESS.md), the experiment harness and how to run a benchmark.
|
|
213
262
|
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
import { Sha256Digest, AgentImprovementActivationResult, AgentImprovementActivation, AgentCandidateBundle, AgentImprovementActivationTarget, AgentImprovementActivationOutcome, AgentImprovementProposal, AgentImprovementReview } from '@tangle-network/agent-interface';
|
|
2
|
+
|
|
3
|
+
interface CreateAgentImprovementActivationResultOptions {
|
|
4
|
+
completedAt: string;
|
|
5
|
+
outcome: AgentImprovementActivationOutcome;
|
|
6
|
+
}
|
|
7
|
+
interface AgentImprovementActivationTargetPlan extends AgentImprovementActivationTarget {
|
|
8
|
+
desiredDigest: Sha256Digest;
|
|
9
|
+
}
|
|
10
|
+
interface AgentImprovementActivationTransitionInput {
|
|
11
|
+
activation: AgentImprovementActivation;
|
|
12
|
+
candidateBundle: AgentCandidateBundle;
|
|
13
|
+
bundle: AgentCandidateBundle;
|
|
14
|
+
targets: [AgentImprovementActivationTargetPlan, ...AgentImprovementActivationTargetPlan[]];
|
|
15
|
+
attemptedAt: string;
|
|
16
|
+
expired: boolean;
|
|
17
|
+
}
|
|
18
|
+
interface AgentImprovementActivationResultStore {
|
|
19
|
+
load(idempotencyKey: Sha256Digest): Promise<unknown | undefined>;
|
|
20
|
+
putIfAbsent(result: AgentImprovementActivationResult): Promise<unknown>;
|
|
21
|
+
}
|
|
22
|
+
/**
|
|
23
|
+
* Product-owned or Runtime-composed transition.
|
|
24
|
+
*
|
|
25
|
+
* Implementations resolve a stored result for `activation.digest`, compare
|
|
26
|
+
* every target, and make the write durably idempotent. Co-located targets store
|
|
27
|
+
* the all-or-none write with its result. Other targets throw when result
|
|
28
|
+
* storage fails so a retry can reconcile it. Runtime never invokes this write
|
|
29
|
+
* function after authorization expires.
|
|
30
|
+
*/
|
|
31
|
+
type AgentImprovementActivationTransition = (input: AgentImprovementActivationTransitionInput) => Promise<unknown>;
|
|
32
|
+
/**
|
|
33
|
+
* Target-read-only check for a prior exact write.
|
|
34
|
+
* It may persist recovered result metadata, but must not change an activation target.
|
|
35
|
+
* Return undefined only when no target write can have committed.
|
|
36
|
+
*/
|
|
37
|
+
type AgentImprovementActivationReconciliation = (input: AgentImprovementActivationTransitionInput) => Promise<unknown | undefined>;
|
|
38
|
+
interface ExecuteAgentImprovementActivationInput {
|
|
39
|
+
proposal: AgentImprovementProposal;
|
|
40
|
+
review: AgentImprovementReview;
|
|
41
|
+
activation: AgentImprovementActivation;
|
|
42
|
+
}
|
|
43
|
+
interface ExecuteAgentImprovementActivationOptions {
|
|
44
|
+
transition: AgentImprovementActivationTransition;
|
|
45
|
+
reconcile?: AgentImprovementActivationReconciliation;
|
|
46
|
+
now?: () => Date;
|
|
47
|
+
}
|
|
48
|
+
/** Create the exact result a product stores in the same transaction as its target write. */
|
|
49
|
+
declare function createAgentImprovementActivationResult(transition: AgentImprovementActivationTransitionInput, options: CreateAgentImprovementActivationResultOptions): AgentImprovementActivationResult;
|
|
50
|
+
/**
|
|
51
|
+
* Recompute one historical activation result against the exact measured proposal and authority.
|
|
52
|
+
* The result records that attempt; it is not a query of the target's current state.
|
|
53
|
+
*/
|
|
54
|
+
declare function verifyAgentImprovementActivationResult(input: {
|
|
55
|
+
proposal: unknown;
|
|
56
|
+
review: unknown;
|
|
57
|
+
activation: unknown;
|
|
58
|
+
result: unknown;
|
|
59
|
+
}): AgentImprovementActivationResult;
|
|
60
|
+
/** Validate and execute one product-owned activation transition. */
|
|
61
|
+
declare function executeAgentImprovementActivation(input: ExecuteAgentImprovementActivationInput, options: ExecuteAgentImprovementActivationOptions): Promise<AgentImprovementActivationResult>;
|
|
62
|
+
|
|
63
|
+
export { type AgentImprovementActivationReconciliation as A, type CreateAgentImprovementActivationResultOptions as C, type ExecuteAgentImprovementActivationInput as E, type AgentImprovementActivationResultStore as a, type AgentImprovementActivationTargetPlan as b, type AgentImprovementActivationTransition as c, type AgentImprovementActivationTransitionInput as d, type ExecuteAgentImprovementActivationOptions as e, createAgentImprovementActivationResult as f, executeAgentImprovementActivation as g, verifyAgentImprovementActivationResult as v };
|
package/dist/agent.d.ts
CHANGED
|
@@ -1,13 +1,12 @@
|
|
|
1
1
|
import * as _tangle_network_agent_eval from '@tangle-network/agent-eval';
|
|
2
|
-
import { TraceAnalystKindSpec
|
|
3
|
-
import { A as ArtifactKind, C as CandidateGenerator, P as PromotionGate } from './generator-YkAQrOoD.js';
|
|
2
|
+
import { TraceAnalystKindSpec } from '@tangle-network/agent-eval';
|
|
4
3
|
import { R as RuntimeStreamEvent } from './types-BwoZWq-i.js';
|
|
5
|
-
import { A as AgentSurfaces } from './improvement-adapter-
|
|
6
|
-
export { C as
|
|
7
|
-
import { K as KnowledgeAdapter, a as RunAnalystLoopResult } from './types-BC3bZpH0.js';
|
|
4
|
+
import { A as AgentSurfaces } from './improvement-adapter-HAZz-7vK.js';
|
|
5
|
+
export { C as CreateSurfaceImprovementProposerOptions, D as DraftPatchInput, a as DraftPatchOutput, R as ResolvedSurface, S as SurfaceImprovementEdit, b as SurfaceValidationIssue, c as createSurfaceImprovementProposer, r as renderSurfaceIssues, d as resolveSubjectPath, v as validateSurfaces } from './improvement-adapter-HAZz-7vK.js';
|
|
8
6
|
import { AgentProfile, AgentProfileFileMount, AgentProfileMcpServer } from '@tangle-network/agent-interface';
|
|
9
7
|
import { SandboxEvent } from '@tangle-network/sandbox';
|
|
10
8
|
import { S as SandboxClient, O as OutputAdapter, A as AgentRunSpec } from './types-B3vAW0Oq.js';
|
|
9
|
+
import './types-CmYCMbFT.js';
|
|
11
10
|
|
|
12
11
|
/**
|
|
13
12
|
* The full agent manifest. Each agent ships ONE of these.
|
|
@@ -82,37 +81,6 @@ interface AgentManifest<TPersona = unknown, TRunOutput = unknown> {
|
|
|
82
81
|
* kinds (override per-kind via `analystKinds` if needed).
|
|
83
82
|
*/
|
|
84
83
|
analyst: AnalystConfig;
|
|
85
|
-
/**
|
|
86
|
-
* Declarative per-surface artifact-lifecycle config the closed loop reads.
|
|
87
|
-
*
|
|
88
|
-
* Each entry names a profile surface (`skill` / `tool` / `prompt` / `mcp` /
|
|
89
|
-
* `hook` / `subagent`) and supplies the `CandidateGenerator` that grows it +
|
|
90
|
-
* the `PromotionGate` (the held-back exam) that decides promotion. `runLifecycle`
|
|
91
|
-
* consumes these: it pools the generators, measures each candidate's marginal
|
|
92
|
-
* lift on the held-back split, gates it, and stores the winners in an
|
|
93
|
-
* `ArtifactRegistry` — then `composeProfile` folds the top-`k` active artifacts
|
|
94
|
-
* back into this agent's profile.
|
|
95
|
-
*
|
|
96
|
-
* Optional — agents that don't self-improve their profile omit it. An empty or
|
|
97
|
-
* absent map means "no lifecycle"; the manifest is otherwise unchanged.
|
|
98
|
-
*/
|
|
99
|
-
lifecycles?: ReadonlyArray<SurfaceLifecycle>;
|
|
100
|
-
}
|
|
101
|
-
/**
|
|
102
|
-
* One profile surface's artifact-lifecycle wiring — the declarative config a
|
|
103
|
-
* `defineAgent` manifest carries and `runLifecycle` reads. It is config, not
|
|
104
|
-
* execution: it names the generator + gate; the loop runs them.
|
|
105
|
-
*/
|
|
106
|
-
interface SurfaceLifecycle {
|
|
107
|
-
/** The profile surface this lifecycle grows. */
|
|
108
|
-
surface: ArtifactKind;
|
|
109
|
-
/** Produces fresh candidate artifacts for `surface` from the agent's history. */
|
|
110
|
-
generator: CandidateGenerator;
|
|
111
|
-
/** The held-back exam that decides promotion of a measured candidate. */
|
|
112
|
-
gate: PromotionGate;
|
|
113
|
-
/** Top-`k` budget for `composeProfile` when folding this surface's promoted
|
|
114
|
-
* artifacts back in. Omit to fold in every active artifact. */
|
|
115
|
-
composeK?: number;
|
|
116
84
|
}
|
|
117
85
|
interface AgentRubric<TRunOutput> {
|
|
118
86
|
/** Dimensions composing the weighted score. Weights sum to 1.0 by convention. */
|
|
@@ -261,138 +229,6 @@ declare class AgentManifestError extends Error {
|
|
|
261
229
|
*/
|
|
262
230
|
declare function defineAgent<TPersona = unknown, TRunOutput = unknown>(manifest: AgentManifest<TPersona, TRunOutput>): AgentManifest<TPersona, TRunOutput>;
|
|
263
231
|
|
|
264
|
-
/**
|
|
265
|
-
* Substrate-default `KnowledgeAdapter` — wraps agent-knowledge's
|
|
266
|
-
* `proposeFromFindings` + `applyKnowledgeWriteBlocks` with substrate
|
|
267
|
-
* defaults (auto-lint after apply, source linkage via finding id).
|
|
268
|
-
*
|
|
269
|
-
* Every agent that ships a `.agent-knowledge/` tree uses this adapter
|
|
270
|
-
* unmodified. Per-agent customization happens at the manifest level
|
|
271
|
-
* (`autoApply.knowledge.confidenceThreshold`, etc.), not by writing a
|
|
272
|
-
* new adapter.
|
|
273
|
-
*
|
|
274
|
-
* Lint discipline: after each apply we run agent-knowledge's
|
|
275
|
-
* `lintKnowledgeIndex` to catch broken links / circular claims /
|
|
276
|
-
* duplicate pages introduced by the new writes. Findings that fail the
|
|
277
|
-
* post-apply lint are recorded in `warnings`; the apply itself is not
|
|
278
|
-
* rolled back (lint failures are soft — humans review the wiki state).
|
|
279
|
-
*/
|
|
280
|
-
|
|
281
|
-
interface CreateSurfaceKnowledgeAdapterOpts {
|
|
282
|
-
/** `.agent-knowledge/` root (absolute path the substrate writes blocks against). */
|
|
283
|
-
knowledgeRoot: string;
|
|
284
|
-
}
|
|
285
|
-
/**
|
|
286
|
-
* Build the adapter. We accept the agent-knowledge functions as DI so
|
|
287
|
-
* the substrate stays decoupled from a specific agent-knowledge
|
|
288
|
-
* version — the agent author imports them in their manifest module
|
|
289
|
-
* and hands them to the factory.
|
|
290
|
-
*
|
|
291
|
-
* `proposeFromFindings(findings)` returns
|
|
292
|
-
* `{ proposals: KnowledgeProposal[]; skipped: number; errors: ... }`.
|
|
293
|
-
*
|
|
294
|
-
* `applyKnowledgeWriteBlocks(root, content)` returns
|
|
295
|
-
* `{ written: string[]; warnings: string[] }`.
|
|
296
|
-
*
|
|
297
|
-
* `lintKnowledgeIndex(index)` (optional) returns `KnowledgeLintFinding[]`.
|
|
298
|
-
*/
|
|
299
|
-
interface KnowledgeAdapterDeps<TProposal> {
|
|
300
|
-
proposeFromFindings: (findings: ReadonlyArray<AnalystFinding>) => {
|
|
301
|
-
proposals: TProposal[];
|
|
302
|
-
skipped: number;
|
|
303
|
-
errors: Array<{
|
|
304
|
-
findingId: string;
|
|
305
|
-
subject: string;
|
|
306
|
-
message: string;
|
|
307
|
-
}>;
|
|
308
|
-
};
|
|
309
|
-
applyKnowledgeWriteBlocks: (root: string, proposalText: string) => Promise<{
|
|
310
|
-
written: string[];
|
|
311
|
-
warnings: string[];
|
|
312
|
-
}>;
|
|
313
|
-
/**
|
|
314
|
-
* Optional post-apply lint hook. The substrate runs it after each
|
|
315
|
-
* batch of writes; failures land in `warnings` (the apply is not
|
|
316
|
-
* rolled back — lint signals drift to review, not block).
|
|
317
|
-
*/
|
|
318
|
-
lintAfterApply?: (root: string) => Promise<ReadonlyArray<string>>;
|
|
319
|
-
}
|
|
320
|
-
/** Wire a surface-based `KnowledgeAdapter` that writes analyst proposals to agent surface files. */
|
|
321
|
-
declare function createSurfaceKnowledgeAdapter<TProposal>(opts: CreateSurfaceKnowledgeAdapterOpts, deps: KnowledgeAdapterDeps<TProposal>): KnowledgeAdapter<TProposal>;
|
|
322
|
-
|
|
323
|
-
/**
|
|
324
|
-
* `OutcomeMeasurement` — the missing metric that turns the analyst
|
|
325
|
-
* loop from "observability" into "self-improvement".
|
|
326
|
-
*
|
|
327
|
-
* Without this hook, the loop reports process counts (`findings: 42`,
|
|
328
|
-
* `applied: 7`) and never proves the applied edits actually improved
|
|
329
|
-
* anything. With this hook, the substrate re-runs the cohort against
|
|
330
|
-
* the same personas after each apply pass and reports a composite
|
|
331
|
-
* score delta. A negative delta is the substrate's strongest signal
|
|
332
|
-
* to either roll back or surface for review.
|
|
333
|
-
*
|
|
334
|
-
* Wiring is intentionally simple: pass the manifest + the `runAgentEval`
|
|
335
|
-
* function and a list of `personaIds` to re-run. The wrapper:
|
|
336
|
-
* 1. Captures the baseline composite from the just-finished run.
|
|
337
|
-
* 2. After `runAnalystLoop` returns, re-invokes `runAgentEval` against
|
|
338
|
-
* the same persona slice.
|
|
339
|
-
* 3. Computes the delta and appends to `loop-report.json`.
|
|
340
|
-
* 4. If `rollbackOnRegression` and delta < 0, reverts applied edits.
|
|
341
|
-
*/
|
|
342
|
-
|
|
343
|
-
interface OutcomeMeasurement {
|
|
344
|
-
/** Baseline composite before applies — captured from the most-recent eval run. */
|
|
345
|
-
baselineComposite: number;
|
|
346
|
-
/** Composite after re-running the cohort with applied edits. */
|
|
347
|
-
afterComposite: number;
|
|
348
|
-
/** `afterComposite - baselineComposite`. Positive = the loop improved the agent. */
|
|
349
|
-
delta: number;
|
|
350
|
-
/** Per-persona deltas for finer-grained review. */
|
|
351
|
-
perPersona: ReadonlyArray<{
|
|
352
|
-
personaId: string;
|
|
353
|
-
before: number;
|
|
354
|
-
after: number;
|
|
355
|
-
delta: number;
|
|
356
|
-
}>;
|
|
357
|
-
/** When the substrate rolled back applies due to regression, the paths reverted. */
|
|
358
|
-
rolledBackPaths: ReadonlyArray<string>;
|
|
359
|
-
}
|
|
360
|
-
interface OutcomeMeasurementOpts {
|
|
361
|
-
/** Composite scores from the run that produced the findings. */
|
|
362
|
-
baseline: ReadonlyArray<{
|
|
363
|
-
personaId: string;
|
|
364
|
-
composite: number;
|
|
365
|
-
}>;
|
|
366
|
-
/**
|
|
367
|
-
* Re-run callback — the substrate invokes this after applies. The
|
|
368
|
-
* agent author provides their `runAgentEval`-equivalent so the
|
|
369
|
-
* substrate can ask "score this persona slice now."
|
|
370
|
-
*
|
|
371
|
-
* The callback SHOULD reuse the same cohort + judges + variant as
|
|
372
|
-
* the baseline run; only the agent's mutable surfaces have changed.
|
|
373
|
-
*/
|
|
374
|
-
reRunCohort: (personaIds: ReadonlyArray<string>) => Promise<ReadonlyArray<{
|
|
375
|
-
personaId: string;
|
|
376
|
-
composite: number;
|
|
377
|
-
}>>;
|
|
378
|
-
/** When `true`, applied edits are reverted on negative delta. Default `false`. */
|
|
379
|
-
rollbackOnRegression?: boolean;
|
|
380
|
-
/** Callback to revert a list of paths (typically `git checkout HEAD --`). */
|
|
381
|
-
revert?: (paths: ReadonlyArray<string>) => Promise<void>;
|
|
382
|
-
}
|
|
383
|
-
/**
|
|
384
|
-
* Run `runAnalystLoop` and stamp an `OutcomeMeasurement` onto the
|
|
385
|
-
* result. The substrate calls this after each canonical eval; the
|
|
386
|
-
* delta lands in `loop-report.json` for cross-run trend analysis.
|
|
387
|
-
*
|
|
388
|
-
* The function returns the original `RunAnalystLoopResult` enriched
|
|
389
|
-
* with `outcome` so callers stay backwards-compatible (the field is
|
|
390
|
-
* optional on the type; missing means no measurement was wired).
|
|
391
|
-
*/
|
|
392
|
-
declare function measureOutcome<TProposal, TEdit>(result: RunAnalystLoopResult<TProposal, TEdit>, opts: OutcomeMeasurementOpts): Promise<RunAnalystLoopResult<TProposal, TEdit> & {
|
|
393
|
-
outcome: OutcomeMeasurement;
|
|
394
|
-
}>;
|
|
395
|
-
|
|
396
232
|
/** Known AgentProfile axes a run path may or may not carry into execution. */
|
|
397
233
|
declare const AGENT_PROFILE_MATERIALIZATION_AXES: readonly ["identity", "name", "model", "prompt", "systemPrompt", "instructions", "resources", "files", "resourceInstructions", "skills", "resourceTools", "resourceAgents", "commands", "tools", "permissions", "mcp", "mcpConnections", "connections", "subagents", "hooks", "modes", "confidential", "metadata", "extensions"];
|
|
398
234
|
type KnownAgentProfileMaterializationAxis = (typeof AGENT_PROFILE_MATERIALIZATION_AXES)[number];
|
|
@@ -508,4 +344,4 @@ interface CreateSandboxActOptions<TPersona, TRunOutput> {
|
|
|
508
344
|
*/
|
|
509
345
|
declare function createSandboxAct<TPersona, TRunOutput>(options: CreateSandboxActOptions<TPersona, TRunOutput>): (persona: TPersona, ctx: AgentRunContext) => AgentRunInvocation<TRunOutput>;
|
|
510
346
|
|
|
511
|
-
export { AGENT_PROFILE_MATERIALIZATION_AXES, type AgentManifest, AgentManifestError, type AgentProfileMaterializationAxis, type AgentRubric, type AgentRunContext, type AgentRunInvocation, type AgentRuntime, AgentSurfaces, type AnalystConfig, type AssertProfileMaterializationOptions, type CreateSandboxActOptions, type
|
|
347
|
+
export { AGENT_PROFILE_MATERIALIZATION_AXES, type AgentManifest, AgentManifestError, type AgentProfileMaterializationAxis, type AgentRubric, type AgentRunContext, type AgentRunInvocation, type AgentRuntime, AgentSurfaces, type AnalystConfig, type AssertProfileMaterializationOptions, type CreateSandboxActOptions, type DefineProfileMaterializationContractOptions, type JudgeConfig, type KnownAgentProfileMaterializationAxis, type ProfileMaterializationContract, type ProfileMaterializationIssue, type RubricDimension, type ValidateProfileMaterializationOptions, assertProfileMaterialization, collectAgentRun, createSandboxAct, defineAgent, defineProfileMaterializationContract, promptOnlyProfileMaterialization, promptResourceProfileMaterialization, renderProfileMaterializationIssues, sandboxActProfileMaterialization, unimplementedAgentRun, validateProfileMaterialization };
|