@tangle-network/agent-runtime 0.94.13 → 0.96.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (110) hide show
  1. package/README.md +64 -15
  2. package/dist/activation-B0ZD7nfX.d.ts +63 -0
  3. package/dist/agent.d.ts +6 -193
  4. package/dist/agent.js +10 -234
  5. package/dist/agent.js.map +1 -1
  6. package/dist/analyst-loop.d.ts +7 -10
  7. package/dist/analyst-loop.js +1 -2
  8. package/dist/candidate-execution/index.d.ts +43 -16
  9. package/dist/candidate-execution/index.js +17 -8
  10. package/dist/{chunk-VYA2YEKA.js → chunk-2KGAN2HM.js} +83 -14
  11. package/dist/chunk-2KGAN2HM.js.map +1 -0
  12. package/dist/chunk-3XKSBI2U.js +474 -0
  13. package/dist/chunk-3XKSBI2U.js.map +1 -0
  14. package/dist/{chunk-TVJQAYQM.js → chunk-6XKPVJAZ.js} +107 -716
  15. package/dist/chunk-6XKPVJAZ.js.map +1 -0
  16. package/dist/chunk-BLQIYRVR.js +699 -0
  17. package/dist/chunk-BLQIYRVR.js.map +1 -0
  18. package/dist/{chunk-QDSOD7RC.js → chunk-FD2MBMOH.js} +13 -101
  19. package/dist/chunk-FD2MBMOH.js.map +1 -0
  20. package/dist/{chunk-U33YZ7B2.js → chunk-FXF2OL34.js} +8 -8
  21. package/dist/{chunk-EP6RVHMX.js → chunk-HZDEXTSL.js} +848 -2
  22. package/dist/chunk-HZDEXTSL.js.map +1 -0
  23. package/dist/{chunk-WIPGQ4GT.js → chunk-IMSNJSXH.js} +1 -1
  24. package/dist/{chunk-WIPGQ4GT.js.map → chunk-IMSNJSXH.js.map} +1 -1
  25. package/dist/{chunk-VSWBYWFK.js → chunk-M6MD6JBS.js} +20 -27
  26. package/dist/chunk-M6MD6JBS.js.map +1 -0
  27. package/dist/chunk-PSOCBNM3.js +2069 -0
  28. package/dist/chunk-PSOCBNM3.js.map +1 -0
  29. package/dist/{chunk-AEG3NGJ2.js → chunk-Q2JSAVQ3.js} +34 -2
  30. package/dist/chunk-Q2JSAVQ3.js.map +1 -0
  31. package/dist/{chunk-XP5KDM3R.js → chunk-SGQ4YIQW.js} +4 -4
  32. package/dist/{chunk-C3UKLQ54.js → chunk-UQ6PNNXM.js} +18 -10
  33. package/dist/chunk-UQ6PNNXM.js.map +1 -0
  34. package/dist/{chunk-33OG2NN3.js → chunk-WYC2XJF2.js} +2 -2
  35. package/dist/{chunk-ZEYAT33L.js → chunk-Y3SRWZMP.js} +2 -2
  36. package/dist/{chunk-CNH7DF7Z.js → chunk-YOLKCWRV.js} +1116 -591
  37. package/dist/chunk-YOLKCWRV.js.map +1 -0
  38. package/dist/{completion-gate-D1gX1-hg.d.ts → completion-gate-C80jiRfN.d.ts} +1 -1
  39. package/dist/conversation.d.ts +12 -1
  40. package/dist/conversation.js +2 -3
  41. package/dist/{coordination-Dr_axlAf.d.ts → coordination-BFE3Den7.d.ts} +10 -11
  42. package/dist/environment-provider.d.ts +2 -2
  43. package/dist/environment-provider.js +1 -2
  44. package/dist/{agentic-generator-DDMM45kZ.d.ts → improve-g75IE2Cx.d.ts} +152 -4
  45. package/dist/{improvement-adapter-BieWeK5J.d.ts → improvement-adapter-HAZz-7vK.d.ts} +8 -31
  46. package/dist/index.d.ts +55 -28
  47. package/dist/index.js +206 -82
  48. package/dist/index.js.map +1 -1
  49. package/dist/intelligence.d.ts +185 -120
  50. package/dist/intelligence.js +509 -345
  51. package/dist/intelligence.js.map +1 -1
  52. package/dist/knowledge.d.ts +40 -12
  53. package/dist/knowledge.js +13 -7
  54. package/dist/{loop-runner-bin-BRQSQdHa.d.ts → loop-runner-bin-Cn1N2rRo.d.ts} +3 -3
  55. package/dist/loop-runner-bin.d.ts +6 -6
  56. package/dist/loop-runner-bin.js +8 -10
  57. package/dist/loops.d.ts +13 -13
  58. package/dist/loops.js +6 -8
  59. package/dist/mcp/bin.js +5 -7
  60. package/dist/mcp/bin.js.map +1 -1
  61. package/dist/mcp/index.d.ts +6 -6
  62. package/dist/mcp/index.js +12 -14
  63. package/dist/mcp/index.js.map +1 -1
  64. package/dist/platform.js +0 -2
  65. package/dist/platform.js.map +1 -1
  66. package/dist/primeintellect/index.js +1 -2
  67. package/dist/primeintellect/index.js.map +1 -1
  68. package/dist/profile-DbfaMTdk.d.ts +233 -0
  69. package/dist/profiles.d.ts +1 -1
  70. package/dist/profiles.js +0 -1
  71. package/dist/profiles.js.map +1 -1
  72. package/dist/{supervise-DmYOug5f.d.ts → supervise-BLPI50-w.d.ts} +3 -3
  73. package/dist/{types-ByAYqlVb.d.ts → types-B3vAW0Oq.d.ts} +1 -1
  74. package/dist/{prepare-CtdtsFNG.d.ts → types-CWqfCO8s.d.ts} +67 -298
  75. package/dist/{types-BC3bZpH0.d.ts → types-CmYCMbFT.d.ts} +12 -54
  76. package/dist/{types-1d5QGK3t.d.ts → types-CmnA2iL3.d.ts} +3 -3
  77. package/dist/{worktree-fanout-CPprU-qI.d.ts → worktree-fanout-DCA3G4bO.d.ts} +3 -3
  78. package/package.json +14 -16
  79. package/skills/build-with-agent-runtime/SKILL.md +122 -213
  80. package/dist/chunk-6O73TRHW.js +0 -142
  81. package/dist/chunk-6O73TRHW.js.map +0 -1
  82. package/dist/chunk-AEG3NGJ2.js.map +0 -1
  83. package/dist/chunk-C3UKLQ54.js.map +0 -1
  84. package/dist/chunk-CNH7DF7Z.js.map +0 -1
  85. package/dist/chunk-D3H7F6L2.js +0 -626
  86. package/dist/chunk-D3H7F6L2.js.map +0 -1
  87. package/dist/chunk-DGUM43GV.js +0 -11
  88. package/dist/chunk-DGUM43GV.js.map +0 -1
  89. package/dist/chunk-EP6RVHMX.js.map +0 -1
  90. package/dist/chunk-HGRW27YY.js +0 -214
  91. package/dist/chunk-HGRW27YY.js.map +0 -1
  92. package/dist/chunk-ISTDY47H.js +0 -849
  93. package/dist/chunk-ISTDY47H.js.map +0 -1
  94. package/dist/chunk-PCURO3DL.js +0 -661
  95. package/dist/chunk-PCURO3DL.js.map +0 -1
  96. package/dist/chunk-QDSOD7RC.js.map +0 -1
  97. package/dist/chunk-TVJQAYQM.js.map +0 -1
  98. package/dist/chunk-VSWBYWFK.js.map +0 -1
  99. package/dist/chunk-VYA2YEKA.js.map +0 -1
  100. package/dist/generator-YkAQrOoD.d.ts +0 -382
  101. package/dist/improve-BN3HyXIO.d.ts +0 -172
  102. package/dist/lifecycle.d.ts +0 -870
  103. package/dist/lifecycle.js +0 -981
  104. package/dist/lifecycle.js.map +0 -1
  105. package/dist/mcp-serve-verifier-DQQDbuyz.d.ts +0 -34
  106. package/skills/agent-runtime-adoption/SKILL.md +0 -246
  107. /package/dist/{chunk-U33YZ7B2.js.map → chunk-FXF2OL34.js.map} +0 -0
  108. /package/dist/{chunk-XP5KDM3R.js.map → chunk-SGQ4YIQW.js.map} +0 -0
  109. /package/dist/{chunk-33OG2NN3.js.map → chunk-WYC2XJF2.js.map} +0 -0
  110. /package/dist/{chunk-ZEYAT33L.js.map → chunk-Y3SRWZMP.js.map} +0 -0
package/README.md CHANGED
@@ -33,7 +33,7 @@ pnpm tsx examples/driver-loop/driver-loop.ts
33
33
  | Run a **chat turn** for a production product agent | `handleChatTurn(...)` |
34
34
  | Have one agent **supervise a team of agents** toward a goal | `supervise(profile, task, opts)` |
35
35
  | **Improve** an agent and prove the gain on fresh tasks | `improve(profile, findings, opts)` |
36
- | **Improve** a knowledge base with agents, checks, and safe promotion | `runKnowledgeImprovementJob(...)` |
36
+ | Produce a measured knowledge-base candidate with agents and checks | `runKnowledgeImprovementJob(...)` |
37
37
  | Evaluate or train the same agent on **PrimeIntellect** | `createPrimeIntellectPackage(...)` |
38
38
 
39
39
  ### Run a chat turn
@@ -70,7 +70,7 @@ const result = await supervise(
70
70
 
71
71
  ### Improve an agent
72
72
 
73
- `improve` optimizes one part of an agent and **only ships a change if it beats the current agent on tasks it never practiced on**.
73
+ `improve` optimizes one part of an agent and returns a detached winner plus a decision. The decision is `ship` only when the candidate beats the current agent on tasks it never practiced on.
74
74
  It accepts prompt, skill document, curated memory, tool, MCP, hook, subagent, whole-profile, and code surfaces through one call.
75
75
  Prompt, skill-document, and memory optimization have built-in generators; structured profile surfaces take an explicit generator, and code runs from isolated incumbent and candidate checkouts.
76
76
  Workflow and rollout-policy files use the code surface so the measured winner is an exact patch that can be sealed and executed; JSON parameter sweeps use agent-eval's `parameterSweepProposer` instead of a runtime-specific optimizer.
@@ -78,26 +78,74 @@ Workflow and rollout-policy files use the code surface so the measured winner is
78
78
  ```ts
79
79
  import { improve } from '@tangle-network/agent-runtime'
80
80
 
81
- const { profile, shipped, lift } = await improve(baseProfile, findings, {
82
- surface: 'prompt', // or skills/memory/tools/mcp/hooks/subagents/agent-profile/code
83
- gate: 'holdout', // certified on a held-back exam, never the practice set
84
- scenarios, judge, agent, // how to measure a candidate
81
+ const { candidate, decision, lift } = await improve(baseProfile, findings, {
82
+ surface: 'prompt',
83
+ gate: 'holdout',
84
+ scenarios,
85
+ judge,
86
+ agent,
85
87
  })
88
+
89
+ if (decision === 'ship') console.log({ candidate, lift })
86
90
  ```
87
91
 
88
- Curated memory is an external lesson document, not a knowledge store. Supply its current text and persist only the promoted winner:
92
+ Skill and curated-memory candidates are exact profile changes, not free-floating text.
93
+ Name one inline skill through `skills.resourceName`; curated memory uses `profile.resources.instructions`.
94
+ Both require `profile.resources.failOnError: true` so an unsupported resource cannot silently disappear.
89
95
 
90
96
  ```ts
91
- await improve(baseProfile, findings, {
92
- surface: 'memory',
93
- memory: { document: currentLessons, writeBack: saveLessons },
97
+ const skillResult = await improve(baseProfile, findings, {
98
+ surface: 'skills',
99
+ skills: { resourceName: 'incident-response' },
94
100
  scenarios, judge, agent,
95
101
  })
96
102
  ```
97
103
 
104
+ `improve` is the search call.
105
+ For production, `proposeAgentImprovement` adds trace analysis and reruns the exact frozen baseline and winner before creating a reviewable proposal.
106
+ Runtime rejects a candidate bundle that differs from the search winner.
107
+
108
+ ```ts
109
+ import {
110
+ createAgentImprovementActivation,
111
+ executeAgentImprovementActivation,
112
+ proposeAgentImprovement,
113
+ reviewAgentImprovementProposal,
114
+ } from '@tangle-network/agent-runtime/intelligence'
115
+
116
+ const result = await proposeAgentImprovement({
117
+ runId,
118
+ profile: liveProfile,
119
+ analysis,
120
+ improvement: { surface: 'prompt', scenarios, judge, agent },
121
+ buildExperiment: ({ improvement }) => freezeExperiment(liveProfile, improvement.candidate),
122
+ placeCell,
123
+ })
124
+
125
+ const review = reviewAgentImprovementProposal(result.proposal, {
126
+ decision: 'approve',
127
+ reviewedBy: user.id,
128
+ reason: 'The measured gain is worth the cost.',
129
+ })
130
+ const activation = createAgentImprovementActivation(result.proposal, review, {
131
+ intent: 'activate-candidate',
132
+ targets: [{ surface: 'prompt', identity: profileId }],
133
+ fundingOwner: tenantId,
134
+ authorizedBy: user.id,
135
+ expiresAt,
136
+ })
137
+ const outcome = await executeAgentImprovementActivation(
138
+ { proposal: result.proposal, review, activation },
139
+ { transition: commitProfileTransaction, reconcile: readCommittedResult },
140
+ )
141
+ ```
142
+
143
+ `freezeExperiment`, `placeCell`, and the transaction functions are application ports because storage and compute differ by product.
144
+ Runtime owns candidate identity, measurement, review binding, expiry, retry identity, and result validation; the application owns its atomic write.
145
+
98
146
  ### Improve a knowledge base
99
147
 
100
- `runKnowledgeImprovementJob` is the runtime-owned front door for KB, wiki, memory-backed, and RAG improvement jobs. It creates a candidate copy, runs supervised agents against it, checks readiness through `@tangle-network/agent-knowledge`, measures spend and timing, and promotes only when the candidate passes. Use `improve(..., { surface: 'memory' })` for the agent's curated lesson document; use this job for source, retrieval, and knowledge-store changes.
148
+ `runKnowledgeImprovementJob` is the runtime-owned front door for KB, wiki, memory-backed, and RAG improvement jobs. It creates a candidate copy, runs supervised agents against it, checks readiness through `@tangle-network/agent-knowledge`, and returns frozen baseline and candidate snapshots with spend and timing. It never changes the live knowledge base. Use `improve(..., { surface: 'memory' })` for the agent's curated lesson document; use this job for source, retrieval, and knowledge-store changes.
101
149
 
102
150
  ```ts
103
151
  import { runKnowledgeImprovementJob } from '@tangle-network/agent-runtime/knowledge'
@@ -110,10 +158,11 @@ const result = await runKnowledgeImprovementJob({
110
158
  backend,
111
159
  })
112
160
 
113
- console.log(result.promoted, result.measurement.supervisedSpent)
161
+ console.log(result.knowledge?.reference.candidateHash, result.measurement.supervisedSpent)
114
162
  ```
115
163
 
116
- Use it when the product needs one knob for "make this knowledge base better" instead of wiring `improveKnowledgeBase`, a runtime supervisor, candidate workspaces, readiness checks, and promotion tracking by hand.
164
+ Use it when the product needs one knob for "make this knowledge base better" instead of wiring `improveKnowledgeBase`, a runtime supervisor, candidate workspaces, and readiness checks by hand.
165
+ Measure the returned bundle pair, record the review, then activate through `executeAgentImprovementActivation`; activation is the only write path.
117
166
 
118
167
  ### Run on PrimeIntellect
119
168
 
@@ -200,14 +249,14 @@ Runnable, grouped by what they show. Copy the one nearest your task:
200
249
  | Evaluate or train a runtime program on PrimeIntellect | `@tangle-network/agent-runtime/primeintellect` |
201
250
  | Study coordination vs raw compute | [`ablation-suite`](./examples/ablation-suite) |
202
251
 
203
- All 29 live in [`examples/`](./examples).
252
+ All 28 live in [`examples/`](./examples).
204
253
 
205
254
  ## Where to go next
206
255
 
207
256
  - New here? [`docs/concepts.md`](./docs/concepts.md), the mental model in plain terms.
208
257
  - [`docs/canonical-api.md`](./docs/canonical-api.md), find the primitive: "I want to ___ → use ___".
209
258
  - [`docs/api/primitive-catalog.md`](./docs/api/primitive-catalog.md), every export in one generated, never-stale list with its import path. Check it before building anything new.
210
- - Import subpaths: the root export is the product surface (`handleChatTurn`, `improve`); deeper capabilities ship as subpaths: `/loops` (multi-agent + the loop kernel), `/conversation` (multi-turn conversations), `/knowledge` (KB improvement), `/primeintellect` (Prime task, runtime, and trace adapter), `/mcp` (tool servers), `/intelligence` (observability drop-in), `/lifecycle`, `/agent`, `/profiles`, `/platform`, `/analyst-loop`, `/environment-provider`.
259
+ - Import subpaths: the root export is the product surface (`handleChatTurn`, `improve`); deeper capabilities ship as subpaths: `/loops` (multi-agent + the loop kernel), `/conversation` (multi-turn conversations), `/knowledge` (KB improvement), `/primeintellect` (Prime task, runtime, and trace adapter), `/mcp` (tool servers), `/intelligence` (observability drop-in), `/agent`, `/profiles`, `/platform`, `/analyst-loop`, `/environment-provider`.
211
260
  - [`docs/architecture.md`](./docs/architecture.md), the design, end to end.
212
261
  - [`bench/HARNESS.md`](./bench/HARNESS.md), the experiment harness and how to run a benchmark.
213
262
 
@@ -0,0 +1,63 @@
1
+ import { Sha256Digest, AgentImprovementActivationResult, AgentImprovementActivation, AgentCandidateBundle, AgentImprovementActivationTarget, AgentImprovementActivationOutcome, AgentImprovementProposal, AgentImprovementReview } from '@tangle-network/agent-interface';
2
+
3
+ interface CreateAgentImprovementActivationResultOptions {
4
+ completedAt: string;
5
+ outcome: AgentImprovementActivationOutcome;
6
+ }
7
+ interface AgentImprovementActivationTargetPlan extends AgentImprovementActivationTarget {
8
+ desiredDigest: Sha256Digest;
9
+ }
10
+ interface AgentImprovementActivationTransitionInput {
11
+ activation: AgentImprovementActivation;
12
+ candidateBundle: AgentCandidateBundle;
13
+ bundle: AgentCandidateBundle;
14
+ targets: [AgentImprovementActivationTargetPlan, ...AgentImprovementActivationTargetPlan[]];
15
+ attemptedAt: string;
16
+ expired: boolean;
17
+ }
18
+ interface AgentImprovementActivationResultStore {
19
+ load(idempotencyKey: Sha256Digest): Promise<unknown | undefined>;
20
+ putIfAbsent(result: AgentImprovementActivationResult): Promise<unknown>;
21
+ }
22
+ /**
23
+ * Product-owned or Runtime-composed transition.
24
+ *
25
+ * Implementations resolve a stored result for `activation.digest`, compare
26
+ * every target, and make the write durably idempotent. Co-located targets store
27
+ * the all-or-none write with its result. Other targets throw when result
28
+ * storage fails so a retry can reconcile it. Runtime never invokes this write
29
+ * function after authorization expires.
30
+ */
31
+ type AgentImprovementActivationTransition = (input: AgentImprovementActivationTransitionInput) => Promise<unknown>;
32
+ /**
33
+ * Target-read-only check for a prior exact write.
34
+ * It may persist recovered result metadata, but must not change an activation target.
35
+ * Return undefined only when no target write can have committed.
36
+ */
37
+ type AgentImprovementActivationReconciliation = (input: AgentImprovementActivationTransitionInput) => Promise<unknown | undefined>;
38
+ interface ExecuteAgentImprovementActivationInput {
39
+ proposal: AgentImprovementProposal;
40
+ review: AgentImprovementReview;
41
+ activation: AgentImprovementActivation;
42
+ }
43
+ interface ExecuteAgentImprovementActivationOptions {
44
+ transition: AgentImprovementActivationTransition;
45
+ reconcile?: AgentImprovementActivationReconciliation;
46
+ now?: () => Date;
47
+ }
48
+ /** Create the exact result a product stores in the same transaction as its target write. */
49
+ declare function createAgentImprovementActivationResult(transition: AgentImprovementActivationTransitionInput, options: CreateAgentImprovementActivationResultOptions): AgentImprovementActivationResult;
50
+ /**
51
+ * Recompute one historical activation result against the exact measured proposal and authority.
52
+ * The result records that attempt; it is not a query of the target's current state.
53
+ */
54
+ declare function verifyAgentImprovementActivationResult(input: {
55
+ proposal: unknown;
56
+ review: unknown;
57
+ activation: unknown;
58
+ result: unknown;
59
+ }): AgentImprovementActivationResult;
60
+ /** Validate and execute one product-owned activation transition. */
61
+ declare function executeAgentImprovementActivation(input: ExecuteAgentImprovementActivationInput, options: ExecuteAgentImprovementActivationOptions): Promise<AgentImprovementActivationResult>;
62
+
63
+ export { type AgentImprovementActivationReconciliation as A, type CreateAgentImprovementActivationResultOptions as C, type ExecuteAgentImprovementActivationInput as E, type AgentImprovementActivationResultStore as a, type AgentImprovementActivationTargetPlan as b, type AgentImprovementActivationTransition as c, type AgentImprovementActivationTransitionInput as d, type ExecuteAgentImprovementActivationOptions as e, createAgentImprovementActivationResult as f, executeAgentImprovementActivation as g, verifyAgentImprovementActivationResult as v };
package/dist/agent.d.ts CHANGED
@@ -1,13 +1,12 @@
1
1
  import * as _tangle_network_agent_eval from '@tangle-network/agent-eval';
2
- import { TraceAnalystKindSpec, AnalystFinding } from '@tangle-network/agent-eval';
3
- import { A as ArtifactKind, C as CandidateGenerator, P as PromotionGate } from './generator-YkAQrOoD.js';
2
+ import { TraceAnalystKindSpec } from '@tangle-network/agent-eval';
4
3
  import { R as RuntimeStreamEvent } from './types-BwoZWq-i.js';
5
- import { A as AgentSurfaces } from './improvement-adapter-BieWeK5J.js';
6
- export { C as CreateSurfaceImprovementAdapterOpts, D as DraftPatchInput, a as DraftPatchOutput, R as ResolvedSurface, S as SurfaceImprovementEdit, b as SurfaceValidationIssue, c as createSurfaceImprovementAdapter, r as renderSurfaceIssues, d as resolveSubjectPath, v as validateSurfaces } from './improvement-adapter-BieWeK5J.js';
7
- import { K as KnowledgeAdapter, a as RunAnalystLoopResult } from './types-BC3bZpH0.js';
4
+ import { A as AgentSurfaces } from './improvement-adapter-HAZz-7vK.js';
5
+ export { C as CreateSurfaceImprovementProposerOptions, D as DraftPatchInput, a as DraftPatchOutput, R as ResolvedSurface, S as SurfaceImprovementEdit, b as SurfaceValidationIssue, c as createSurfaceImprovementProposer, r as renderSurfaceIssues, d as resolveSubjectPath, v as validateSurfaces } from './improvement-adapter-HAZz-7vK.js';
8
6
  import { AgentProfile, AgentProfileFileMount, AgentProfileMcpServer } from '@tangle-network/agent-interface';
9
7
  import { SandboxEvent } from '@tangle-network/sandbox';
10
- import { S as SandboxClient, O as OutputAdapter, A as AgentRunSpec } from './types-ByAYqlVb.js';
8
+ import { S as SandboxClient, O as OutputAdapter, A as AgentRunSpec } from './types-B3vAW0Oq.js';
9
+ import './types-CmYCMbFT.js';
11
10
 
12
11
  /**
13
12
  * The full agent manifest. Each agent ships ONE of these.
@@ -82,48 +81,6 @@ interface AgentManifest<TPersona = unknown, TRunOutput = unknown> {
82
81
  * kinds (override per-kind via `analystKinds` if needed).
83
82
  */
84
83
  analyst: AnalystConfig;
85
- /**
86
- * Auto-apply policy. Knowledge / improvement edits land only when
87
- * `enabled === true` AND the source finding's confidence meets the
88
- * threshold. `mode` controls how applies happen: `'write'` mutates
89
- * files in-place; `'open-pr'` writes to a branch and opens a PR.
90
- *
91
- * Default: knowledge auto-applies at confidence ≥0.85 in `'write'`
92
- * mode (wiki edits are git-reversible); improvement stays at
93
- * `enabled: false` until the agent author has measured precision.
94
- */
95
- autoApply?: AutoApplyPolicy;
96
- /**
97
- * Declarative per-surface artifact-lifecycle config the closed loop reads.
98
- *
99
- * Each entry names a profile surface (`skill` / `tool` / `prompt` / `mcp` /
100
- * `hook` / `subagent`) and supplies the `CandidateGenerator` that grows it +
101
- * the `PromotionGate` (the held-back exam) that decides promotion. `runLifecycle`
102
- * consumes these: it pools the generators, measures each candidate's marginal
103
- * lift on the held-back split, gates it, and stores the winners in an
104
- * `ArtifactRegistry` — then `composeProfile` folds the top-`k` active artifacts
105
- * back into this agent's profile.
106
- *
107
- * Optional — agents that don't self-improve their profile omit it. An empty or
108
- * absent map means "no lifecycle"; the manifest is otherwise unchanged.
109
- */
110
- lifecycles?: ReadonlyArray<SurfaceLifecycle>;
111
- }
112
- /**
113
- * One profile surface's artifact-lifecycle wiring — the declarative config a
114
- * `defineAgent` manifest carries and `runLifecycle` reads. It is config, not
115
- * execution: it names the generator + gate; the loop runs them.
116
- */
117
- interface SurfaceLifecycle {
118
- /** The profile surface this lifecycle grows. */
119
- surface: ArtifactKind;
120
- /** Produces fresh candidate artifacts for `surface` from the agent's history. */
121
- generator: CandidateGenerator;
122
- /** The held-back exam that decides promotion of a measured candidate. */
123
- gate: PromotionGate;
124
- /** Top-`k` budget for `composeProfile` when folding this surface's promoted
125
- * artifacts back in. Omit to fold in every active artifact. */
126
- composeK?: number;
127
84
  }
128
85
  interface AgentRubric<TRunOutput> {
129
86
  /** Dimensions composing the weighted score. Weights sum to 1.0 by convention. */
@@ -251,18 +208,6 @@ interface AnalystConfig {
251
208
  baseUrl?: string;
252
209
  };
253
210
  }
254
- interface AutoApplyPolicy {
255
- knowledge?: {
256
- enabled: boolean;
257
- confidenceThreshold?: number;
258
- mode?: 'write' | 'open-pr';
259
- };
260
- improvement?: {
261
- enabled: boolean;
262
- confidenceThreshold?: number;
263
- mode?: 'write' | 'open-pr';
264
- };
265
- }
266
211
  /** Thrown when `defineAgent` finds a required surface missing on disk. */
267
212
  declare class AgentManifestError extends Error {
268
213
  readonly agentId: string;
@@ -284,138 +229,6 @@ declare class AgentManifestError extends Error {
284
229
  */
285
230
  declare function defineAgent<TPersona = unknown, TRunOutput = unknown>(manifest: AgentManifest<TPersona, TRunOutput>): AgentManifest<TPersona, TRunOutput>;
286
231
 
287
- /**
288
- * Substrate-default `KnowledgeAdapter` — wraps agent-knowledge's
289
- * `proposeFromFindings` + `applyKnowledgeWriteBlocks` with substrate
290
- * defaults (auto-lint after apply, source linkage via finding id).
291
- *
292
- * Every agent that ships a `.agent-knowledge/` tree uses this adapter
293
- * unmodified. Per-agent customization happens at the manifest level
294
- * (`autoApply.knowledge.confidenceThreshold`, etc.), not by writing a
295
- * new adapter.
296
- *
297
- * Lint discipline: after each apply we run agent-knowledge's
298
- * `lintKnowledgeIndex` to catch broken links / circular claims /
299
- * duplicate pages introduced by the new writes. Findings that fail the
300
- * post-apply lint are recorded in `warnings`; the apply itself is not
301
- * rolled back (lint failures are soft — humans review the wiki state).
302
- */
303
-
304
- interface CreateSurfaceKnowledgeAdapterOpts {
305
- /** `.agent-knowledge/` root (absolute path the substrate writes blocks against). */
306
- knowledgeRoot: string;
307
- }
308
- /**
309
- * Build the adapter. We accept the agent-knowledge functions as DI so
310
- * the substrate stays decoupled from a specific agent-knowledge
311
- * version — the agent author imports them in their manifest module
312
- * and hands them to the factory.
313
- *
314
- * `proposeFromFindings(findings)` returns
315
- * `{ proposals: KnowledgeProposal[]; skipped: number; errors: ... }`.
316
- *
317
- * `applyKnowledgeWriteBlocks(root, content)` returns
318
- * `{ written: string[]; warnings: string[] }`.
319
- *
320
- * `lintKnowledgeIndex(index)` (optional) returns `KnowledgeLintFinding[]`.
321
- */
322
- interface KnowledgeAdapterDeps<TProposal> {
323
- proposeFromFindings: (findings: ReadonlyArray<AnalystFinding>) => {
324
- proposals: TProposal[];
325
- skipped: number;
326
- errors: Array<{
327
- findingId: string;
328
- subject: string;
329
- message: string;
330
- }>;
331
- };
332
- applyKnowledgeWriteBlocks: (root: string, proposalText: string) => Promise<{
333
- written: string[];
334
- warnings: string[];
335
- }>;
336
- /**
337
- * Optional post-apply lint hook. The substrate runs it after each
338
- * batch of writes; failures land in `warnings` (the apply is not
339
- * rolled back — lint signals drift to review, not block).
340
- */
341
- lintAfterApply?: (root: string) => Promise<ReadonlyArray<string>>;
342
- }
343
- /** Wire a surface-based `KnowledgeAdapter` that writes analyst proposals to agent surface files. */
344
- declare function createSurfaceKnowledgeAdapter<TProposal>(opts: CreateSurfaceKnowledgeAdapterOpts, deps: KnowledgeAdapterDeps<TProposal>): KnowledgeAdapter<TProposal>;
345
-
346
- /**
347
- * `OutcomeMeasurement` — the missing metric that turns the analyst
348
- * loop from "observability" into "self-improvement".
349
- *
350
- * Without this hook, the loop reports process counts (`findings: 42`,
351
- * `applied: 7`) and never proves the applied edits actually improved
352
- * anything. With this hook, the substrate re-runs the cohort against
353
- * the same personas after each apply pass and reports a composite
354
- * score delta. A negative delta is the substrate's strongest signal
355
- * to either roll back or surface for review.
356
- *
357
- * Wiring is intentionally simple: pass the manifest + the `runAgentEval`
358
- * function and a list of `personaIds` to re-run. The wrapper:
359
- * 1. Captures the baseline composite from the just-finished run.
360
- * 2. After `runAnalystLoop` returns, re-invokes `runAgentEval` against
361
- * the same persona slice.
362
- * 3. Computes the delta and appends to `loop-report.json`.
363
- * 4. If `rollbackOnRegression` and delta < 0, reverts applied edits.
364
- */
365
-
366
- interface OutcomeMeasurement {
367
- /** Baseline composite before applies — captured from the most-recent eval run. */
368
- baselineComposite: number;
369
- /** Composite after re-running the cohort with applied edits. */
370
- afterComposite: number;
371
- /** `afterComposite - baselineComposite`. Positive = the loop improved the agent. */
372
- delta: number;
373
- /** Per-persona deltas for finer-grained review. */
374
- perPersona: ReadonlyArray<{
375
- personaId: string;
376
- before: number;
377
- after: number;
378
- delta: number;
379
- }>;
380
- /** When the substrate rolled back applies due to regression, the paths reverted. */
381
- rolledBackPaths: ReadonlyArray<string>;
382
- }
383
- interface OutcomeMeasurementOpts {
384
- /** Composite scores from the run that produced the findings. */
385
- baseline: ReadonlyArray<{
386
- personaId: string;
387
- composite: number;
388
- }>;
389
- /**
390
- * Re-run callback — the substrate invokes this after applies. The
391
- * agent author provides their `runAgentEval`-equivalent so the
392
- * substrate can ask "score this persona slice now."
393
- *
394
- * The callback SHOULD reuse the same cohort + judges + variant as
395
- * the baseline run; only the agent's mutable surfaces have changed.
396
- */
397
- reRunCohort: (personaIds: ReadonlyArray<string>) => Promise<ReadonlyArray<{
398
- personaId: string;
399
- composite: number;
400
- }>>;
401
- /** When `true`, applied edits are reverted on negative delta. Default `false`. */
402
- rollbackOnRegression?: boolean;
403
- /** Callback to revert a list of paths (typically `git checkout HEAD --`). */
404
- revert?: (paths: ReadonlyArray<string>) => Promise<void>;
405
- }
406
- /**
407
- * Run `runAnalystLoop` and stamp an `OutcomeMeasurement` onto the
408
- * result. The substrate calls this after each canonical eval; the
409
- * delta lands in `loop-report.json` for cross-run trend analysis.
410
- *
411
- * The function returns the original `RunAnalystLoopResult` enriched
412
- * with `outcome` so callers stay backwards-compatible (the field is
413
- * optional on the type; missing means no measurement was wired).
414
- */
415
- declare function measureOutcome<TProposal, TEdit>(result: RunAnalystLoopResult<TProposal, TEdit>, opts: OutcomeMeasurementOpts): Promise<RunAnalystLoopResult<TProposal, TEdit> & {
416
- outcome: OutcomeMeasurement;
417
- }>;
418
-
419
232
  /** Known AgentProfile axes a run path may or may not carry into execution. */
420
233
  declare const AGENT_PROFILE_MATERIALIZATION_AXES: readonly ["identity", "name", "model", "prompt", "systemPrompt", "instructions", "resources", "files", "resourceInstructions", "skills", "resourceTools", "resourceAgents", "commands", "tools", "permissions", "mcp", "mcpConnections", "connections", "subagents", "hooks", "modes", "confidential", "metadata", "extensions"];
421
234
  type KnownAgentProfileMaterializationAxis = (typeof AGENT_PROFILE_MATERIALIZATION_AXES)[number];
@@ -531,4 +344,4 @@ interface CreateSandboxActOptions<TPersona, TRunOutput> {
531
344
  */
532
345
  declare function createSandboxAct<TPersona, TRunOutput>(options: CreateSandboxActOptions<TPersona, TRunOutput>): (persona: TPersona, ctx: AgentRunContext) => AgentRunInvocation<TRunOutput>;
533
346
 
534
- export { AGENT_PROFILE_MATERIALIZATION_AXES, type AgentManifest, AgentManifestError, type AgentProfileMaterializationAxis, type AgentRubric, type AgentRunContext, type AgentRunInvocation, type AgentRuntime, AgentSurfaces, type AnalystConfig, type AssertProfileMaterializationOptions, type AutoApplyPolicy, type CreateSandboxActOptions, type CreateSurfaceKnowledgeAdapterOpts, type DefineProfileMaterializationContractOptions, type JudgeConfig, type KnowledgeAdapterDeps, type KnownAgentProfileMaterializationAxis, type OutcomeMeasurement, type OutcomeMeasurementOpts, type ProfileMaterializationContract, type ProfileMaterializationIssue, type RubricDimension, type SurfaceLifecycle, type ValidateProfileMaterializationOptions, assertProfileMaterialization, collectAgentRun, createSandboxAct, createSurfaceKnowledgeAdapter, defineAgent, defineProfileMaterializationContract, measureOutcome, promptOnlyProfileMaterialization, promptResourceProfileMaterialization, renderProfileMaterializationIssues, sandboxActProfileMaterialization, unimplementedAgentRun, validateProfileMaterialization };
347
+ export { AGENT_PROFILE_MATERIALIZATION_AXES, type AgentManifest, AgentManifestError, type AgentProfileMaterializationAxis, type AgentRubric, type AgentRunContext, type AgentRunInvocation, type AgentRuntime, AgentSurfaces, type AnalystConfig, type AssertProfileMaterializationOptions, type CreateSandboxActOptions, type DefineProfileMaterializationContractOptions, type JudgeConfig, type KnownAgentProfileMaterializationAxis, type ProfileMaterializationContract, type ProfileMaterializationIssue, type RubricDimension, type ValidateProfileMaterializationOptions, assertProfileMaterialization, collectAgentRun, createSandboxAct, defineAgent, defineProfileMaterializationContract, promptOnlyProfileMaterialization, promptResourceProfileMaterialization, renderProfileMaterializationIssues, sandboxActProfileMaterialization, unimplementedAgentRun, validateProfileMaterialization };