@tangle-network/agent-runtime 0.194.0 → 0.195.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (93) hide show
  1. package/dist/{activation-BQFIiyUG.js → activation-B4adh0sA.js} +3 -3
  2. package/dist/{activation-BQFIiyUG.js.map → activation-B4adh0sA.js.map} +1 -1
  3. package/dist/agent.d.ts +1 -1
  4. package/dist/agent.js +2 -2
  5. package/dist/candidate-execution/index.js +4 -4
  6. package/dist/{candidate-execution-B6CDW-fp.js → candidate-execution-nvqVIMyS.js} +5 -4
  7. package/dist/{candidate-execution-B6CDW-fp.js.map → candidate-execution-nvqVIMyS.js.map} +1 -1
  8. package/dist/{coordination-driver-iSN-m3VX.js → coordination-driver-CuShDjRV.js} +3 -3
  9. package/dist/{coordination-driver-iSN-m3VX.js.map → coordination-driver-CuShDjRV.js.map} +1 -1
  10. package/dist/{delegate-DCC2muUd.js → delegate-CSFC-Kh3.js} +2 -2
  11. package/dist/{delegate-DCC2muUd.js.map → delegate-CSFC-Kh3.js.map} +1 -1
  12. package/dist/durable.d.ts +1 -1
  13. package/dist/durable.js +2 -2
  14. package/dist/{environment-provider-Dr-wfnHg.js → environment-provider-Z1JazbJ1.js} +108 -10
  15. package/dist/environment-provider-Z1JazbJ1.js.map +1 -0
  16. package/dist/environment-provider.js +1 -1
  17. package/dist/{graph-DghidDx5.js → graph-DqPUeri6.js} +3 -3
  18. package/dist/{graph-DghidDx5.js.map → graph-DqPUeri6.js.map} +1 -1
  19. package/dist/graph.d.ts +1 -1
  20. package/dist/graph.js +4 -4
  21. package/dist/{improve-COGzLCiY.d.ts → improve-7VTwse1c.d.ts} +4 -4
  22. package/dist/{improvement-cycle-DeS1ZC5B.js → improvement-cycle-DEbciBTL.js} +20 -6
  23. package/dist/{improvement-cycle-DeS1ZC5B.js.map → improvement-cycle-DEbciBTL.js.map} +1 -1
  24. package/dist/{index-CrBgLCIf.d.ts → index-CVD30soi.d.ts} +2 -2
  25. package/dist/{index-BEPjOPwH.d.ts → index-DNx0nWZH.d.ts} +9 -2
  26. package/dist/index.d.ts +6 -5
  27. package/dist/index.js +72 -26
  28. package/dist/index.js.map +1 -1
  29. package/dist/intelligence.d.ts +1 -1
  30. package/dist/intelligence.js +7 -6
  31. package/dist/intelligence.js.map +1 -1
  32. package/dist/kernel.d.ts +1 -1
  33. package/dist/kernel.js +9 -9
  34. package/dist/{knowledge-B5okOzBQ.js → knowledge-B0Mrp-XB.js} +7 -17
  35. package/dist/knowledge-B0Mrp-XB.js.map +1 -0
  36. package/dist/knowledge.d.ts +1 -1
  37. package/dist/knowledge.js +1 -1
  38. package/dist/{loop-runner-bin-U1J6x_UF.js → loop-runner-bin-C5K46H0Q.js} +3 -3
  39. package/dist/{loop-runner-bin-U1J6x_UF.js.map → loop-runner-bin-C5K46H0Q.js.map} +1 -1
  40. package/dist/{loop-runner-bin-Clc5IAwu.d.ts → loop-runner-bin-b4585v9Z.d.ts} +2 -2
  41. package/dist/loop-runner-bin.d.ts +1 -1
  42. package/dist/loop-runner-bin.js +1 -1
  43. package/dist/mcp/bin.js +3 -3
  44. package/dist/mcp/index.d.ts +1 -1
  45. package/dist/mcp/index.js +5 -5
  46. package/dist/{prepare-C29kNAon.js → prepare-DDGp0-rW.js} +4 -54
  47. package/dist/prepare-DDGp0-rW.js.map +1 -0
  48. package/dist/profiles.js +454 -1
  49. package/dist/profiles.js.map +1 -1
  50. package/dist/{protected-model-port-CKYND416.js → protected-model-port-DxFN8DLS.js} +3 -2
  51. package/dist/{protected-model-port-CKYND416.js.map → protected-model-port-DxFN8DLS.js.map} +1 -1
  52. package/dist/{protected-redaction--F3v1oo8.js → protected-redaction-wGo44k2K.js} +54 -3
  53. package/dist/protected-redaction-wGo44k2K.js.map +1 -0
  54. package/dist/{provision-supervisor-CRPbnJiB.js → provision-supervisor-chyz0_Aj.js} +4 -4
  55. package/dist/{provision-supervisor-CRPbnJiB.js.map → provision-supervisor-chyz0_Aj.js.map} +1 -1
  56. package/dist/{redact-BOU77QfQ.js → redact-Dv6PeuQz.js} +2 -2
  57. package/dist/{redact-BOU77QfQ.js.map → redact-Dv6PeuQz.js.map} +1 -1
  58. package/dist/{runtime-BVrqSu0-.js → runtime-BZryF6gD.js} +140 -65
  59. package/dist/runtime-BZryF6gD.js.map +1 -0
  60. package/dist/{server-D6XmE1Du.js → server-BorSFmn8.js} +3 -3
  61. package/dist/{server-D6XmE1Du.js.map → server-BorSFmn8.js.map} +1 -1
  62. package/dist/{structural-rollout-BcPXFzVb.js → structural-rollout-pOqANr8Y.js} +23 -10
  63. package/dist/structural-rollout-pOqANr8Y.js.map +1 -0
  64. package/dist/{supervise-C0V3pZXK.js → supervise-BReI5ATC.js} +4 -4
  65. package/dist/{supervise-C0V3pZXK.js.map → supervise-BReI5ATC.js.map} +1 -1
  66. package/dist/testing.d.ts +1 -1
  67. package/dist/testing.js +14 -13
  68. package/dist/testing.js.map +1 -1
  69. package/dist/tui/index.d.ts +1 -1
  70. package/dist/tui/index.js +1 -1
  71. package/dist/{workspace-archive-BCezkeIk.js → workspace-archive-Ybomp7AN.js} +3 -2
  72. package/dist/{workspace-archive-BCezkeIk.js.map → workspace-archive-Ybomp7AN.js.map} +1 -1
  73. package/package.json +4 -4
  74. package/skills/agent-graphs/SKILL.md +28 -123
  75. package/skills/agent-graphs/references/authoring.md +34 -0
  76. package/skills/agent-graphs/references/measurement.md +18 -0
  77. package/skills/build-with-agent-runtime/SKILL.md +31 -117
  78. package/skills/build-with-agent-runtime/references/improvement.md +32 -0
  79. package/skills/codemode/SKILL.md +22 -34
  80. package/skills/codemode/references/runtime-execution.md +14 -0
  81. package/skills/generate-eval/SKILL.md +2 -1
  82. package/skills/loop-writer/SKILL.md +30 -84
  83. package/skills/supervise/SKILL.md +43 -95
  84. package/skills/supervise/references/profile-authoring.md +19 -0
  85. package/dist/environment-provider-Dr-wfnHg.js.map +0 -1
  86. package/dist/knowledge-B5okOzBQ.js.map +0 -1
  87. package/dist/prepare-C29kNAon.js.map +0 -1
  88. package/dist/protected-redaction--F3v1oo8.js.map +0 -1
  89. package/dist/researcher-Cp4JRbFp.js +0 -457
  90. package/dist/researcher-Cp4JRbFp.js.map +0 -1
  91. package/dist/runtime-BVrqSu0-.js.map +0 -1
  92. package/dist/structural-rollout-BcPXFzVb.js.map +0 -1
  93. package/skills/agent-graphs/IMPROVE.md +0 -58
@@ -1,139 +1,44 @@
1
1
  ---
2
2
  name: agent-graphs
3
- description: Author runGraph programs from AgentProfiles and versioned prompt directives.
3
+ description: Author fixed AgentGraph roles with explicit delegation, analysis, completion, and budgets.
4
4
  ---
5
5
 
6
- # Agent graphs
6
+ # Agent Graphs
7
7
 
8
- Use this skill when every role is known before execution and the relationship between roles must be reviewable as data.
9
- The output is an `AgentGraph` executed by `runGraph`, not a new coordinator or workflow framework.
8
+ Use `runGraph` when roles are known before execution and their relationships must be explicit Runtime data.
9
+ Use a smaller maintained composition when it provides the required behavior.
10
+ Use dynamic supervision when the agent must discover or create roles while working.
11
+ A request for review alone does not require a graph.
10
12
 
11
- ## Choose the existing entry point
13
+ Read the current [API decision table](https://github.com/tangle-network/agent-runtime/blob/main/docs/canonical-api.md), [AgentGraph contract](https://github.com/tangle-network/agent-runtime/blob/main/src/runtime/supervise/graph.ts), and a relevant [runnable example](https://github.com/tangle-network/agent-runtime/tree/main/examples/graphs).
14
+ These distinguish the fixed AgentGraph API from other graph or supervision contracts.
12
15
 
13
- | Need | Use |
14
- | --- | --- |
15
- | Known roles with versioned work and analysis instructions | `runGraph` |
16
- | A standard fixed shape such as parallel attempts, a chain, or a review panel | `fanout`, `pipeline`, `verify`, or `panel` |
17
- | A model decides which workers to create while it works | `supervise` |
18
- | One profile can complete the task directly | Run that profile without composition |
16
+ ## Author the required relationships
19
17
 
20
- Do not force a dynamic task into a static graph.
21
- Do not use a graph when a smaller shipped primitive already expresses the work.
18
+ Define the artifact and independent completion check before choosing roles.
19
+ Give each required role a complete AgentProfile and capabilities appropriate to its task.
20
+ Preserve requested parallel instances and independent reviewers rather than collapsing their distinct work into a root prompt.
22
21
 
23
- ### Strict authoring decisions (Do not under-graph)
22
+ When authoring nodes, directives, traversal limits, or analyst routes, read [the graph contract](references/authoring.md).
23
+ The runtime validates structure and prompt references before execution.
24
+ Keep the shared budget, per-worker allocation, concurrency, and supported limits in their actual API fields.
25
+ Use measured execution cost when available; do not turn a prior run into a universal minimum budget.
24
26
 
25
- - **Cheapness is not the dialect test:** Do not bail to `single-agent` just because a brief sounds trivial (e.g., "write a one-line file"). If the brief implies roles, observers, or a specific tight budget, author the graph.
26
- - **Identical-Role Parallelism:** If a brief requests N parallel instances of the same role, you MUST create N distinct worker nodes and N `delegates` edges. Do not collapse identical parallel workers into a single node.
27
- - **Mandatory Analysts:** If a brief requires independent observation, review, or post-settle findings (e.g., "neutral decider", "review by two perspectives", "watch the worker"), you MUST author `analyzes` edges. Do not omit analysts and attempt to merge their logic into the root's prompt.
28
- - **Caps are not stops:** Do not use an analysis edge `maxTraversals` cap as a global stop condition. To stop after N findings, use `deliverable.check` or `maxTraversals` on a `delegates` edge.
27
+ ## Prove the graph
29
28
 
30
- ## Author the complete contract
29
+ Use the maintained example pattern with injected test execution to check routes, directives, traversal limits, and both successful and rejected completion.
30
+ Then exercise the intended backend, profiles, tools, and completion check on a real representative task.
31
+ Offline control-flow tests do not establish that a real agent solves the task.
31
32
 
32
- An `AgentGraph` has four required fields: `nodes`, `edges`, `deliverable`, and `budget`.
33
- `runGraph(graph, options)` validates graph structure and prompt references before it spends compute.
33
+ Inspect the terminal result, complete cost and token accounting, edge delivery records, exhausted edges, and journal evidence.
34
+ An expected edge with zero traversals did not exercise its intended relationship.
35
+ Unknown usage is missing evidence, not zero cost.
36
+ A passing check proves only the outcome it actually tests.
34
37
 
35
- ### Nodes
36
-
37
- Each node is `{ id, profile }`, where `profile` is a complete canonical `AgentProfile`.
38
- Set `profile.name` equal to `id` because Runtime uses that value to select and route the node.
39
- Put the standing role in `profile.prompt.systemPrompt` and capabilities in the profile's tools, MCP, resources, hooks, and subagents.
40
- Do not rebuild profile materialization in graph code.
41
-
42
- ### Delegation edges
43
-
44
- A delegation edge is `{ kind: 'delegates', from, to, directive, maxTraversals? }`.
45
- The directive is a registered, versioned `PromptHandle`, such as `promptHandle('delegates/research-brief/v1')`.
46
- Each spawn and each later steer over the same edge consumes one traversal.
47
- The default cap is `defaultEdgeTraversalCap`; exhaustion refuses further delegation.
48
-
49
- The current graph form has one root and a static set of worker nodes.
50
- Every delegation edge starts at the root, and each worker has exactly one incoming delegation edge.
51
- Use a new directive version to change a brief instead of adding a second edge to the same worker.
52
-
53
- ### Analysis edges
54
-
55
- An analysis edge is `{ kind: 'analyzes', analyst, over, to, directive, maxTraversals? }`.
56
- It runs after a listed worker settles and routes findings to one node.
57
-
58
- `analyst` has two supported forms:
59
-
60
- - A lens id from `options.analysts` runs a caller-supplied analysis function.
61
- - A graph node id runs that node's pinned `AgentProfile` as a tool-equipped analyst.
62
-
63
- An analyst node has no incoming delegation edge, so the root cannot hand it ordinary work.
64
- An id cannot be both a registered lens and an analyst node.
65
- `over` lists delegated worker nodes only; Runtime refuses the root and analyst nodes because neither settles as an ordinary worker.
66
- An analysis traversal cap records excess findings as `unpropagated`; it does not stop the run.
67
-
68
- ### Completion and budget
69
-
70
- `deliverable.check(output)` is the independent completion test.
71
- It must accept a genuinely complete result and reject junk.
72
- Put the concrete mission in `deliverable.describe`; Runtime uses that text as the root's task.
73
-
74
- `budget` is one conserved pool for the full graph.
75
- Set `options.perWorker` explicitly from the actual executor cost.
76
- Size each allocation from measurements of the actual profile, mounted context, tools, and task shape when those measurements exist.
77
- Do not turn one run's cumulative spend into a universal harness minimum; Runtime enforces the caller's conserved pool, not guessed per-harness floors.
78
- Analyst nodes spend from the same pool and need the same honest accounting as ordinary workers.
79
-
80
- ## Authoring procedure
81
-
82
- 1. **Classify correctly:** Verify if this needs `single-agent`, `dynamic-workflow`, or a static `runGraph`. If independent review or parallel workers are requested, use `runGraph`.
83
- 2. **Define completion first:** Write the completion test and its description.
84
- 3. **Select entry point:** Choose the smallest shipped entry point from the table above.
85
- 4. **Define Roles:** Give every distinct role one complete `AgentProfile`. If N parallel instances of a role are requested, create N nodes. Merge roles only if their standing prompts and capabilities are identical.
86
- 5. **Register directives:** Register a versioned directive for every edge.
87
- 6. **Delegate work:** Add one delegation edge per ordinary worker from the root.
88
- 7. **Attach analysts:** Add `analyzes` edges only when findings must be produced independently after a worker settles. Do not skip this if the brief asked for a watcher/reviewer.
89
- 8. **Size the pool:** Set budget, per-worker allocation, traversal caps, time, and concurrency from comparable measured runs when available.
90
- 9. **Prove and inspect:** Run the structure offline, then run the real backend and inspect its result.
91
-
92
- ## Prove the graph before spending
93
-
94
- Use an injected `brain` plus `makeWorkerAgent` to exercise graph structure without a network call.
95
- Cover invalid profiles, unknown directives, impossible analysis routes, traversal exhaustion, successful completion, and rejected junk.
96
- Start from the runnable programs in `examples/graphs/` rather than creating a second graph runner.
97
-
98
- Offline execution proves control flow only.
99
- A real task must still use the intended backend, profiles, tools, completion test, and budget before claiming the graph solves that task.
100
-
101
- ## Read the complete result
102
-
103
- | Field | Meaning |
104
- | --- | --- |
105
- | `result.result.kind` and `reason` | Whether a result won and why execution ended |
106
- | `result.result.spentTotal` | Tokens and money, including whether each total is known |
107
- | `result.ledger` | Every delivered, stripped, empty, or unpropagated edge traversal with byte counts |
108
- | `result.exhaustedEdges` | Every edge whose cap was reached, including normal lifecycle endings |
109
- | Journal `edge` events | Durable copies of traversal evidence |
110
-
111
- Zero traversals on an expected edge means the graph did not exercise that relationship.
112
- `usdKnown: false` means cost is missing, not free.
113
- A passing completion test proves only what that test checks.
114
-
115
- ## Common mistakes
116
-
117
- - Bailing to `single-agent` because a brief sounds trivial, instead of respecting requested roles.
118
- - Collapsing N requested parallel identical roles into a single worker node.
119
- - Skipping `analyzes` edges when an observer or reviewer is explicitly requested.
120
- - Putting the task only in a spawn prompt instead of `deliverable.describe`.
121
- - Giving a node a `profile.name` different from its id.
122
- - Delegating ordinary work to an analyst node.
123
- - Listing the root or an analyst node in `analyzes.over`.
124
- - Using an analysis cap as a stop condition.
125
- - Allowing a driver-authored spawn profile to add capabilities instead of defining them on the pinned node profile.
126
- - Reading only thrown cap errors and missing `result.exhaustedEdges` on budget or cancellation endings.
127
- - Treating unknown spend as zero.
128
- - Claiming recursive or runtime-discovered structure when the current graph is a static root with workers and analysts.
129
-
130
- ## Improve only after measurement
131
-
132
- Runtime already optimizes one inline skill through `improve(profile, { surface: 'skills', skills: { resourceName }, ... })`.
133
- Put the exact skill bytes in `profile.resources.skills`, set `profile.resources.failOnError: true`, supply disjoint development and final-test tasks, and pass a complete Agent Eval optimization method.
134
- Do not create a graph-specific optimizer, campaign runner, candidate store, or promotion path.
38
+ When measuring a proposed change to this skill, read [measurement](references/measurement.md) before reusing its historical cases or results.
39
+ Ordinary graph construction does not require an optimization campaign.
135
40
 
136
41
  ## Then consider
137
42
 
138
- - `loop-writer` when the required dynamic structure still cannot be expressed by `supervise` or another shipped primitive; pass the exact missing behavior and the completion test.
139
- - `verify` before publishing a graph consumer; pass the real backend command, expected result fields, and failure cases.
43
+ - `loop-writer` when a required dynamic policy cannot be expressed through existing Runtime APIs.
44
+ - `verify` when the real task works and publishing or consumer integration checks remain.
@@ -0,0 +1,34 @@
1
+ # AgentGraph authoring contract
2
+
3
+ Use the current [graph types and validation](https://github.com/tangle-network/agent-runtime/blob/main/src/runtime/supervise/graph.ts) for exact fields.
4
+ Start from a matching [example](https://github.com/tangle-network/agent-runtime/tree/main/examples/graphs) rather than copying a second graph runner.
5
+
6
+ ## Nodes and work
7
+
8
+ AgentGraph requires nodes, edges, a deliverable, and a shared budget.
9
+ Each node holds an id and a canonical AgentProfile; its profile name matches the node id for routing.
10
+ Standing roles belong in the profile prompt, with explicit tools and resources.
11
+ Put the concrete mission in the deliverable description and supply an independent completion check that rejects incomplete output.
12
+
13
+ ## Edges
14
+
15
+ Delegation edges carry work from the root to a worker through registered prompt references.
16
+ A worker has one incoming delegation edge in this fixed graph form.
17
+ Keep the exact directive identity so recorded delivery can be traced to the text the worker received.
18
+ Spawns and steers consume delegation traversals; inspect exhaustion rather than assuming all requested work ran.
19
+
20
+ Analysis edges run after listed workers settle and route findings to their configured recipient.
21
+ The analyst is either a registered analysis function or a graph node with a complete profile.
22
+ An analyst node receives no ordinary delegation, and its id cannot also name a registered analysis function.
23
+ Analysis targets list delegated workers, not the root or other analysts.
24
+
25
+ Analysis traversal caps limit propagated findings; they do not stop the whole run.
26
+ Use the deliverable check or an appropriate execution limit to terminate work.
27
+ Analyst execution consumes the same shared budget as other workers.
28
+
29
+ ## Check the result
30
+
31
+ Inspect delivered, stripped, empty, and unpropagated edge records and their byte counts.
32
+ Read exhausted edges even when execution ended through completion, budget, or cancellation rather than a thrown error.
33
+ Check journal records when recovery or auditability is part of the task.
34
+ Preserve run, profile, directive, and artifact identity so a resumed or measured run cannot silently change its inputs.
@@ -0,0 +1,18 @@
1
+ # Measuring changes to graph-authoring guidance
2
+
3
+ The [case files](../cases) and [generation records](../generations) preserve previous experiments.
4
+ Read their recorded limitations before reuse; historical scores do not describe the current skill or API.
5
+ The generation records include an invalidated result and its reasons.
6
+
7
+ Use the current Runtime improvement API and a complete Eval method.
8
+ For the shared search and activation constraints, read [improvement and activation](../../build-with-agent-runtime/references/improvement.md).
9
+ Deliver the exact candidate skill resource to the authoring agent and retain its identity in the run evidence.
10
+
11
+ Check authored graph behavior through the actual graph implementation.
12
+ A case should distinguish the required relationship and reject a plausible wrong graph, not reward preferred words or unnecessary graph complexity.
13
+ Keep cases used to revise the skill separate from final decision cases.
14
+ If a reference is conditionally required by the skill, make it reachable in the measured resource package and inspect whether it was read.
15
+
16
+ Use real execution evidence when claiming improvement on a real backend.
17
+ Offline graph execution tests structure only; it cannot establish agent quality, deployed reliability, or paid execution cost.
18
+ Keep invalidated results as evidence without promoting their conclusions.
@@ -1,136 +1,50 @@
1
1
  ---
2
2
  name: build-with-agent-runtime
3
- description: Choose and compose current runtime, eval, knowledge, and interface APIs before adding wrappers.
3
+ description: Choose maintained runtime APIs and compose execution, evaluation, and controlled improvement.
4
4
  ---
5
5
 
6
- # Build with agent-runtime
6
+ # Build with Agent Runtime
7
7
 
8
- Use this skill before writing product-local agent infrastructure.
9
- The goal is one portable agent definition, one execution path, one measurement system, and one reviewed activation path.
8
+ Build on the maintained execution path while keeping product policy and storage in the consumer.
9
+ Read the current [API decision table](https://github.com/tangle-network/agent-runtime/blob/main/docs/canonical-api.md) and [package exports](https://github.com/tangle-network/agent-runtime/blob/main/package.json).
10
+ Follow the chosen entrypoint to its implementation and nearest runnable example.
11
+ For an existing consumer, confirm the actual installed package supports the chosen contract.
10
12
 
11
- ## Read first
13
+ ## Choose by the required outcome
12
14
 
13
- 1. Read `docs/canonical-api.md` for the current decision table.
14
- 2. Check exports in `src/index.ts`, `src/runtime/index.ts`, `src/improvement/index.ts`, `src/intelligence/index.ts`, and `src/knowledge/index.ts`.
15
- 3. Read the nearest runnable example.
16
- 4. Treat source as authoritative when docs disagree, then correct the stale doc in the same change.
17
-
18
- ## Ownership
15
+ Use the existing entrypoint for one turn, a bounded task, fixed composition, dynamic supervision, or a measured improvement.
16
+ Avoid copying the API catalog into product code or creating a wrapper that only renames it.
19
17
 
20
18
  | Concern | Owner |
21
19
  |---|---|
22
- | Portable prompt, skills, tools, MCP, hooks, subagents, model hints | `AgentProfile` from `@tangle-network/agent-interface` |
23
- | Agent execution, supervision, budgets, streaming, candidate execution | `@tangle-network/agent-runtime` |
24
- | Tasks, graders, search, paired statistics, cost and latency comparison | `@tangle-network/agent-eval` |
25
- | Sources, retrieval, citations, freshness, memory adapters, knowledge promotion | `@tangle-network/agent-knowledge` |
26
- | Product records, permissions, funding, UI, and atomic storage writes | The consuming product |
27
-
28
- Do not move shared measurement into Runtime or product code.
29
- Do not move product storage transactions into a provider-neutral package.
30
-
31
- ## Choose the entry point
32
-
33
- | Need | Use |
34
- |---|---|
35
- | One product chat turn | `handleChatTurn(...)` |
36
- | One normalized streamed agent turn | `streamAgentTurn(...)` and `collectAgentTurn(...)` |
37
- | One task or multi-turn loop | `runAgentTask(...)`, `runAgentTaskStream(...)`, or `runAgentRounds(...)` |
38
- | Supervisor and workers | `supervise(...)` or `superviseSurface(...)` |
39
- | Static roles with versioned delegation and analysis directives | `runGraph(...)` |
40
- | Parallel work with a shared budget | `fanout(...)` |
41
- | Fixed composition | `pipeline(...)`, `panel(...)`, or `verify(...)` |
42
- | Product benchmark | `defineLeaderboard(...)` |
43
- | Profile matrix | `expandProfileAxes(...)` and `runProfileMatrix(...)` from agent-eval |
44
- | Search one agent surface | `improve(...)` |
45
- | Analyze traces through a measured proposal | `proposeAgentImprovement(...)` |
46
- | Review and authorize an exact proposal | `reviewAgentImprovementProposal(...)` and `createAgentImprovementActivation(...)` |
47
- | Apply or restore an approved candidate | `executeAgentImprovementActivation(...)` with a product transaction |
48
- | Build a knowledge candidate | `runKnowledgeImprovementJob(...)` |
49
- | Apply a knowledge candidate | `createKnowledgeImprovementActivationExecutor(...)` through the same activation path |
50
- | Observe and pull approved changes on a live agent | `withIntelligence(...)` |
51
-
52
- ## Improvement flow
53
-
54
- `improve(profile, options)` searches one surface and returns a detached winner.
55
- It never changes a profile, document, repository, memory store, or knowledge base.
56
-
57
- For a profile field, pass one complete agent-eval `OptimizationMethod`, explicit train, selection, and final-test partitions, judges, and the candidate execution function.
58
- Use `officialGepa(...)` with an explicit recipe when upstream GEPA should own search.
59
- Use `officialSkillOpt(...)` when Microsoft's SkillOpt should own search.
60
- Both require `evaluationId`; change it whenever dispatch, judges, models, or scoring behavior changes.
61
- Resumable runs accept `never`, `if-compatible`, or `required` and reuse state only when agent-eval derives the same run identity.
62
- Runtime has no local prompt, skill, memory, or profile optimizer fallback.
63
- Code uses Runtime's isolated worktrees and returns a sealed patch candidate.
64
- Knowledge uses `runKnowledgeImprovementJob(...)` and returns paired snapshots.
65
-
66
- Use `proposeAgentImprovement(...)` for a production proposal.
67
- It performs these steps in order:
68
-
69
- 1. Analyze completed traces.
70
- 2. Search for a candidate on development tasks.
71
- 3. Build the frozen baseline, candidate, and held-back work.
72
- 4. Return only baseline, candidate, held-back tasks, and policy; Runtime adds the optimizer ancestry and seals the final experiment.
73
- 5. Reject the experiment if its candidate differs from the search winner.
74
- 6. Run baseline and candidate on the same held-back tasks.
75
- 7. Produce findings, confidence intervals, quality, cost, latency, and a decision.
76
-
77
- After a person or tenant policy approves the proposal, call `createAgentImprovementActivation(...)` with target identities, funding owner, authority, intent, and expiry.
78
- Runtime derives the expected current digests from the measured experiment.
79
- Call `executeAgentImprovementActivation(...)` with one product-owned transaction that compares current state, writes every target atomically, and stores the result under the activation digest.
80
- Pass a read-only reconciliation function so retries can distinguish committed, uncommitted, and uncertain outcomes.
81
-
82
- Never apply a change from analyst confidence alone.
83
- Never measure one candidate and apply another.
84
- Never let search code write live state.
85
- Never treat a lost response as a failed write without reconciling it.
86
-
87
- ## Surface rules
88
-
89
- - Prompt changes `profile.prompt` only and requires a complete method.
90
- - Skill optimization selects one inline skill by `skills.resourceName`, requires a complete method, and requires profile resources to fail closed.
91
- - Curated memory changes `profile.resources.instructions`; retrieval stores and memory databases belong in the knowledge flow.
92
- - Tools, MCP, hooks, subagents, curated memory, rollout policy, and whole-profile changes require a complete method.
93
- - Code candidates must come from the Runtime worktree path so patch identity and cleanup stay intact.
94
- - Workflow files are code surfaces. Parameter sweeps belong in a complete agent-eval method.
95
- - Knowledge candidates remain detached until the shared activation path applies or restores their frozen snapshots.
96
-
97
- ## Product integration
98
-
99
- The product supplies only the pieces that vary by deployment:
20
+ | Portable prompt, skills, tools, MCP, hooks, and model hints | AgentProfile from agent-interface |
21
+ | Execution, supervision, budgets, streaming, and candidate execution | Agent-runtime |
22
+ | Cases, grading, search, statistics, and comparison | Agent-eval |
23
+ | Retrieval, citations, freshness, memory stores, and knowledge promotion | Agent-knowledge |
24
+ | Users, permissions, funding, UI, persistence, and atomic writes | The product |
100
25
 
101
- - How traces and current profiles are loaded.
102
- - How exact candidate execution is placed on compute.
103
- - How proposal, review, activation, and result records are persisted.
104
- - How a target is changed atomically.
105
- - Who may approve, reject, request changes, fund, apply, or restore.
106
- - How those records and actions appear in the UI or API.
26
+ Keep measurement in Eval and product storage transactions in the consumer.
27
+ Use the same agent definition and execution path in the product and its evaluation.
107
28
 
108
- The product must not recreate candidate hashing, paired comparison, confidence intervals, review binding, expiry, retry identity, or result validation.
29
+ ## Integrate the selected capability
109
30
 
110
- ## Do not duplicate
31
+ Search for the existing product adapter and current package usage before adding infrastructure.
32
+ Supply only the policy, storage, credential, and execution-placement boundaries the consumer needs.
33
+ Preserve explicit failures, cost and usage capture, cancellation, and recovery behavior.
111
34
 
112
- - Do not write a provider-specific profile wrapper; extend `AgentProfile` and its materializer.
113
- - Do not write a second optimizer loop; pass a complete agent-eval method to `improve(...)`.
114
- - Do not use Runtime's code generator to approximate GEPA, SkillOpt, or another upstream profile optimizer.
115
- - Do not write a second candidate catalog; persist the immutable proposal records.
116
- - Do not let an analyst or adapter commit, push, open a pull request, or edit a live store.
117
- - Do not hand-roll SSE parsing, usage totals, profile matrices, bootstrap statistics, sandbox acquisition, or worktree cleanup.
118
- - Do not attach completed Runtime totals to an Eval campaign. Use `loopDispatch` or `loopCampaignDispatch` so admission and receipt capture surround the paid work.
119
- - Do not add a product-local approval format for knowledge, code, or profile changes.
35
+ When changing prompts, skills, code, or knowledge through measured search, read [improvement and activation](references/improvement.md) before implementing that path.
36
+ Ordinary execution work does not need an optimizer or activation workflow.
120
37
 
121
- ## Finish
38
+ ## Prove the integration
122
39
 
123
- - The same agent definition runs in product and measurement paths.
124
- - The held-back tasks were not visible during search.
125
- - Candidate identity is checked before execution and again before activation.
126
- - Quality, cost, latency, sample count, and uncertainty are retained.
127
- - Rejection and request-changes are first-class outcomes.
128
- - Activation is authorized, expiring, idempotent, and reconcilable.
129
- - No customer write, message, trigger, or billing occurs in read-only proof mode.
130
- - Public examples, package exports, generated API docs, type checks, tests, build, and package verification pass.
40
+ Run a real task through the selected backend and inspect its result and execution evidence.
41
+ Test the changed contract's denial, failure, cancellation, or recovery cases as applicable.
42
+ An in-process test proves only its own path; it does not prove a deployed sandbox path.
43
+ Run the repository's required checks, public-import checks when exports change, and the consumer's affected flow.
44
+ Report retained product adapters, adopted exports, observable results, and unchecked boundaries.
131
45
 
132
46
  ## Then consider
133
47
 
134
- - Use `build-with-agent-knowledge` when agents should improve retrieval, memory, or a knowledge base.
135
- - Use `critical-audit` when the change introduces or alters a public contract.
136
- - Use `verify` before publishing or adopting the package in a product.
48
+ - `build-with-agent-knowledge` when the remaining work concerns retrieval or memory integration.
49
+ - `critical-audit` when a changed public contract needs independent review.
50
+ - `verify` when implementation is complete and release checks remain.
@@ -0,0 +1,32 @@
1
+ # Measured improvement and activation
2
+
3
+ Read the current [improvement exports](https://github.com/tangle-network/agent-runtime/blob/main/src/improvement/index.ts), [intelligence exports](https://github.com/tangle-network/agent-runtime/blob/main/src/intelligence/index.ts), and relevant sections of the [API decision table](https://github.com/tangle-network/agent-runtime/blob/main/docs/canonical-api.md).
4
+ For knowledge changes, also read the [knowledge exports](https://github.com/tangle-network/agent-runtime/blob/main/src/knowledge/index.ts).
5
+ Use the selected function's current types and maintained example rather than copying a method signature from this guide.
6
+
7
+ ## Search without changing the live system
8
+
9
+ Use the existing improvement API and a complete Eval optimization method for the chosen surface.
10
+ Keep development, selection, and final decision cases separate.
11
+ Record the delivered profile resources and execution identity so resumed work cannot silently reuse incompatible measurements.
12
+ Search returns a detached candidate; it cannot edit the live product, knowledge store, or repository.
13
+
14
+ Prompt, tool, resource, and profile changes remain portable profile data.
15
+ Code candidates use Runtime's isolated worktree and patch identity path.
16
+ Knowledge candidates use the existing snapshot and promotion contract.
17
+ Do not rebuild candidate hashing, statistics, or search history in the consumer.
18
+
19
+ ## Apply only the measured candidate
20
+
21
+ Use the maintained proposal, review, and activation path.
22
+ The proposal must compare the unchanged baseline and exact candidate on tasks hidden during search.
23
+ Keep candidate identity checked before execution and before activation.
24
+ Retain quality, cost, latency, sample count, uncertainty, and rejected outcomes.
25
+
26
+ The product supplies authority, funding, target identities, persistence, and an atomic transaction.
27
+ That transaction compares expected current state, writes the authorized targets, and records the activation outcome under its retry identity.
28
+ Use read-only reconciliation to distinguish committed, uncommitted, and uncertain outcomes after a lost response.
29
+ Preserve expiry and authority checks; review evidence does not itself grant write authority.
30
+
31
+ Prove rejection, successful activation, expired or mismatched activation, and retry after an uncertain write.
32
+ A read-only experiment must not send customer messages, mutate customer data, or incur product billing side effects.
@@ -1,49 +1,37 @@
1
1
  ---
2
2
  name: codemode
3
- description: Batch mechanical tool work as one program so loops and intermediates stay out of context.
3
+ description: Batch mechanical tool work in code while preserving judgment, authorization, and accounting.
4
4
  ---
5
5
 
6
6
  # Codemode
7
7
 
8
- Use this policy when a task needs three or more mechanical tool or command calls whose intermediate results need no judgment.
9
- One call per model turn spends a round trip per step and pushes every intermediate value through the context window.
10
- Write one program instead: the loop, the branch, and the intermediates stay in the program, and only the decision-relevant summary returns.
8
+ Batch stretches of mechanical tool work whose intermediate results require no judgment.
9
+ Keep decisions that could change the plan in the agent's turn.
11
10
 
12
- This is the pattern the ecosystem calls code mode (Cloudflare's Code Mode, Anthropic's code execution with MCP, the CodeAct paper).
13
- In a coding harness you already have the whole capability: a shell, a filesystem, and the tools this profile grants.
11
+ ## Batch the work
14
12
 
15
- ## Run The Work
13
+ Identify independent calls and the points where an observed result must change the next action.
14
+ Use the session's permitted execution tool to hold intermediate data in variables or workspace files.
15
+ Return the decision-relevant values, failures, and artifact locations.
16
+ Inspect every result; a successful batch must not hide a failed item.
17
+ Retain results needed later instead of rerunning expensive work to recover them.
16
18
 
17
- 1. List the calls the task needs and mark which results require your judgment.
18
- 2. Put every judgment-free stretch into one script; keep each judgment point in your own turn.
19
- 3. Hold intermediates in variables or files inside the workspace, never in your reply.
20
- 4. Make the script print only the decision-relevant summary: counts, failures, the final value.
21
- 5. Prefer one script that fans out over N items to N separate tool calls with identical shape.
22
- 6. Stop batching the moment a result changes what you would do next; read it, decide, then batch again.
19
+ Respect each tool's concurrency, cancellation, and authorization contract.
20
+ A batch does not expand the permission of its individual operations.
21
+ Meter paid operations on the owning execution path and preserve their usage records.
22
+ Keep dependent actions sequential unless their contract supports safe composition.
23
23
 
24
- ## Boundaries That Are Not Yours To Move
24
+ ## Runtime-supervised code
25
25
 
26
- Code may spawn or steer only through Runtime-provided API bindings such as `api.spawn_worker`.
27
- Never reach coordination verbs over HTTP or create a second scheduler; that bypasses the budget pool and journal.
28
- An operation that costs money must run where the runtime meters it; do not wrap metered work in a script that hides the spend.
29
- The lint on authored code refuses imports, `process`, and network access; it is a lint, not a sandbox, so treat generated code you did not review as untrusted.
26
+ When configuring code execution for a Runtime supervisor, read [the execution boundary](references/runtime-execution.md).
27
+ That branch supplies a generated API over Runtime's existing coordination tools and requires an explicit runner.
28
+ For ordinary shell or session-tool batching, no additional runtime is needed.
30
29
 
31
- ## Router-Brained Supervisors
32
-
33
- A raw chat model has no shell, so give it the runtime's code mode: pass `codeModeSupervisorTools()` as `resolveSupervisorTools` and the supervisor's tool surface becomes `search` and `execute`.
34
- `search` answers a TypeScript API generated from the live coordination grant; `execute` runs the model's program through a caller-supplied runner, and every `api.spawn_worker` call crosses the kernel's pool, authorization, and journal.
35
- Supply a jailed runner for an untrusted model: the in-process runner is not an isolation boundary.
36
- The lifecycle verbs (`submit_result`, `stop`, `ask_parent`) stay model tools: the program does the mechanics, the model keeps the judgment.
37
-
38
- ## Common Mistakes
39
-
40
- - Batching a step whose output should have changed your plan, then discovering it three steps later.
41
- - Printing a whole dataset into the reply instead of writing it to a file and printing the summary.
42
- - Re-running an expensive script to re-read a value the first run already produced; write results to files.
43
- - Moving supervision into a script because the coordination verbs are reachable over local HTTP.
30
+ Complete the requested work and inspect the resulting artifact.
31
+ Code reduces mechanical round trips; it does not replace judgment or prove the quality of the result.
44
32
 
45
33
  ## Then consider
46
34
 
47
- - `supervise` when the batched work is really delegation to workers with their own judgment.
48
- - `agent-graphs` when the shape of the work is a fixed topology rather than one agent's loop.
49
- - `loop-writer` when no shipped composition API can express the control policy you need.
35
+ - `supervise` when the work needs workers with their own judgment.
36
+ - `agent-graphs` when fixed roles require explicit Runtime relationships and shared accounting.
37
+ - `loop-writer` when no maintained composition expresses the required control policy.
@@ -0,0 +1,14 @@
1
+ # Runtime code execution boundary
2
+
3
+ Read [code-mode implementation and types](https://github.com/tangle-network/agent-runtime/blob/main/src/runtime/supervise/code-mode.ts) and its [contract tests](https://github.com/tangle-network/agent-runtime/blob/main/tests/kernel/code-mode.test.ts).
4
+ Use `codeModeSupervisorTools(runner)` with an explicit `CodeModeRunner`; there is no default runner.
5
+ For untrusted model output, supply a real isolated execution environment.
6
+ The in-process runner and source lint are not security boundaries.
7
+
8
+ The generated API follows the live coordination grant.
9
+ Code can spawn or steer through Runtime-provided bindings such as `api.spawn_worker`; these retain authorization, shared budgets, and journal records.
10
+ Direct coordination requests over HTTP or a second scheduler bypass that contract.
11
+
12
+ Lifecycle decisions remain model tools: `submit_result`, `stop`, and `ask_parent` are outside the generated code API.
13
+ Keep judgment in the model and mechanics in the program.
14
+ Before broad use, exercise one allowed call, one denied call, a failed operation, and cancellation through the selected runner.
@@ -13,8 +13,9 @@ Do not use it for general coding quality or subjective output.
13
13
  - `TARGET`: a pinned package version, repository commit, or release.
14
14
  - `OUT`: the path for one candidate JSON object.
15
15
 
16
- Read `bench/src/generate-eval/schema.ts` and `bench/src/generate-eval/certify.ts` before authoring the candidate.
16
+ Read the current [candidate schema](https://github.com/tangle-network/agent-runtime/blob/main/bench/src/generate-eval/schema.ts) and [execution checks](https://github.com/tangle-network/agent-runtime/blob/main/bench/src/generate-eval/certify.ts) before authoring the candidate.
17
17
  Those files define the current format and checks.
18
+ Use a maintained target for new cases, then freeze its exact identity so later runs compare the same behavior.
18
19
 
19
20
  ## Build One Case
20
21