@tangle-network/agent-runtime 0.123.1 → 0.128.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/dist/agent.d.ts +1 -1
- package/dist/agent.js +2 -2
- package/dist/{environment-provider-PM9PeW_J.d.ts → environment-provider-CUFsyymu.d.ts} +7 -1
- package/dist/environment-provider.d.ts +1 -1
- package/dist/{index-xP27vtnt.d.ts → index-BhZhQw77.d.ts} +247 -145
- package/dist/{index-4NcThsDc.d.ts → index-BhuzfG2r.d.ts} +3 -3
- package/dist/{index-CQBioeAj.d.ts → index-DLM0W1h1.d.ts} +5 -5
- package/dist/index.d.ts +5 -5
- package/dist/index.js +5 -5
- package/dist/index.js.map +1 -1
- package/dist/intelligence.d.ts +1 -1
- package/dist/kernel.d.ts +3 -3
- package/dist/kernel.js +3 -3
- package/dist/{knowledge-BOHj6nGh.js → knowledge-DF63xPr4.js} +2 -2
- package/dist/{knowledge-BOHj6nGh.js.map → knowledge-DF63xPr4.js.map} +1 -1
- package/dist/knowledge.d.ts +1 -1
- package/dist/knowledge.js +1 -1
- package/dist/{loop-runner-bin-DUM37LLw.js → loop-runner-bin-CWqOpCEw.js} +3 -3
- package/dist/{loop-runner-bin-DUM37LLw.js.map → loop-runner-bin-CWqOpCEw.js.map} +1 -1
- package/dist/{loop-runner-bin-DMnNxZHw.d.ts → loop-runner-bin-Ckp_9tmD.d.ts} +3 -3
- package/dist/loop-runner-bin.d.ts +1 -1
- package/dist/loop-runner-bin.js +1 -1
- package/dist/mcp/bin.js +1 -1
- package/dist/mcp/index.d.ts +3 -7
- package/dist/mcp/index.js +3 -3
- package/dist/{openai-tools-Bm1SDJIp.js → openai-tools-D3XfrrQ6.js} +2 -2
- package/dist/{openai-tools-Bm1SDJIp.js.map → openai-tools-D3XfrrQ6.js.map} +1 -1
- package/dist/primeintellect/index.d.ts +1 -1
- package/dist/{runtime-DZblIf3R.js → runtime-5uDVVfER.js} +444 -132
- package/dist/runtime-5uDVVfER.js.map +1 -0
- package/dist/{supervise-Cx24b3lw.js → supervise-CsTKbH9R.js} +209 -24
- package/dist/supervise-CsTKbH9R.js.map +1 -0
- package/dist/testing.js +8 -8
- package/package.json +4 -4
- package/skills/agent-graphs/IMPROVE.md +58 -0
- package/skills/agent-graphs/SKILL.md +140 -0
- package/skills/agent-graphs/cases/artifact-mission-release-notes.json +10 -0
- package/skills/agent-graphs/cases/audited-single-writer.json +9 -0
- package/skills/agent-graphs/cases/cap-as-stop-mistake.json +8 -0
- package/skills/agent-graphs/cases/floor-trap-pi.json +11 -0
- package/skills/agent-graphs/cases/mission-in-deliverable.json +8 -0
- package/skills/agent-graphs/cases/review-pipeline.json +14 -0
- package/skills/agent-graphs/cases/runtime-discovered-fanout.json +8 -0
- package/skills/agent-graphs/cases/single-agent-suffices.json +7 -0
- package/skills/agent-graphs/cases/steer-heavy-drafting.json +9 -0
- package/skills/agent-graphs/cases/unmeasured-harness.json +9 -0
- package/skills/agent-graphs/generations/gen1-baseline.json +248 -0
- package/skills/agent-graphs/generations/gen2.json +375 -0
- package/skills/agent-graphs/generations/gen3.json +702 -0
- package/skills/build-with-agent-runtime/SKILL.md +1 -0
- package/dist/runtime-DZblIf3R.js.map +0 -1
- package/dist/supervise-Cx24b3lw.js.map +0 -1
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
{
|
|
2
|
+
"skillVersion": "v1",
|
|
3
|
+
"surfaceSource": "working-tree",
|
|
4
|
+
"authorModel": "glm-5.2",
|
|
5
|
+
"temperature": 0.2,
|
|
6
|
+
"date": "2026-08-03T17:42:45.828Z",
|
|
7
|
+
"n": 8,
|
|
8
|
+
"aggregate": {
|
|
9
|
+
"mean": 0.6375,
|
|
10
|
+
"median": 0.8,
|
|
11
|
+
"min": 0,
|
|
12
|
+
"max": 1,
|
|
13
|
+
"n": 8,
|
|
14
|
+
"correction": "first publication reported median 1.0 from an upper-middle indexing bug; corrected to the interpolated median",
|
|
15
|
+
"samplePerCase": 1,
|
|
16
|
+
"varianceNote": "decision-level variance observed at temp 0.2 (cap-as-stop-mistake flipped graph->single-agent on rerun); treat per-case scores as n=1 draws, not stable properties. Reps belong to the v2 gate round."
|
|
17
|
+
},
|
|
18
|
+
"cases": [
|
|
19
|
+
{
|
|
20
|
+
"id": "cap-as-stop-mistake",
|
|
21
|
+
"decision": "graph",
|
|
22
|
+
"score": 1,
|
|
23
|
+
"reasons": [
|
|
24
|
+
"PASS trapIsAnalyzesCapAsStop: analyzes caps=true delegates caps=true",
|
|
25
|
+
"PASS correctStopIsDelegatesCapOrDeliverable: delegates caps=true, reason mentions deliverable=true"
|
|
26
|
+
],
|
|
27
|
+
"reason": "The brief asks for an analyst to watch a worker \u2014 a legitimate unaddressable observation lens \u2014 so a static graph is the right dialect. But the brief's stated stop mechanism is the textbook pothole: an analyzes edge's maxTraversals is observability-only; on exhaustion the traversal is ledgered 'unpropagated' and it never refuses, so it cannot terminate a run. Only a delegates cap or the deliverable check can. I therefore encode 'three findings' in the deliverable (the real termination oracle) and let the worker settle once three findings exist; the findings-lens analyst observes and routes to the root with maxTraversals=3 as the intended observation budget, but it is NOT load-bearing for termination. If the brief truly needs hard termination at exactly three analyst findings regardless of worker output, that is not expressible in the current P0 topology and must be re-scoped before running.",
|
|
28
|
+
"authoredGraph": {
|
|
29
|
+
"nodes": [
|
|
30
|
+
{
|
|
31
|
+
"id": "root",
|
|
32
|
+
"systemPrompt": "You are the root orchestrator. Hold the mission: produce at least three distinct findings on the delegated subject. Delegate to the worker, accept the findings-lens observations routed back to you, and stop spawning the moment the worker's settled output contains three findings. Do not delegate further once three are present."
|
|
33
|
+
},
|
|
34
|
+
{
|
|
35
|
+
"id": "worker",
|
|
36
|
+
"systemPrompt": "You are the findings worker. Investigate the delegated brief and emit discrete, numbered findings, one paragraph each. Continue until you have produced at least three distinct findings, then settle. Do not pad or repeat."
|
|
37
|
+
}
|
|
38
|
+
],
|
|
39
|
+
"edges": [
|
|
40
|
+
{
|
|
41
|
+
"kind": "delegates",
|
|
42
|
+
"from": "root",
|
|
43
|
+
"to": "worker",
|
|
44
|
+
"maxTraversals": 6
|
|
45
|
+
},
|
|
46
|
+
{
|
|
47
|
+
"kind": "analyzes",
|
|
48
|
+
"analyst": "findings-lens",
|
|
49
|
+
"over": [
|
|
50
|
+
"worker"
|
|
51
|
+
],
|
|
52
|
+
"to": "root",
|
|
53
|
+
"maxTraversals": 3
|
|
54
|
+
}
|
|
55
|
+
],
|
|
56
|
+
"budget": {
|
|
57
|
+
"maxIterations": 8,
|
|
58
|
+
"maxTokens": 240000
|
|
59
|
+
},
|
|
60
|
+
"perWorker": {
|
|
61
|
+
"maxIterations": 4,
|
|
62
|
+
"maxTokens": 80000
|
|
63
|
+
},
|
|
64
|
+
"deliverableDescribe": "Investigate the delegated subject and report at least three distinct, numbered findings. The run terminates when the worker's settled output contains three findings; the findings-lens analyst observes the worker and routes its observations to the root (observation only \u2014 it does not stop the run)."
|
|
65
|
+
},
|
|
66
|
+
"runResultKind": "winner",
|
|
67
|
+
"ledgerRows": 2,
|
|
68
|
+
"exhaustedEdges": []
|
|
69
|
+
},
|
|
70
|
+
{
|
|
71
|
+
"id": "floor-trap-pi",
|
|
72
|
+
"decision": "single-agent",
|
|
73
|
+
"score": 0,
|
|
74
|
+
"reasons": [
|
|
75
|
+
"FAIL mustBudgetAtLeast: no graph authored (decision=single-agent)",
|
|
76
|
+
"FAIL nodes: no graph authored",
|
|
77
|
+
"FAIL analyzesWarranted: no graph authored"
|
|
78
|
+
],
|
|
79
|
+
"reason": "The task is trivially one-shot-able by a single pi agent (writing a one-line file) without requiring parallel fan-out or independent verification. Furthermore, the brief's tight budget constraint of a few thousand tokens falls drastically below the measured 31,211 input token floor (`WORKER_TOKEN_FLOOR`) required for a `runGraph` spawn, meaning any graph attempt would fail with a `below-runtime-floor` refusal."
|
|
80
|
+
},
|
|
81
|
+
{
|
|
82
|
+
"id": "mission-in-deliverable",
|
|
83
|
+
"decision": "single-agent",
|
|
84
|
+
"score": 0,
|
|
85
|
+
"reasons": [
|
|
86
|
+
"FAIL deliverableDescribeCarriesMission: no graph authored"
|
|
87
|
+
],
|
|
88
|
+
"reason": "A single agent can author a CHANGELOG entry that satisfies a format check in a single pass, requiring no topology, parallelism, or independent verification."
|
|
89
|
+
},
|
|
90
|
+
{
|
|
91
|
+
"id": "review-pipeline",
|
|
92
|
+
"decision": "graph",
|
|
93
|
+
"score": 0.6,
|
|
94
|
+
"reasons": [
|
|
95
|
+
"PASS nodes: workers=2 total=3 expected=3",
|
|
96
|
+
"FAIL analyzesWarranted: analyzes edges=0 warranted=true",
|
|
97
|
+
"PASS wrongIfAnalystIsNode: no analyst id collides with a node id",
|
|
98
|
+
"PASS edge:delegates to each reviewer: workers [security-reviewer, maintainability-reviewer] each delegated-to with >0 ledger traversals: true",
|
|
99
|
+
"FAIL edge:analyzes routing findings to root: analyzes\u2192root edges=0, fired with >0 traversals=false"
|
|
100
|
+
],
|
|
101
|
+
"reason": "The case calls for two genuinely distinct review perspectives (security/correctness vs maintainability) that must settle independently, then a neutral arbiter synthesizing both \u2014 a fixed topology with independent verification that earns composition. One agent cannot maintain two addressable, independently-settled roles plus a neutral third.",
|
|
102
|
+
"authoredGraph": {
|
|
103
|
+
"nodes": [
|
|
104
|
+
{
|
|
105
|
+
"id": "merge-arbiter",
|
|
106
|
+
"systemPrompt": "You are the neutral merge arbiter. You delegate code changes to two independent reviewers \u2014 one for security/correctness, one for maintainability \u2014 then collect both reviews and render a final merge decision: APPROVE, REQUEST-CHANGES, or REJECT. You must cite specific findings from each reviewer in your decision rationale. You do not review code yourself; you synthesize and decide."
|
|
107
|
+
},
|
|
108
|
+
{
|
|
109
|
+
"id": "security-reviewer",
|
|
110
|
+
"systemPrompt": "You are a security and correctness reviewer. Examine code changes for vulnerabilities, input-validation gaps, race conditions, data-integrity risks, logic errors, and unhandled edge cases. Report each finding with a severity rating (critical, high, medium, low) and a concrete location in the diff. Do not comment on style or naming unless it introduces a correctness risk."
|
|
111
|
+
},
|
|
112
|
+
{
|
|
113
|
+
"id": "maintainability-reviewer",
|
|
114
|
+
"systemPrompt": "You are a maintainability and clarity reviewer. Examine code changes for naming quality, structural clarity, test coverage, documentation gaps, convention adherence, and future changeability. Report each finding with a priority rating (blocker, important, minor) and a concrete location in the diff. Do not comment on security unless a pattern directly blocks maintainability."
|
|
115
|
+
}
|
|
116
|
+
],
|
|
117
|
+
"edges": [
|
|
118
|
+
{
|
|
119
|
+
"kind": "delegates",
|
|
120
|
+
"from": "merge-arbiter",
|
|
121
|
+
"to": "security-reviewer",
|
|
122
|
+
"maxTraversals": 2
|
|
123
|
+
},
|
|
124
|
+
{
|
|
125
|
+
"kind": "delegates",
|
|
126
|
+
"from": "merge-arbiter",
|
|
127
|
+
"to": "maintainability-reviewer",
|
|
128
|
+
"maxTraversals": 2
|
|
129
|
+
}
|
|
130
|
+
],
|
|
131
|
+
"budget": {
|
|
132
|
+
"maxIterations": 6,
|
|
133
|
+
"maxTokens": 300000
|
|
134
|
+
},
|
|
135
|
+
"perWorker": {
|
|
136
|
+
"maxIterations": 3,
|
|
137
|
+
"maxTokens": 80000
|
|
138
|
+
},
|
|
139
|
+
"deliverableDescribe": "Review the provided code change from two independent perspectives \u2014 security/correctness and maintainability \u2014 then produce a merge decision (APPROVE, REQUEST-CHANGES, or REJECT) with a rationale that cites specific findings from each reviewer."
|
|
140
|
+
},
|
|
141
|
+
"runResultKind": "winner",
|
|
142
|
+
"ledgerRows": 2,
|
|
143
|
+
"exhaustedEdges": []
|
|
144
|
+
},
|
|
145
|
+
{
|
|
146
|
+
"id": "runtime-discovered-fanout",
|
|
147
|
+
"decision": "dynamic-workflow",
|
|
148
|
+
"score": 1,
|
|
149
|
+
"reasons": [
|
|
150
|
+
"PASS correctAnswerIsDynamicWorkflow: decision=dynamic-workflow"
|
|
151
|
+
],
|
|
152
|
+
"reason": "The task requires identifying an unknown number of failing tests at runtime and fixing them in parallel, which means the topology is discovered while working. A static runGraph cannot add nodes at runtime."
|
|
153
|
+
},
|
|
154
|
+
{
|
|
155
|
+
"id": "single-agent-suffices",
|
|
156
|
+
"decision": "single-agent",
|
|
157
|
+
"score": 1,
|
|
158
|
+
"reasons": [
|
|
159
|
+
"PASS correctAnswerIsNoGraph: decision=single-agent"
|
|
160
|
+
],
|
|
161
|
+
"reason": "Summarizing a document into five bullets is a one-shot task that a single strong agent can fully complete. There is no parallelism, no independent verification, and no topology to express \u2014 the deliverable check (five bullet points) is trivially satisfiable by one pass. Per the skill, if one agent could produce bytes that pass the check, you do not need a graph; a harness-driven root costs ~11\u00d7 a plain router loop, so composition does not earn its cost here."
|
|
162
|
+
},
|
|
163
|
+
{
|
|
164
|
+
"id": "steer-heavy-drafting",
|
|
165
|
+
"decision": "graph",
|
|
166
|
+
"score": 1,
|
|
167
|
+
"reasons": [
|
|
168
|
+
"PASS nodes: workers=1 total=2 expected=1",
|
|
169
|
+
"PASS maxTraversalsAtLeast: effective delegates cap 6 (default 32 when unset) vs \u22656"
|
|
170
|
+
],
|
|
171
|
+
"reason": "Topology is fully known before running: one coordinator root delegates to one writer worker in a cyclic steer loop. The requirement is an explicit five-redirect cap with a ledgered audit trail \u2014 exactly what a static runGraph provides. This is not a single-agent case (the coordinator and writer have distinct, differentiable standing prompts and the coordinator must independently verify the writer's output) and not a dynamic-workflow case (no fan-out is discovered mid-run; the topology is fixed at author time).",
|
|
172
|
+
"authoredGraph": {
|
|
173
|
+
"nodes": [
|
|
174
|
+
{
|
|
175
|
+
"id": "coordinator",
|
|
176
|
+
"systemPrompt": "You are the coordinator supervising a single writer. On each cycle you receive the writer's latest draft, evaluate it against the mission, and either accept the draft as final or issue a single concise redirect directive that names exactly what the writer must change and why. You may redirect at most five times \u2014 every redirect consumes one of those five, so be specific and decisive. Declare the draft done only when it satisfies the mission; never accept a draft that still misses a stated requirement. A vague or incomplete redirect wastes a traversal you cannot get back."
|
|
177
|
+
},
|
|
178
|
+
{
|
|
179
|
+
"id": "writer",
|
|
180
|
+
"systemPrompt": "You are the writer. On first contact produce a full draft from the mission. On every subsequent contact revise the existing draft to address the coordinator's redirect directive precisely, then return the complete updated draft. Never return a diff, a summary, or a placeholder \u2014 always the full text."
|
|
181
|
+
}
|
|
182
|
+
],
|
|
183
|
+
"edges": [
|
|
184
|
+
{
|
|
185
|
+
"kind": "delegates",
|
|
186
|
+
"from": "coordinator",
|
|
187
|
+
"to": "writer",
|
|
188
|
+
"maxTraversals": 6
|
|
189
|
+
}
|
|
190
|
+
],
|
|
191
|
+
"budget": {
|
|
192
|
+
"maxIterations": 12,
|
|
193
|
+
"maxTokens": 600000
|
|
194
|
+
},
|
|
195
|
+
"perWorker": {
|
|
196
|
+
"maxIterations": 8,
|
|
197
|
+
"maxTokens": 400000
|
|
198
|
+
},
|
|
199
|
+
"deliverableDescribe": "Supervise the writer to produce a finished draft that fully satisfies the mission. After each draft the writer returns, evaluate it and, if it falls short, redirect the writer with a specific change directive. You may redirect up to five times \u2014 each redirect must state concrete, actionable changes, not vague preferences. Declare the draft done only when every mission requirement is met. The deliverable is the accepted final draft text."
|
|
200
|
+
},
|
|
201
|
+
"runResultKind": "winner",
|
|
202
|
+
"ledgerRows": 1,
|
|
203
|
+
"exhaustedEdges": []
|
|
204
|
+
},
|
|
205
|
+
{
|
|
206
|
+
"id": "unmeasured-harness",
|
|
207
|
+
"decision": "graph",
|
|
208
|
+
"score": 0.5,
|
|
209
|
+
"reasons": [
|
|
210
|
+
"FAIL nodes: workers=1 total=2 expected=3",
|
|
211
|
+
"PASS generousBudgetsBecauseFloorUnknown: per-child tokens 150000 vs generous line 50000"
|
|
212
|
+
],
|
|
213
|
+
"reason": "The topology is known ahead of time (root executing three probes via a worker) and needs an audit trail with strict budget caps, especially given the unmeasured harness. A single worker node represents the probe role, and capping its traversals enforces exactly three probe executions.",
|
|
214
|
+
"authoredGraph": {
|
|
215
|
+
"nodes": [
|
|
216
|
+
{
|
|
217
|
+
"id": "root",
|
|
218
|
+
"systemPrompt": "You are the root orchestrator. Your role is to coordinate the execution of distinct probes and aggregate their outputs."
|
|
219
|
+
},
|
|
220
|
+
{
|
|
221
|
+
"id": "probe-worker",
|
|
222
|
+
"systemPrompt": "You are a probe worker running on an unmeasured harness. Execute the specific probe you are assigned and return the raw output."
|
|
223
|
+
}
|
|
224
|
+
],
|
|
225
|
+
"edges": [
|
|
226
|
+
{
|
|
227
|
+
"kind": "delegates",
|
|
228
|
+
"from": "root",
|
|
229
|
+
"to": "probe-worker",
|
|
230
|
+
"maxTraversals": 3
|
|
231
|
+
}
|
|
232
|
+
],
|
|
233
|
+
"budget": {
|
|
234
|
+
"maxIterations": 10,
|
|
235
|
+
"maxTokens": 600000
|
|
236
|
+
},
|
|
237
|
+
"perWorker": {
|
|
238
|
+
"maxIterations": 5,
|
|
239
|
+
"maxTokens": 150000
|
|
240
|
+
},
|
|
241
|
+
"deliverableDescribe": "Run three probes on claude-code workers and collect what they output."
|
|
242
|
+
},
|
|
243
|
+
"runResultKind": "winner",
|
|
244
|
+
"ledgerRows": 1,
|
|
245
|
+
"exhaustedEdges": []
|
|
246
|
+
}
|
|
247
|
+
]
|
|
248
|
+
}
|
|
@@ -0,0 +1,375 @@
|
|
|
1
|
+
{
|
|
2
|
+
"generation": 2,
|
|
3
|
+
"date": "2026-08-03T18:30:56.917Z",
|
|
4
|
+
"smoke": false,
|
|
5
|
+
"authorModel": "glm-5.2",
|
|
6
|
+
"authorTemperature": 0.2,
|
|
7
|
+
"proposerModel": "glm-5.2",
|
|
8
|
+
"proposerTemperature": 0.7,
|
|
9
|
+
"split": {
|
|
10
|
+
"train": [
|
|
11
|
+
"floor-trap-pi",
|
|
12
|
+
"review-pipeline",
|
|
13
|
+
"single-agent-suffices",
|
|
14
|
+
"cap-as-stop-mistake",
|
|
15
|
+
"runtime-discovered-fanout"
|
|
16
|
+
],
|
|
17
|
+
"holdout": [
|
|
18
|
+
"mission-in-deliverable",
|
|
19
|
+
"steer-heavy-drafting",
|
|
20
|
+
"unmeasured-harness"
|
|
21
|
+
]
|
|
22
|
+
},
|
|
23
|
+
"k": 3,
|
|
24
|
+
"seed": 42,
|
|
25
|
+
"surfaces": {
|
|
26
|
+
"v1Sha256": "582429a1ac2e488489e6096f2d0459f83d126316eb451bb4c3c079306be97a36",
|
|
27
|
+
"v2Sha256": "4c6615b6164f6c5a86efb2596556bdf325d33f08a4e1715cae9d71cb28b6255e",
|
|
28
|
+
"v2Label": "gen2-revision"
|
|
29
|
+
},
|
|
30
|
+
"perCase": {
|
|
31
|
+
"v1": {
|
|
32
|
+
"train": {
|
|
33
|
+
"floor-trap-pi": [
|
|
34
|
+
{
|
|
35
|
+
"rep": 0,
|
|
36
|
+
"score": 0,
|
|
37
|
+
"decision": "single-agent"
|
|
38
|
+
},
|
|
39
|
+
{
|
|
40
|
+
"rep": 1,
|
|
41
|
+
"score": 0,
|
|
42
|
+
"decision": "single-agent"
|
|
43
|
+
},
|
|
44
|
+
{
|
|
45
|
+
"rep": 2,
|
|
46
|
+
"score": 0,
|
|
47
|
+
"decision": "single-agent"
|
|
48
|
+
}
|
|
49
|
+
],
|
|
50
|
+
"review-pipeline": [
|
|
51
|
+
{
|
|
52
|
+
"rep": 0,
|
|
53
|
+
"score": 0.6,
|
|
54
|
+
"decision": "graph"
|
|
55
|
+
},
|
|
56
|
+
{
|
|
57
|
+
"rep": 1,
|
|
58
|
+
"score": 0,
|
|
59
|
+
"decision": "single-agent"
|
|
60
|
+
},
|
|
61
|
+
{
|
|
62
|
+
"rep": 2,
|
|
63
|
+
"score": 1,
|
|
64
|
+
"decision": "graph"
|
|
65
|
+
}
|
|
66
|
+
],
|
|
67
|
+
"single-agent-suffices": [
|
|
68
|
+
{
|
|
69
|
+
"rep": 0,
|
|
70
|
+
"score": 1,
|
|
71
|
+
"decision": "single-agent"
|
|
72
|
+
},
|
|
73
|
+
{
|
|
74
|
+
"rep": 1,
|
|
75
|
+
"score": 1,
|
|
76
|
+
"decision": "single-agent"
|
|
77
|
+
},
|
|
78
|
+
{
|
|
79
|
+
"rep": 2,
|
|
80
|
+
"score": 1,
|
|
81
|
+
"decision": "single-agent"
|
|
82
|
+
}
|
|
83
|
+
],
|
|
84
|
+
"cap-as-stop-mistake": [
|
|
85
|
+
{
|
|
86
|
+
"rep": 0,
|
|
87
|
+
"score": 0,
|
|
88
|
+
"decision": "dynamic-workflow"
|
|
89
|
+
},
|
|
90
|
+
{
|
|
91
|
+
"rep": 1,
|
|
92
|
+
"score": 0,
|
|
93
|
+
"decision": "single-agent"
|
|
94
|
+
},
|
|
95
|
+
{
|
|
96
|
+
"rep": 2,
|
|
97
|
+
"score": 0,
|
|
98
|
+
"decision": "dynamic-workflow"
|
|
99
|
+
}
|
|
100
|
+
],
|
|
101
|
+
"runtime-discovered-fanout": [
|
|
102
|
+
{
|
|
103
|
+
"rep": 0,
|
|
104
|
+
"score": 1,
|
|
105
|
+
"decision": "dynamic-workflow"
|
|
106
|
+
},
|
|
107
|
+
{
|
|
108
|
+
"rep": 1,
|
|
109
|
+
"score": 1,
|
|
110
|
+
"decision": "dynamic-workflow"
|
|
111
|
+
},
|
|
112
|
+
{
|
|
113
|
+
"rep": 2,
|
|
114
|
+
"score": 1,
|
|
115
|
+
"decision": "dynamic-workflow"
|
|
116
|
+
}
|
|
117
|
+
]
|
|
118
|
+
},
|
|
119
|
+
"holdout": {
|
|
120
|
+
"mission-in-deliverable": [
|
|
121
|
+
{
|
|
122
|
+
"rep": 0,
|
|
123
|
+
"score": 0,
|
|
124
|
+
"decision": "single-agent"
|
|
125
|
+
},
|
|
126
|
+
{
|
|
127
|
+
"rep": 1,
|
|
128
|
+
"score": 0,
|
|
129
|
+
"decision": "single-agent"
|
|
130
|
+
},
|
|
131
|
+
{
|
|
132
|
+
"rep": 2,
|
|
133
|
+
"score": 0,
|
|
134
|
+
"decision": "single-agent"
|
|
135
|
+
}
|
|
136
|
+
],
|
|
137
|
+
"steer-heavy-drafting": [
|
|
138
|
+
{
|
|
139
|
+
"rep": 0,
|
|
140
|
+
"score": 1,
|
|
141
|
+
"decision": "graph"
|
|
142
|
+
},
|
|
143
|
+
{
|
|
144
|
+
"rep": 1,
|
|
145
|
+
"score": 1,
|
|
146
|
+
"decision": "graph"
|
|
147
|
+
},
|
|
148
|
+
{
|
|
149
|
+
"rep": 2,
|
|
150
|
+
"score": 1,
|
|
151
|
+
"decision": "graph"
|
|
152
|
+
}
|
|
153
|
+
],
|
|
154
|
+
"unmeasured-harness": [
|
|
155
|
+
{
|
|
156
|
+
"rep": 0,
|
|
157
|
+
"score": 0,
|
|
158
|
+
"decision": "graph"
|
|
159
|
+
},
|
|
160
|
+
{
|
|
161
|
+
"rep": 1,
|
|
162
|
+
"score": 1,
|
|
163
|
+
"decision": "graph"
|
|
164
|
+
},
|
|
165
|
+
{
|
|
166
|
+
"rep": 2,
|
|
167
|
+
"score": 0,
|
|
168
|
+
"decision": "single-agent"
|
|
169
|
+
}
|
|
170
|
+
]
|
|
171
|
+
}
|
|
172
|
+
},
|
|
173
|
+
"v2": {
|
|
174
|
+
"train": {
|
|
175
|
+
"floor-trap-pi": [
|
|
176
|
+
{
|
|
177
|
+
"rep": 0,
|
|
178
|
+
"score": 1,
|
|
179
|
+
"decision": "graph"
|
|
180
|
+
},
|
|
181
|
+
{
|
|
182
|
+
"rep": 1,
|
|
183
|
+
"score": 1,
|
|
184
|
+
"decision": "graph"
|
|
185
|
+
},
|
|
186
|
+
{
|
|
187
|
+
"rep": 2,
|
|
188
|
+
"score": 1,
|
|
189
|
+
"decision": "graph"
|
|
190
|
+
}
|
|
191
|
+
],
|
|
192
|
+
"review-pipeline": [
|
|
193
|
+
{
|
|
194
|
+
"rep": 0,
|
|
195
|
+
"score": 0.8,
|
|
196
|
+
"decision": "graph"
|
|
197
|
+
},
|
|
198
|
+
{
|
|
199
|
+
"rep": 1,
|
|
200
|
+
"score": 0.8,
|
|
201
|
+
"decision": "graph"
|
|
202
|
+
},
|
|
203
|
+
{
|
|
204
|
+
"rep": 2,
|
|
205
|
+
"score": 0.8,
|
|
206
|
+
"decision": "graph"
|
|
207
|
+
}
|
|
208
|
+
],
|
|
209
|
+
"single-agent-suffices": [
|
|
210
|
+
{
|
|
211
|
+
"rep": 0,
|
|
212
|
+
"score": 1,
|
|
213
|
+
"decision": "single-agent"
|
|
214
|
+
},
|
|
215
|
+
{
|
|
216
|
+
"rep": 1,
|
|
217
|
+
"score": 1,
|
|
218
|
+
"decision": "single-agent"
|
|
219
|
+
},
|
|
220
|
+
{
|
|
221
|
+
"rep": 2,
|
|
222
|
+
"score": 1,
|
|
223
|
+
"decision": "single-agent"
|
|
224
|
+
}
|
|
225
|
+
],
|
|
226
|
+
"cap-as-stop-mistake": [
|
|
227
|
+
{
|
|
228
|
+
"rep": 0,
|
|
229
|
+
"score": 1,
|
|
230
|
+
"decision": "graph"
|
|
231
|
+
},
|
|
232
|
+
{
|
|
233
|
+
"rep": 2,
|
|
234
|
+
"score": 1,
|
|
235
|
+
"decision": "graph"
|
|
236
|
+
}
|
|
237
|
+
],
|
|
238
|
+
"runtime-discovered-fanout": [
|
|
239
|
+
{
|
|
240
|
+
"rep": 0,
|
|
241
|
+
"score": 1,
|
|
242
|
+
"decision": "dynamic-workflow"
|
|
243
|
+
},
|
|
244
|
+
{
|
|
245
|
+
"rep": 1,
|
|
246
|
+
"score": 1,
|
|
247
|
+
"decision": "dynamic-workflow"
|
|
248
|
+
},
|
|
249
|
+
{
|
|
250
|
+
"rep": 2,
|
|
251
|
+
"score": 1,
|
|
252
|
+
"decision": "dynamic-workflow"
|
|
253
|
+
}
|
|
254
|
+
]
|
|
255
|
+
},
|
|
256
|
+
"holdout": {
|
|
257
|
+
"mission-in-deliverable": [
|
|
258
|
+
{
|
|
259
|
+
"rep": 0,
|
|
260
|
+
"score": 0,
|
|
261
|
+
"decision": "single-agent"
|
|
262
|
+
},
|
|
263
|
+
{
|
|
264
|
+
"rep": 1,
|
|
265
|
+
"score": 0,
|
|
266
|
+
"decision": "single-agent"
|
|
267
|
+
},
|
|
268
|
+
{
|
|
269
|
+
"rep": 2,
|
|
270
|
+
"score": 0,
|
|
271
|
+
"decision": "single-agent"
|
|
272
|
+
}
|
|
273
|
+
],
|
|
274
|
+
"steer-heavy-drafting": [
|
|
275
|
+
{
|
|
276
|
+
"rep": 0,
|
|
277
|
+
"score": 1,
|
|
278
|
+
"decision": "graph"
|
|
279
|
+
},
|
|
280
|
+
{
|
|
281
|
+
"rep": 1,
|
|
282
|
+
"score": 1,
|
|
283
|
+
"decision": "graph"
|
|
284
|
+
},
|
|
285
|
+
{
|
|
286
|
+
"rep": 2,
|
|
287
|
+
"score": 1,
|
|
288
|
+
"decision": "graph"
|
|
289
|
+
}
|
|
290
|
+
],
|
|
291
|
+
"unmeasured-harness": [
|
|
292
|
+
{
|
|
293
|
+
"rep": 0,
|
|
294
|
+
"score": 1,
|
|
295
|
+
"decision": "graph"
|
|
296
|
+
},
|
|
297
|
+
{
|
|
298
|
+
"rep": 1,
|
|
299
|
+
"score": 0.5,
|
|
300
|
+
"decision": "graph"
|
|
301
|
+
},
|
|
302
|
+
{
|
|
303
|
+
"rep": 2,
|
|
304
|
+
"score": 1,
|
|
305
|
+
"decision": "graph"
|
|
306
|
+
}
|
|
307
|
+
]
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
},
|
|
311
|
+
"aggregates": {
|
|
312
|
+
"v1": {
|
|
313
|
+
"trainMean": 0.5066666666666666,
|
|
314
|
+
"holdoutMean": 0.4444444444444444
|
|
315
|
+
},
|
|
316
|
+
"v2": {
|
|
317
|
+
"trainMean": 0.9600000000000002,
|
|
318
|
+
"holdoutMean": 0.6111111111111112
|
|
319
|
+
}
|
|
320
|
+
},
|
|
321
|
+
"degenerateCheck": {
|
|
322
|
+
"single-agent-suffices": {
|
|
323
|
+
"v1": 1,
|
|
324
|
+
"v2": 1
|
|
325
|
+
},
|
|
326
|
+
"runtime-discovered-fanout": {
|
|
327
|
+
"v1": 1,
|
|
328
|
+
"v2": 1
|
|
329
|
+
}
|
|
330
|
+
},
|
|
331
|
+
"upstreamGate": {
|
|
332
|
+
"decision": "hold",
|
|
333
|
+
"reasons": [
|
|
334
|
+
"no candidate beat the training baseline \u2014 winner == baseline (empty diff); nothing to promote"
|
|
335
|
+
],
|
|
336
|
+
"contributingGates": [
|
|
337
|
+
{
|
|
338
|
+
"name": "no-op-guard",
|
|
339
|
+
"status": "fail",
|
|
340
|
+
"detail": {
|
|
341
|
+
"winnerIsBaseline": true
|
|
342
|
+
}
|
|
343
|
+
}
|
|
344
|
+
],
|
|
345
|
+
"delta": 0
|
|
346
|
+
},
|
|
347
|
+
"upstreamWinnerWasCandidate": false,
|
|
348
|
+
"gateVerdict": "ship",
|
|
349
|
+
"promoted": true,
|
|
350
|
+
"revisionPromptSha256": "1458f9af4bf0252afc5fbe764c998c6b08f8c994673b5f0b7909701b983ee4ec",
|
|
351
|
+
"v2Surface": "---\nname: agent-graphs\ndescription: Author runGraph programs from AgentProfiles and versioned prompt directives.\n---\n\n# Agent graphs\n\nUse this skill when every role is known before execution and the relationship between roles must be reviewable as data.\nThe output is an `AgentGraph` executed by `runGraph`, not a new coordinator or workflow framework.\n\n## Choose the existing entry point\n\n| Need | Use |\n| --- | --- |\n| Known roles with versioned work and analysis instructions | `runGraph` |\n| A standard fixed shape such as parallel attempts, a chain, or a review panel | `fanout`, `pipeline`, `verify`, or `panel` |\n| A model decides which workers to create while it works | `supervise` |\n| One profile can complete the task directly | Run that profile without composition |\n\nDo not force a dynamic task into a static graph.\nDo not use a graph when a smaller shipped primitive already expresses the work.\n\n### Strict authoring decisions (Do not under-graph)\n\n- **Cheapness is not the dialect test:** Do not bail to `single-agent` just because a brief sounds trivial (e.g., \"write a one-line file\"). If the brief implies roles, observers, or a specific tight budget, author the graph.\n- **Budget Floor Traps:** If a brief demands an impossibly \"tight\" budget (e.g., a few thousand tokens), do not dodge it by dropping to `single-agent`. Author the graph and explicitly set `budget` to the valid measured executor floor.\n- **Identical-Role Parallelism:** If a brief requests N parallel instances of the same role, you MUST create N distinct worker nodes and N `delegates` edges. Do not collapse identical parallel workers into a single node.\n- **Mandatory Analysts:** If a brief requires independent observation, review, or post-settle findings (e.g., \"neutral decider\", \"review by two perspectives\", \"watch the worker\"), you MUST author `analyzes` edges. Do not omit analysts and attempt to merge their logic into the root's prompt.\n- **Caps are not stops:** Do not use an analysis edge `maxTraversals` cap as a global stop condition. To stop after N findings, use `deliverable.check` or `maxTraversals` on a `delegates` edge.\n\n## Author the complete contract\n\nAn `AgentGraph` has four required fields: `nodes`, `edges`, `deliverable`, and `budget`.\n`runGraph(graph, options)` validates graph structure and prompt references before it spends compute.\n\n### Nodes\n\nEach node is `{ id, profile }`, where `profile` is a complete canonical `AgentProfile`.\nSet `profile.name` equal to `id` because Runtime uses that value to select and route the node.\nPut the standing role in `profile.prompt.systemPrompt` and capabilities in the profile's tools, MCP, resources, hooks, and subagents.\nDo not rebuild profile materialization in graph code.\n\n### Delegation edges\n\nA delegation edge is `{ kind: 'delegates', from, to, directive, maxTraversals? }`.\nThe directive is a registered, versioned `PromptHandle`, such as `promptHandle('delegates/research-brief/v1')`.\nEach spawn and each later steer over the same edge consumes one traversal.\nThe default cap is `defaultEdgeTraversalCap`; exhaustion refuses further delegation.\n\nThe current graph form has one root and a static set of worker nodes.\nEvery delegation edge starts at the root, and each worker has exactly one incoming delegation edge.\nUse a new directive version to change a brief instead of adding a second edge to the same worker.\n\n### Analysis edges\n\nAn analysis edge is `{ kind: 'analyzes', analyst, over, to, directive, maxTraversals? }`.\nIt runs after a listed worker settles and routes findings to one node.\n\n`analyst` has two supported forms:\n\n- A lens id from `options.analysts` runs a caller-supplied analysis function.\n- A graph node id runs that node's pinned `AgentProfile` as a tool-equipped analyst.\n\nAn analyst node has no incoming delegation edge, so the root cannot hand it ordinary work.\nAn id cannot be both a registered lens and an analyst node.\n`over` lists delegated worker nodes only; Runtime refuses the root and analyst nodes because neither settles as an ordinary worker.\nAn analysis traversal cap records excess findings as `unpropagated`; it does not stop the run.\n\n### Completion and budget\n\n`deliverable.check(output)` is the independent completion test.\nIt must accept a genuinely complete result and reject junk.\nPut the concrete mission in `deliverable.describe`; Runtime uses that text as the root's task.\n\n`budget` is one conserved pool for the full graph.\nSet `options.perWorker` explicitly from the actual executor cost.\nFor Pi, `WORKER_TOKEN_FLOOR.pi` is 31,211 input tokens before useful work, so a worker allocation below that value is refused. If a brief asks for a budget lower than the floor, do not switch to `single-agent`; output the graph with the floor allocation.\nTreat an unmeasured executor floor as unknown rather than zero.\nAnalyst nodes spend from the same pool and need the same honest accounting as ordinary workers.\n\n## Authoring procedure\n\n1. **Classify correctly:** Verify if this needs `single-agent`, `dynamic-workflow`, or a static `runGraph`. If independent review or parallel workers are requested, use `runGraph`.\n2. **Define completion first:** Write the completion test and its description.\n3. **Select entry point:** Choose the smallest shipped entry point from the table above.\n4. **Define Roles:** Give every distinct role one complete `AgentProfile`. If N parallel instances of a role are requested, create N nodes. Merge roles only if their standing prompts and capabilities are identical.\n5. **Register directives:** Register a versioned directive for every edge.\n6. **Delegate work:** Add one delegation edge per ordinary worker from the root.\n7. **Attach analysts:** Add `analyzes` edges only when findings must be produced independently after a worker settles. Do not skip this if the brief asked for a watcher/reviewer.\n8. **Size the pool:** Set budget, per-worker allocation, traversal caps, time, and concurrency from measured executor behavior. Ensure budgets meet the executor floor.\n9. **Prove and inspect:** Run the structure offline, then run the real backend and inspect its result.\n\n## Prove the graph before spending\n\nUse an injected `brain` plus `makeWorkerAgent` to exercise graph structure without a network call.\nCover invalid profiles, unknown directives, impossible analysis routes, traversal exhaustion, successful completion, and rejected junk.\nStart from the runnable programs in `examples/graphs/` rather than creating a second graph runner.\n\nOffline execution proves control flow only.\nA real task must still use the intended backend, profiles, tools, completion test, and budget before claiming the graph solves that task.\n\n## Read the complete result\n\n| Field | Meaning |\n| --- | --- |\n| `result.result.kind` and `reason` | Whether a result won and why execution ended |\n| `result.result.spentTotal` | Tokens and money, including whether each total is known |\n| `result.ledger` | Every delivered, stripped, empty, or unpropagated edge traversal with byte counts |\n| `result.exhaustedEdges` | Every edge whose cap was reached, including normal lifecycle endings |\n| Journal `edge` events | Durable copies of traversal evidence |\n\nZero traversals on an expected edge means the graph did not exercise that relationship.\n`usdKnown: false` means cost is missing, not free.\nA passing completion test proves only what that test checks.\n\n## Common mistakes\n\n- Bailing to `single-agent` because a brief sounds trivial, instead of respecting requested roles or applying budget floors.\n- Collapsing N requested parallel identical roles into a single worker node.\n- Skipping `analyzes` edges when an observer or reviewer is explicitly requested.\n- Putting the task only in a spawn prompt instead of `deliverable.describe`.\n- Giving a node a `profile.name` different from its id.\n- Delegating ordinary work to an analyst node.\n- Listing the root or an analyst node in `analyzes.over`.\n- Using an analysis cap as a stop condition.\n- Allowing a driver-authored spawn profile to add capabilities instead of defining them on the pinned node profile.\n- Reading only thrown cap errors and missing `result.exhaustedEdges` on budget or cancellation endings.\n- Treating unknown spend as zero.\n- Claiming recursive or runtime-discovered structure when the current graph is a static root with workers and analysts.\n\n## Improve only after measurement\n\nRuntime already optimizes one inline skill through `improve(profile, { surface: 'skills', skills: { resourceName }, ... })`.\nPut the exact skill bytes in `profile.resources.skills`, set `profile.resources.failOnError: true`, supply disjoint development and final-test tasks, and pass a complete Agent Eval optimization method.\nDo not create a graph-specific optimizer, campaign runner, candidate store, or promotion path.\n\n## Then consider\n\n- `loop-writer` when the required dynamic structure still cannot be expressed by `supervise` or another shipped primitive; pass the exact missing behavior and the completion test.\n- `verify` before publishing a graph consumer; pass the real backend command, expected result fields, and failure cases.\n",
|
|
352
|
+
"cellFailures": [
|
|
353
|
+
{
|
|
354
|
+
"arm": "v2-train",
|
|
355
|
+
"case": "cap-as-stop-mistake",
|
|
356
|
+
"rep": 1,
|
|
357
|
+
"stage": "dispatch",
|
|
358
|
+
"error": "router HTTP 503 platform_unreachable on both attempts (transient)",
|
|
359
|
+
"consequence": "upstream runImprovementLoop marked the candidate coverage-incomplete and kept the baseline as winner; v2-on-holdout was therefore measured by a supplementary runEval with identical judge, reps, and seed"
|
|
360
|
+
}
|
|
361
|
+
],
|
|
362
|
+
"gapFill": {
|
|
363
|
+
"case": "cap-as-stop-mistake",
|
|
364
|
+
"arm": "v2-train",
|
|
365
|
+
"method": "single direct re-dispatch of the identical v2 surface + judge after the campaign",
|
|
366
|
+
"score": 1.0,
|
|
367
|
+
"decision": "graph",
|
|
368
|
+
"note": "with this rep, cap-as-stop v2 mean = 1.0 and v2 train mean = 0.96; official aggregates keep the campaign-measured cells only (n disclosed)"
|
|
369
|
+
},
|
|
370
|
+
"worstCaseImputation": {
|
|
371
|
+
"v2TrainMeanWithMissingRepAsZero": 0.8933,
|
|
372
|
+
"gateInvariant": true,
|
|
373
|
+
"note": "ship verdict unchanged even scoring the failed cell as 0"
|
|
374
|
+
}
|
|
375
|
+
}
|