@tangle-network/agent-runtime 0.126.0 → 0.131.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (146) hide show
  1. package/README.md +70 -20
  2. package/dist/{activation-DhWJ3p8N.js → activation-BCzMOTaV.js} +3 -3
  3. package/dist/{activation-DhWJ3p8N.js.map → activation-BCzMOTaV.js.map} +1 -1
  4. package/dist/agent.d.ts +2 -3
  5. package/dist/agent.js +4 -5
  6. package/dist/agent.js.map +1 -1
  7. package/dist/{analyst-loop-DvSciOfB.js → analyst-loop-BE8cDs5Q.js} +2 -2
  8. package/dist/{analyst-loop-DvSciOfB.js.map → analyst-loop-BE8cDs5Q.js.map} +1 -1
  9. package/dist/analyst-loop.d.ts +1 -1
  10. package/dist/analyst-loop.js +1 -1
  11. package/dist/authoring-CvHwo1oW.js +163 -0
  12. package/dist/authoring-CvHwo1oW.js.map +1 -0
  13. package/dist/candidate-execution/index.js +4 -4
  14. package/dist/{candidate-execution-BFpq-Xi6.js → candidate-execution-BNxKr-Bu.js} +5 -5
  15. package/dist/{candidate-execution-BFpq-Xi6.js.map → candidate-execution-BNxKr-Bu.js.map} +1 -1
  16. package/dist/{conversation-BpLQZGPH.js → conversation-DNtxaJ1Z.js} +113 -31
  17. package/dist/conversation-DNtxaJ1Z.js.map +1 -0
  18. package/dist/conversation.d.ts +2 -2
  19. package/dist/conversation.js +2 -2
  20. package/dist/environment-provider-CxvSd1W6.d.ts +86 -0
  21. package/dist/{environment-provider-DqFS6FSZ.js → environment-provider-Dyg8DtLK.js} +8 -4
  22. package/dist/environment-provider-Dyg8DtLK.js.map +1 -0
  23. package/dist/environment-provider.d.ts +1 -1
  24. package/dist/environment-provider.js +1 -1
  25. package/dist/graph-BJTxGOFB.js +471 -0
  26. package/dist/graph-BJTxGOFB.js.map +1 -0
  27. package/dist/{improvement-cycle-IJgCbKWQ.js → improvement-cycle-Csp38cWg.js} +133 -224
  28. package/dist/improvement-cycle-Csp38cWg.js.map +1 -0
  29. package/dist/{index-EdjCQBV9.d.ts → index-CoO7atyo.d.ts} +640 -1184
  30. package/dist/{index-DIV33AF5.d.ts → index-DwGtu9nc.d.ts} +7 -9
  31. package/dist/{index-Efjb3nrQ.d.ts → index-qYHpsmG2.d.ts} +36 -18
  32. package/dist/index.d.ts +353 -11
  33. package/dist/index.js +111 -354
  34. package/dist/index.js.map +1 -1
  35. package/dist/intelligence.d.ts +6 -6
  36. package/dist/intelligence.js +9 -8
  37. package/dist/intelligence.js.map +1 -1
  38. package/dist/kernel.d.ts +7 -5
  39. package/dist/kernel.js +13 -9
  40. package/dist/{knowledge-EnuEqm_Y.js → knowledge-ce0_uKCl.js} +19 -17
  41. package/dist/knowledge-ce0_uKCl.js.map +1 -0
  42. package/dist/knowledge.d.ts +1 -1
  43. package/dist/knowledge.js +1 -1
  44. package/dist/{loop-runner-bin-Bo29_fiD.d.ts → loop-runner-bin-BwgZ8m_7.d.ts} +6 -3
  45. package/dist/{loop-runner-bin-qwT_4F5I.js → loop-runner-bin-DSbuDDqM.js} +5 -27
  46. package/dist/loop-runner-bin-DSbuDDqM.js.map +1 -0
  47. package/dist/loop-runner-bin.d.ts +1 -1
  48. package/dist/loop-runner-bin.js +1 -1
  49. package/dist/materialization-COJ1UYQ-.js +272 -0
  50. package/dist/materialization-COJ1UYQ-.js.map +1 -0
  51. package/dist/mcp/bin.js +39 -47
  52. package/dist/mcp/bin.js.map +1 -1
  53. package/dist/mcp/index.d.ts +24 -30
  54. package/dist/mcp/index.js +66 -83
  55. package/dist/mcp/index.js.map +1 -1
  56. package/dist/mcp/memory-bin.js +1 -1
  57. package/dist/{memory-server-DL6cE2Ag.js → memory-server-5HEJH672.js} +2 -2
  58. package/dist/{memory-server-DL6cE2Ag.js.map → memory-server-5HEJH672.js.map} +1 -1
  59. package/dist/model-policy-CqziaqS1.js +232 -0
  60. package/dist/model-policy-CqziaqS1.js.map +1 -0
  61. package/dist/{openai-tools-B68JaOCx.d.ts → openai-tools-D3oMyVGl.d.ts} +2 -2
  62. package/dist/{openai-tools-Bp1KSkP6.js → openai-tools-ru75mLjq.js} +2 -2
  63. package/dist/openai-tools-ru75mLjq.js.map +1 -0
  64. package/dist/{prepare--8EvLqCr.js → prepare-DYWjVcPx.js} +169 -153
  65. package/dist/prepare-DYWjVcPx.js.map +1 -0
  66. package/dist/primeintellect/index.d.ts +7 -6
  67. package/dist/primeintellect/index.js +9 -11
  68. package/dist/primeintellect/index.js.map +1 -1
  69. package/dist/profiles.d.ts +21 -174
  70. package/dist/profiles.js +67 -276
  71. package/dist/profiles.js.map +1 -1
  72. package/dist/{protected-model-port-CXVfOUu_.js → protected-model-port-48ALLGxT.js} +2 -2
  73. package/dist/{protected-model-port-CXVfOUu_.js.map → protected-model-port-48ALLGxT.js.map} +1 -1
  74. package/dist/{redact-PQzmE1Jn.d.ts → redact-9_Gf-8m3.d.ts} +58 -69
  75. package/dist/{researcher-CoVqNhfI.js → researcher-Skz5-Uc8.js} +50 -26
  76. package/dist/researcher-Skz5-Uc8.js.map +1 -0
  77. package/dist/{run-layout-C2FGmZ3v.js → run-layout-CeJsEAom.js} +2 -2
  78. package/dist/{run-layout-C2FGmZ3v.js.map → run-layout-CeJsEAom.js.map} +1 -1
  79. package/dist/runtime-D-QfLbSd.d.ts +893 -0
  80. package/dist/{runtime-BzXz7OjS.js → runtime-hiAABiTk.js} +329 -854
  81. package/dist/runtime-hiAABiTk.js.map +1 -0
  82. package/dist/{sandbox-events-Yhd1GYWl.js → sandbox-events-CRDwc5WN.js} +42 -5
  83. package/dist/sandbox-events-CRDwc5WN.js.map +1 -0
  84. package/dist/snapshot-CXiiuHhL.js +21 -0
  85. package/dist/snapshot-CXiiuHhL.js.map +1 -0
  86. package/dist/{spawn-journal-DsZKDqeh.js → spawn-journal-saHQzqYi.js} +15 -21
  87. package/dist/spawn-journal-saHQzqYi.js.map +1 -0
  88. package/dist/stream-agent-turn-C852AgMT.d.ts +128 -0
  89. package/dist/stream-agent-turn-rYgaOLO0.js +910 -0
  90. package/dist/stream-agent-turn-rYgaOLO0.js.map +1 -0
  91. package/dist/{structural-rollout-zY0oqQzO.js → structural-rollout-3uxVGcg2.js} +459 -235
  92. package/dist/structural-rollout-3uxVGcg2.js.map +1 -0
  93. package/dist/{supervise-Ds8FtyI9.js → supervise-iPN27pO0.js} +932 -4771
  94. package/dist/supervise-iPN27pO0.js.map +1 -0
  95. package/dist/{supervisor-DpjO0Gmy.js → supervisor-CV6Jh28D.js} +7439 -2897
  96. package/dist/supervisor-CV6Jh28D.js.map +1 -0
  97. package/dist/testing.d.ts +3 -1
  98. package/dist/testing.js +271 -221
  99. package/dist/testing.js.map +1 -1
  100. package/dist/{top-app-5unxqovu.js → top-app-G4M18b6i.js} +3 -3
  101. package/dist/{top-app-5unxqovu.js.map → top-app-G4M18b6i.js.map} +1 -1
  102. package/dist/tui/bin.js +1 -1
  103. package/dist/tui/index.js +1 -1
  104. package/dist/{environment-provider-PM9PeW_J.d.ts → types-C6Q-J0Dt.d.ts} +57 -114
  105. package/dist/{types-DnNGJ5Gz.d.ts → types-ebIY0dMG.d.ts} +556 -30
  106. package/dist/{util-MVgdwuIS.js → util-Bw6srryQ.js} +3 -2
  107. package/dist/{util-MVgdwuIS.js.map → util-Bw6srryQ.js.map} +1 -1
  108. package/dist/{workspace-archive-BQxvkypI.js → workspace-archive-aOfJ47ms.js} +3 -3
  109. package/dist/{workspace-archive-BQxvkypI.js.map → workspace-archive-aOfJ47ms.js.map} +1 -1
  110. package/package.json +12 -15
  111. package/skills/agent-graphs/IMPROVE.md +58 -0
  112. package/skills/agent-graphs/SKILL.md +139 -0
  113. package/skills/agent-graphs/cases/artifact-mission-release-notes.json +10 -0
  114. package/skills/agent-graphs/cases/audited-single-writer.json +9 -0
  115. package/skills/agent-graphs/cases/cap-as-stop-mistake.json +8 -0
  116. package/skills/agent-graphs/cases/mission-in-deliverable.json +8 -0
  117. package/skills/agent-graphs/cases/review-pipeline.json +13 -0
  118. package/skills/agent-graphs/cases/runtime-discovered-fanout.json +8 -0
  119. package/skills/agent-graphs/cases/single-agent-suffices.json +7 -0
  120. package/skills/agent-graphs/cases/steer-heavy-drafting.json +9 -0
  121. package/skills/agent-graphs/cases/unmeasured-harness.json +7 -0
  122. package/skills/agent-graphs/generations/gen1-baseline.json +248 -0
  123. package/skills/agent-graphs/generations/gen2.json +375 -0
  124. package/skills/agent-graphs/generations/gen3.json +702 -0
  125. package/skills/build-with-agent-runtime/SKILL.md +1 -0
  126. package/dist/backends-CiOCyRHb.js +0 -743
  127. package/dist/backends-CiOCyRHb.js.map +0 -1
  128. package/dist/conversation-BpLQZGPH.js.map +0 -1
  129. package/dist/environment-provider-DqFS6FSZ.js.map +0 -1
  130. package/dist/improvement-cycle-IJgCbKWQ.js.map +0 -1
  131. package/dist/index-D_M4d1_B.d.ts +0 -545
  132. package/dist/knowledge-EnuEqm_Y.js.map +0 -1
  133. package/dist/local-harness-BIajef4A.d.ts +0 -465
  134. package/dist/loop-runner-bin-qwT_4F5I.js.map +0 -1
  135. package/dist/model-resolution-Btd9iIKV.js +0 -98
  136. package/dist/model-resolution-Btd9iIKV.js.map +0 -1
  137. package/dist/openai-tools-Bp1KSkP6.js.map +0 -1
  138. package/dist/prepare--8EvLqCr.js.map +0 -1
  139. package/dist/researcher-CoVqNhfI.js.map +0 -1
  140. package/dist/runtime-BzXz7OjS.js.map +0 -1
  141. package/dist/sandbox-events-Yhd1GYWl.js.map +0 -1
  142. package/dist/spawn-journal-DsZKDqeh.js.map +0 -1
  143. package/dist/structural-rollout-zY0oqQzO.js.map +0 -1
  144. package/dist/supervise-Ds8FtyI9.js.map +0 -1
  145. package/dist/supervisor-DpjO0Gmy.js.map +0 -1
  146. package/dist/types-C9j4qg6l.d.ts +0 -500
@@ -0,0 +1,10 @@
1
+ {
2
+ "id": "artifact-mission-release-notes",
3
+ "brief": "Produce a release-notes file for version 2.0 that passes our repo format checker.",
4
+ "expect": {
5
+ "correctAnswerIsGraph": true,
6
+ "nodes": 1,
7
+ "deliverableDescribeCarriesMission": true,
8
+ "checkIsMechanical": true
9
+ }
10
+ }
@@ -0,0 +1,9 @@
1
+ {
2
+ "id": "audited-single-writer",
3
+ "brief": "Have a separate worker write the summary document so I can audit exactly what it did afterwards.",
4
+ "expect": {
5
+ "correctAnswerIsGraph": true,
6
+ "nodes": 1,
7
+ "reason": "auditability warrants delegation despite a cheap-sounding task"
8
+ }
9
+ }
@@ -0,0 +1,8 @@
1
+ {
2
+ "id": "cap-as-stop-mistake",
3
+ "brief": "Have an analyst watch the worker and stop the whole thing after three findings.",
4
+ "expect": {
5
+ "trapIsAnalyzesCapAsStop": true,
6
+ "correctStopIsDelegatesCapOrDeliverable": true
7
+ }
8
+ }
@@ -0,0 +1,8 @@
1
+ {
2
+ "id": "mission-in-deliverable",
3
+ "brief": "Produce a CHANGELOG entry for last week that passes our format check.",
4
+ "expect": {
5
+ "deliverableDescribeCarriesMission": true,
6
+ "checkIsMechanical": true
7
+ }
8
+ }
@@ -0,0 +1,13 @@
1
+ {
2
+ "id": "review-pipeline",
3
+ "brief": "I want code changes reviewed by two different perspectives before anything merges, and someone neutral deciding.",
4
+ "expect": {
5
+ "nodes": 3,
6
+ "analyzesWarranted": true,
7
+ "edges": [
8
+ "delegates to each reviewer",
9
+ "analyzes routing findings to root"
10
+ ],
11
+ "wrongIfAnalystIsNode": true
12
+ }
13
+ }
@@ -0,0 +1,8 @@
1
+ {
2
+ "id": "runtime-discovered-fanout",
3
+ "brief": "Find every failing test in the repo and fix each one in parallel.",
4
+ "expect": {
5
+ "correctAnswerIsDynamicWorkflow": true,
6
+ "reason": "topology discovered mid-run"
7
+ }
8
+ }
@@ -0,0 +1,7 @@
1
+ {
2
+ "id": "single-agent-suffices",
3
+ "brief": "Summarize this document into five bullets.",
4
+ "expect": {
5
+ "correctAnswerIsNoGraph": true
6
+ }
7
+ }
@@ -0,0 +1,9 @@
1
+ {
2
+ "id": "steer-heavy-drafting",
3
+ "brief": "One writer drafts, I want the coordinator to redirect it up to five times based on how the draft evolves.",
4
+ "expect": {
5
+ "nodes": 1,
6
+ "maxTraversalsAtLeast": 6,
7
+ "reason": "spawns and steers share the traversal count"
8
+ }
9
+ }
@@ -0,0 +1,7 @@
1
+ {
2
+ "id": "unmeasured-harness",
3
+ "brief": "Run three probes on claude-code workers and collect what they output.",
4
+ "expect": {
5
+ "nodes": 3
6
+ }
7
+ }
@@ -0,0 +1,248 @@
1
+ {
2
+ "skillVersion": "v1",
3
+ "surfaceSource": "working-tree",
4
+ "authorModel": "glm-5.2",
5
+ "temperature": 0.2,
6
+ "date": "2026-08-03T17:42:45.828Z",
7
+ "n": 8,
8
+ "aggregate": {
9
+ "mean": 0.6375,
10
+ "median": 0.8,
11
+ "min": 0,
12
+ "max": 1,
13
+ "n": 8,
14
+ "correction": "first publication reported median 1.0 from an upper-middle indexing bug; corrected to the interpolated median",
15
+ "samplePerCase": 1,
16
+ "varianceNote": "decision-level variance observed at temp 0.2 (cap-as-stop-mistake flipped graph->single-agent on rerun); treat per-case scores as n=1 draws, not stable properties. Reps belong to the v2 gate round."
17
+ },
18
+ "cases": [
19
+ {
20
+ "id": "cap-as-stop-mistake",
21
+ "decision": "graph",
22
+ "score": 1,
23
+ "reasons": [
24
+ "PASS trapIsAnalyzesCapAsStop: analyzes caps=true delegates caps=true",
25
+ "PASS correctStopIsDelegatesCapOrDeliverable: delegates caps=true, reason mentions deliverable=true"
26
+ ],
27
+ "reason": "The brief asks for an analyst to watch a worker \u2014 a legitimate unaddressable observation lens \u2014 so a static graph is the right dialect. But the brief's stated stop mechanism is the textbook pothole: an analyzes edge's maxTraversals is observability-only; on exhaustion the traversal is ledgered 'unpropagated' and it never refuses, so it cannot terminate a run. Only a delegates cap or the deliverable check can. I therefore encode 'three findings' in the deliverable (the real termination oracle) and let the worker settle once three findings exist; the findings-lens analyst observes and routes to the root with maxTraversals=3 as the intended observation budget, but it is NOT load-bearing for termination. If the brief truly needs hard termination at exactly three analyst findings regardless of worker output, that is not expressible in the current P0 topology and must be re-scoped before running.",
28
+ "authoredGraph": {
29
+ "nodes": [
30
+ {
31
+ "id": "root",
32
+ "systemPrompt": "You are the root orchestrator. Hold the mission: produce at least three distinct findings on the delegated subject. Delegate to the worker, accept the findings-lens observations routed back to you, and stop spawning the moment the worker's settled output contains three findings. Do not delegate further once three are present."
33
+ },
34
+ {
35
+ "id": "worker",
36
+ "systemPrompt": "You are the findings worker. Investigate the delegated brief and emit discrete, numbered findings, one paragraph each. Continue until you have produced at least three distinct findings, then settle. Do not pad or repeat."
37
+ }
38
+ ],
39
+ "edges": [
40
+ {
41
+ "kind": "delegates",
42
+ "from": "root",
43
+ "to": "worker",
44
+ "maxTraversals": 6
45
+ },
46
+ {
47
+ "kind": "analyzes",
48
+ "analyst": "findings-lens",
49
+ "over": [
50
+ "worker"
51
+ ],
52
+ "to": "root",
53
+ "maxTraversals": 3
54
+ }
55
+ ],
56
+ "budget": {
57
+ "maxIterations": 8,
58
+ "maxTokens": 240000
59
+ },
60
+ "perWorker": {
61
+ "maxIterations": 4,
62
+ "maxTokens": 80000
63
+ },
64
+ "deliverableDescribe": "Investigate the delegated subject and report at least three distinct, numbered findings. The run terminates when the worker's settled output contains three findings; the findings-lens analyst observes the worker and routes its observations to the root (observation only \u2014 it does not stop the run)."
65
+ },
66
+ "runResultKind": "winner",
67
+ "ledgerRows": 2,
68
+ "exhaustedEdges": []
69
+ },
70
+ {
71
+ "id": "floor-trap-pi",
72
+ "decision": "single-agent",
73
+ "score": 0,
74
+ "reasons": [
75
+ "FAIL mustBudgetAtLeast: no graph authored (decision=single-agent)",
76
+ "FAIL nodes: no graph authored",
77
+ "FAIL analyzesWarranted: no graph authored"
78
+ ],
79
+ "reason": "The task is trivially one-shot-able by a single pi agent (writing a one-line file) without requiring parallel fan-out or independent verification. Furthermore, the brief's tight budget constraint of a few thousand tokens falls drastically below the measured 31,211 input token floor (`WORKER_TOKEN_FLOOR`) required for a `runGraph` spawn, meaning any graph attempt would fail with a `below-runtime-floor` refusal."
80
+ },
81
+ {
82
+ "id": "mission-in-deliverable",
83
+ "decision": "single-agent",
84
+ "score": 0,
85
+ "reasons": [
86
+ "FAIL deliverableDescribeCarriesMission: no graph authored"
87
+ ],
88
+ "reason": "A single agent can author a CHANGELOG entry that satisfies a format check in a single pass, requiring no topology, parallelism, or independent verification."
89
+ },
90
+ {
91
+ "id": "review-pipeline",
92
+ "decision": "graph",
93
+ "score": 0.6,
94
+ "reasons": [
95
+ "PASS nodes: workers=2 total=3 expected=3",
96
+ "FAIL analyzesWarranted: analyzes edges=0 warranted=true",
97
+ "PASS wrongIfAnalystIsNode: no analyst id collides with a node id",
98
+ "PASS edge:delegates to each reviewer: workers [security-reviewer, maintainability-reviewer] each delegated-to with >0 ledger traversals: true",
99
+ "FAIL edge:analyzes routing findings to root: analyzes\u2192root edges=0, fired with >0 traversals=false"
100
+ ],
101
+ "reason": "The case calls for two genuinely distinct review perspectives (security/correctness vs maintainability) that must settle independently, then a neutral arbiter synthesizing both \u2014 a fixed topology with independent verification that earns composition. One agent cannot maintain two addressable, independently-settled roles plus a neutral third.",
102
+ "authoredGraph": {
103
+ "nodes": [
104
+ {
105
+ "id": "merge-arbiter",
106
+ "systemPrompt": "You are the neutral merge arbiter. You delegate code changes to two independent reviewers \u2014 one for security/correctness, one for maintainability \u2014 then collect both reviews and render a final merge decision: APPROVE, REQUEST-CHANGES, or REJECT. You must cite specific findings from each reviewer in your decision rationale. You do not review code yourself; you synthesize and decide."
107
+ },
108
+ {
109
+ "id": "security-reviewer",
110
+ "systemPrompt": "You are a security and correctness reviewer. Examine code changes for vulnerabilities, input-validation gaps, race conditions, data-integrity risks, logic errors, and unhandled edge cases. Report each finding with a severity rating (critical, high, medium, low) and a concrete location in the diff. Do not comment on style or naming unless it introduces a correctness risk."
111
+ },
112
+ {
113
+ "id": "maintainability-reviewer",
114
+ "systemPrompt": "You are a maintainability and clarity reviewer. Examine code changes for naming quality, structural clarity, test coverage, documentation gaps, convention adherence, and future changeability. Report each finding with a priority rating (blocker, important, minor) and a concrete location in the diff. Do not comment on security unless a pattern directly blocks maintainability."
115
+ }
116
+ ],
117
+ "edges": [
118
+ {
119
+ "kind": "delegates",
120
+ "from": "merge-arbiter",
121
+ "to": "security-reviewer",
122
+ "maxTraversals": 2
123
+ },
124
+ {
125
+ "kind": "delegates",
126
+ "from": "merge-arbiter",
127
+ "to": "maintainability-reviewer",
128
+ "maxTraversals": 2
129
+ }
130
+ ],
131
+ "budget": {
132
+ "maxIterations": 6,
133
+ "maxTokens": 300000
134
+ },
135
+ "perWorker": {
136
+ "maxIterations": 3,
137
+ "maxTokens": 80000
138
+ },
139
+ "deliverableDescribe": "Review the provided code change from two independent perspectives \u2014 security/correctness and maintainability \u2014 then produce a merge decision (APPROVE, REQUEST-CHANGES, or REJECT) with a rationale that cites specific findings from each reviewer."
140
+ },
141
+ "runResultKind": "winner",
142
+ "ledgerRows": 2,
143
+ "exhaustedEdges": []
144
+ },
145
+ {
146
+ "id": "runtime-discovered-fanout",
147
+ "decision": "dynamic-workflow",
148
+ "score": 1,
149
+ "reasons": [
150
+ "PASS correctAnswerIsDynamicWorkflow: decision=dynamic-workflow"
151
+ ],
152
+ "reason": "The task requires identifying an unknown number of failing tests at runtime and fixing them in parallel, which means the topology is discovered while working. A static runGraph cannot add nodes at runtime."
153
+ },
154
+ {
155
+ "id": "single-agent-suffices",
156
+ "decision": "single-agent",
157
+ "score": 1,
158
+ "reasons": [
159
+ "PASS correctAnswerIsNoGraph: decision=single-agent"
160
+ ],
161
+ "reason": "Summarizing a document into five bullets is a one-shot task that a single strong agent can fully complete. There is no parallelism, no independent verification, and no topology to express \u2014 the deliverable check (five bullet points) is trivially satisfiable by one pass. Per the skill, if one agent could produce bytes that pass the check, you do not need a graph; a harness-driven root costs ~11\u00d7 a plain router loop, so composition does not earn its cost here."
162
+ },
163
+ {
164
+ "id": "steer-heavy-drafting",
165
+ "decision": "graph",
166
+ "score": 1,
167
+ "reasons": [
168
+ "PASS nodes: workers=1 total=2 expected=1",
169
+ "PASS maxTraversalsAtLeast: effective delegates cap 6 (default 32 when unset) vs \u22656"
170
+ ],
171
+ "reason": "Topology is fully known before running: one coordinator root delegates to one writer worker in a cyclic steer loop. The requirement is an explicit five-redirect cap with a ledgered audit trail \u2014 exactly what a static runGraph provides. This is not a single-agent case (the coordinator and writer have distinct, differentiable standing prompts and the coordinator must independently verify the writer's output) and not a dynamic-workflow case (no fan-out is discovered mid-run; the topology is fixed at author time).",
172
+ "authoredGraph": {
173
+ "nodes": [
174
+ {
175
+ "id": "coordinator",
176
+ "systemPrompt": "You are the coordinator supervising a single writer. On each cycle you receive the writer's latest draft, evaluate it against the mission, and either accept the draft as final or issue a single concise redirect directive that names exactly what the writer must change and why. You may redirect at most five times \u2014 every redirect consumes one of those five, so be specific and decisive. Declare the draft done only when it satisfies the mission; never accept a draft that still misses a stated requirement. A vague or incomplete redirect wastes a traversal you cannot get back."
177
+ },
178
+ {
179
+ "id": "writer",
180
+ "systemPrompt": "You are the writer. On first contact produce a full draft from the mission. On every subsequent contact revise the existing draft to address the coordinator's redirect directive precisely, then return the complete updated draft. Never return a diff, a summary, or a placeholder \u2014 always the full text."
181
+ }
182
+ ],
183
+ "edges": [
184
+ {
185
+ "kind": "delegates",
186
+ "from": "coordinator",
187
+ "to": "writer",
188
+ "maxTraversals": 6
189
+ }
190
+ ],
191
+ "budget": {
192
+ "maxIterations": 12,
193
+ "maxTokens": 600000
194
+ },
195
+ "perWorker": {
196
+ "maxIterations": 8,
197
+ "maxTokens": 400000
198
+ },
199
+ "deliverableDescribe": "Supervise the writer to produce a finished draft that fully satisfies the mission. After each draft the writer returns, evaluate it and, if it falls short, redirect the writer with a specific change directive. You may redirect up to five times \u2014 each redirect must state concrete, actionable changes, not vague preferences. Declare the draft done only when every mission requirement is met. The deliverable is the accepted final draft text."
200
+ },
201
+ "runResultKind": "winner",
202
+ "ledgerRows": 1,
203
+ "exhaustedEdges": []
204
+ },
205
+ {
206
+ "id": "unmeasured-harness",
207
+ "decision": "graph",
208
+ "score": 0.5,
209
+ "reasons": [
210
+ "FAIL nodes: workers=1 total=2 expected=3",
211
+ "PASS generousBudgetsBecauseFloorUnknown: per-child tokens 150000 vs generous line 50000"
212
+ ],
213
+ "reason": "The topology is known ahead of time (root executing three probes via a worker) and needs an audit trail with strict budget caps, especially given the unmeasured harness. A single worker node represents the probe role, and capping its traversals enforces exactly three probe executions.",
214
+ "authoredGraph": {
215
+ "nodes": [
216
+ {
217
+ "id": "root",
218
+ "systemPrompt": "You are the root orchestrator. Your role is to coordinate the execution of distinct probes and aggregate their outputs."
219
+ },
220
+ {
221
+ "id": "probe-worker",
222
+ "systemPrompt": "You are a probe worker running on an unmeasured harness. Execute the specific probe you are assigned and return the raw output."
223
+ }
224
+ ],
225
+ "edges": [
226
+ {
227
+ "kind": "delegates",
228
+ "from": "root",
229
+ "to": "probe-worker",
230
+ "maxTraversals": 3
231
+ }
232
+ ],
233
+ "budget": {
234
+ "maxIterations": 10,
235
+ "maxTokens": 600000
236
+ },
237
+ "perWorker": {
238
+ "maxIterations": 5,
239
+ "maxTokens": 150000
240
+ },
241
+ "deliverableDescribe": "Run three probes on claude-code workers and collect what they output."
242
+ },
243
+ "runResultKind": "winner",
244
+ "ledgerRows": 1,
245
+ "exhaustedEdges": []
246
+ }
247
+ ]
248
+ }