@tangle-network/agent-eval 0.94.0 → 0.95.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +32 -0
- package/README.md +44 -30
- package/dist/adapters/http.d.ts +8 -7
- package/dist/adapters/http.js.map +1 -1
- package/dist/adapters/langchain.d.ts +3 -2
- package/dist/adapters/otel.d.ts +5 -4
- package/dist/analyst/index.d.ts +11 -31
- package/dist/analyst/index.js +5 -65
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-B6Ljo_dI.d.ts → analyze-runs-DtT6F_6T.d.ts} +3 -3
- package/dist/belief-state/index.d.ts +4 -3
- package/dist/benchmarks/index.d.ts +3 -2
- package/dist/campaign/index.d.ts +727 -616
- package/dist/campaign/index.js +1863 -1316
- package/dist/campaign/index.js.map +1 -1
- package/dist/{chunk-2K6UUZ7P.js → chunk-2T4EZACH.js} +1 -1
- package/dist/chunk-2T4EZACH.js.map +1 -0
- package/dist/{chunk-CTBHKLEU.js → chunk-77T4STFI.js} +59 -86
- package/dist/chunk-77T4STFI.js.map +1 -0
- package/dist/{chunk-EGPMSBEZ.js → chunk-7QTQKIDD.js} +178 -177
- package/dist/chunk-7QTQKIDD.js.map +1 -0
- package/dist/{chunk-MIFZUPEK.js → chunk-AQ5WQAIV.js} +21 -6
- package/dist/chunk-AQ5WQAIV.js.map +1 -0
- package/dist/chunk-DJWX3GVS.js +81 -0
- package/dist/chunk-DJWX3GVS.js.map +1 -0
- package/dist/{chunk-S6OZEZQK.js → chunk-HMA63UEO.js} +37 -9
- package/dist/{chunk-S6OZEZQK.js.map → chunk-HMA63UEO.js.map} +1 -1
- package/dist/{chunk-TBDR6PAI.js → chunk-IZCEK2HR.js} +2 -2
- package/dist/{chunk-SD2YFWQQ.js → chunk-KKWJD5E6.js} +20 -20
- package/dist/chunk-KKWJD5E6.js.map +1 -0
- package/dist/{chunk-KWRRMR3J.js → chunk-LO6IOIJ2.js} +10 -10
- package/dist/chunk-LO6IOIJ2.js.map +1 -0
- package/dist/{chunk-E4GH6USR.js → chunk-NZEQVRH5.js} +2 -2
- package/dist/chunk-NZEQVRH5.js.map +1 -0
- package/dist/{chunk-MPQWFX6Y.js → chunk-PSWWQXHF.js} +13 -88
- package/dist/chunk-PSWWQXHF.js.map +1 -0
- package/dist/{chunk-Q5LIB7BC.js → chunk-S4SYLDFX.js} +2 -2
- package/dist/chunk-S4SYLDFX.js.map +1 -0
- package/dist/{chunk-KW53MSA5.js → chunk-X74V6ESX.js} +2 -2
- package/dist/{chunk-QMUEXQJS.js → chunk-YBIGNSCZ.js} +81 -4
- package/dist/chunk-YBIGNSCZ.js.map +1 -0
- package/dist/{chunk-2KNZHH3P.js → chunk-Z6L6YSU6.js} +2 -2
- package/dist/{code-agent-session-BO8nCnv3.d.ts → code-agent-session-CPHRCb4-.d.ts} +1 -1
- package/dist/contract/index.d.ts +91 -43
- package/dist/contract/index.js +127 -17
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-D6qwHXIR.d.ts → control-Doncu-B_.d.ts} +2 -2
- package/dist/control.d.ts +3 -2
- package/dist/control.js +2 -2
- package/dist/{corpus-B8A4BDR3.d.ts → corpus-D4YW9UoJ.d.ts} +1 -1
- package/dist/{default-registry-6dhErQbs.d.ts → default-registry-GyE8X5SP.d.ts} +3 -3
- package/dist/diagnose.d.ts +4 -3
- package/dist/diagnose.js +1 -1
- package/dist/{run-improvement-loop-DBahB8Ax.d.ts → gepa-C1NCIZ9o.d.ts} +117 -130
- package/dist/hosted/index.d.ts +5 -4
- package/dist/{index-Bx3gZ8xl.d.ts → index-_Y4oNOOb.d.ts} +1 -1
- package/dist/index.d.ts +76 -81
- package/dist/index.js +66 -31
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DWl3z9tl.d.ts → insight-report-BnRjTibG.d.ts} +1 -1
- package/dist/{kind-factory-0BhLSI27.d.ts → kind-factory-X3eDYbKn.d.ts} +2 -3
- package/dist/matrix/index.d.ts +1 -1
- package/dist/meta-eval/index.d.ts +3 -2
- package/dist/multishot/index.d.ts +4 -4
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{pre-registration-mAnCugl9.d.ts → pre-registration-nfUdc9EQ.d.ts} +2 -42
- package/dist/{provenance-P-bCL2Fo.d.ts → provenance-CncDq9qE.d.ts} +26 -41
- package/dist/{release-report-BEbWmVYj.d.ts → release-report-pidWUMZ2.d.ts} +2 -2
- package/dist/reporting.d.ts +5 -4
- package/dist/{researcher-B0C2_fVO.d.ts → researcher-Jr8ME1dZ.d.ts} +2 -2
- package/dist/rl.d.ts +516 -515
- package/dist/rl.js +612 -612
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-Cy_W-hWZ.d.ts → rubric-predictive-validity-C2hDKM8Z.d.ts} +1 -1
- package/dist/{run-campaign-7WNXMDSN.js → run-campaign-WXY7KI67.js} +2 -2
- package/dist/{run-record-e7vj1uZQ.d.ts → run-record-CP2ObebC.d.ts} +14 -18
- package/dist/{runtime-trajectory-BDgfGZSr.d.ts → runtime-trajectory-BOUUjI0y.d.ts} +1 -1
- package/dist/{semantic-concept-judge-B9MgmBnM.d.ts → semantic-concept-judge-DSBB2Cfp.d.ts} +2 -2
- package/dist/{summary-report-BDOFevaT.d.ts → summary-report-CInXwsza.d.ts} +1 -1
- package/dist/testing-C21CHsq2.d.ts +20 -0
- package/dist/testing.d.ts +1 -0
- package/dist/testing.js +8 -0
- package/dist/testing.js.map +1 -0
- package/dist/traces.d.ts +26 -10
- package/dist/traces.js +41 -11
- package/dist/{types-Ce17tDlG.d.ts → types-B5x54y6n.d.ts} +1 -1
- package/dist/{types-mn5Aqk7x.d.ts → types-BUxNaJ8c.d.ts} +2 -4
- package/dist/{types-BU-7W85F.d.ts → types-DQRY8ZT-.d.ts} +60 -58
- package/dist/workflow/index.d.ts +5 -4
- package/dist/workflow/index.js +1 -1
- package/docs/campaign-proposers.md +170 -0
- package/docs/concepts.md +8 -4
- package/docs/customer-journeys.md +15 -13
- package/docs/design/loop-taxonomy.md +34 -66
- package/docs/distributed-driver.md +14 -14
- package/docs/feature-guide.md +1 -1
- package/docs/hosted-ingest-spec.md +2 -3
- package/docs/multi-shot-optimization.md +8 -8
- package/docs/product-eval-adoption.md +1 -1
- package/docs/self-improvement-map.md +33 -29
- package/package.json +8 -14
- package/dist/chunk-2K6UUZ7P.js.map +0 -1
- package/dist/chunk-CTBHKLEU.js.map +0 -1
- package/dist/chunk-E4GH6USR.js.map +0 -1
- package/dist/chunk-EGPMSBEZ.js.map +0 -1
- package/dist/chunk-KWRRMR3J.js.map +0 -1
- package/dist/chunk-MIFZUPEK.js.map +0 -1
- package/dist/chunk-MPQWFX6Y.js.map +0 -1
- package/dist/chunk-Q5LIB7BC.js.map +0 -1
- package/dist/chunk-QMUEXQJS.js.map +0 -1
- package/dist/chunk-SD2YFWQQ.js.map +0 -1
- package/docs/design/external-agent-wedge.md +0 -89
- package/docs/design/phase-d-rfc.md +0 -125
- package/docs/design/phase4-consumer-migration.md +0 -70
- package/docs/design/primitives-integration-spec.md +0 -393
- package/docs/design/product-self-improvement-loop.md +0 -146
- package/docs/design/self-improvement-engine.md +0 -140
- package/docs/design/self-improvement-protocol.md +0 -223
- package/docs/design/self-improvement-roadmap.md +0 -106
- package/docs/design/substrate-gaps.md +0 -118
- package/docs/phase-b-pairing-kit.md +0 -188
- package/docs/phase-b-runbook.md +0 -176
- package/docs/pilot/README.md +0 -62
- package/docs/pilot/customer-checklist.md +0 -90
- package/docs/pilot/integration-foreign-stack.md +0 -296
- package/docs/pilot/integration-tangle-stack.md +0 -248
- package/docs/pilot/one-pager.md +0 -161
- package/docs/pilot/sample-insight-report.json +0 -172
- package/docs/quickstart-external.md +0 -229
- package/docs/research/belief-state-agent-eval-roadmap.md +0 -593
- package/docs/research/research-roadmap.md +0 -205
- package/docs/specs/driver-honest-spec.md +0 -251
- package/docs/specs/hermes-self-improvement-audit.md +0 -93
- package/docs/specs/profile-versioning.md +0 -291
- package/docs/three-package-architecture.md +0 -168
- /package/dist/{chunk-TBDR6PAI.js.map → chunk-IZCEK2HR.js.map} +0 -0
- /package/dist/{chunk-KW53MSA5.js.map → chunk-X74V6ESX.js.map} +0 -0
- /package/dist/{chunk-2KNZHH3P.js.map → chunk-Z6L6YSU6.js.map} +0 -0
- /package/dist/{run-campaign-7WNXMDSN.js.map → run-campaign-WXY7KI67.js.map} +0 -0
|
@@ -1,248 +0,0 @@
|
|
|
1
|
-
# Integration — Tangle Intelligence on the Tangle stack (sandbox + tcloud)
|
|
2
|
-
|
|
3
|
-
Step-by-step. This is what we run with you on the onboarding call.
|
|
4
|
-
|
|
5
|
-
## Zero-setup demo first (30 seconds, no install)
|
|
6
|
-
|
|
7
|
-
```sh
|
|
8
|
-
npx @tangle-network/intelligence demo
|
|
9
|
-
```
|
|
10
|
-
|
|
11
|
-
End-to-end loop against synthetic data — agent + judge + scenarios + selfImprove. Prints the `InsightReport` shape you'll get on your real data. Useful to confirm the output is what you want before any integration. Hosted equivalent: open **[staging-intelligence.tangle.tools](https://staging-intelligence.tangle.tools)**.
|
|
12
|
-
|
|
13
|
-
## Prerequisites you already have
|
|
14
|
-
|
|
15
|
-
- `@tangle-network/sandbox` running your agent in a session
|
|
16
|
-
- `@tangle-network/tcloud` for LLM routing (or any OpenAI-compat router)
|
|
17
|
-
- Your scenarios (the inputs your agent handles) listed somewhere — even as YAML or a TS array
|
|
18
|
-
- A judge function for scoring outputs — LLM-as-judge is fine for v1
|
|
19
|
-
|
|
20
|
-
## Install
|
|
21
|
-
|
|
22
|
-
The CLI scaffolds and runs everything; you only add the substrate package if your code calls primitives directly:
|
|
23
|
-
|
|
24
|
-
```sh
|
|
25
|
-
# CLI (zero-install via npx, or add to your repo as a dev-dep)
|
|
26
|
-
npx @tangle-network/intelligence init
|
|
27
|
-
|
|
28
|
-
# Optional — only if your code imports analyzeRuns / selfImprove directly
|
|
29
|
-
pnpm add @tangle-network/agent-eval
|
|
30
|
-
# or for Python customers:
|
|
31
|
-
pip install agent-eval-rpc
|
|
32
|
-
```
|
|
33
|
-
|
|
34
|
-
`@tangle-network/intelligence` is the customer-facing CLI + hosted product (binary `tangle-intel`). `@tangle-network/agent-eval` is the substrate it wraps — install only if you want to script directly against the primitives.
|
|
35
|
-
|
|
36
|
-
## Step 1 — Ingest your trace stream
|
|
37
|
-
|
|
38
|
-
You already emit traces via sandbox sessions. Pull them into canonical `RunRecord[]`:
|
|
39
|
-
|
|
40
|
-
```ts
|
|
41
|
-
import { fromTangleSandbox } from '@tangle-network/agent-eval/adapters/sandbox'
|
|
42
|
-
|
|
43
|
-
const runs = await fromTangleSandbox({
|
|
44
|
-
sessionIds: ['session_abc', 'session_def'], // your current week
|
|
45
|
-
fromMs: lastReportTime,
|
|
46
|
-
toMs: Date.now(),
|
|
47
|
-
})
|
|
48
|
-
// runs is RunRecord[] — canonical wire shape, ready for any downstream substrate primitive
|
|
49
|
-
```
|
|
50
|
-
|
|
51
|
-
If your agent emits OTel directly instead of going through `@tangle-network/sandbox`:
|
|
52
|
-
|
|
53
|
-
```ts
|
|
54
|
-
import { fromOtelSpans } from '@tangle-network/agent-eval'
|
|
55
|
-
|
|
56
|
-
const runs = fromOtelSpans({ spans: yourOtelSpans })
|
|
57
|
-
```
|
|
58
|
-
|
|
59
|
-
## Step 2 — Get the decision packet (no LLM cost)
|
|
60
|
-
|
|
61
|
-
```ts
|
|
62
|
-
import { analyzeRuns } from '@tangle-network/agent-eval/contract'
|
|
63
|
-
|
|
64
|
-
const report = await analyzeRuns({
|
|
65
|
-
runs: thisWeek,
|
|
66
|
-
baselineRuns: lastWeek, // optional — gives you the "did my change help?" answer
|
|
67
|
-
baselineLabel: 'vs prior 7 days',
|
|
68
|
-
})
|
|
69
|
-
|
|
70
|
-
console.log(report.composite.mean) // overall score
|
|
71
|
-
console.log(report.composite.tailRuns) // worst 5 runs by name
|
|
72
|
-
console.log(report.priorPeriodComparison?.improvedMetrics) // ['composite'] if significantly better
|
|
73
|
-
console.log(report.priorPeriodComparison?.regressedMetrics) // ['cost'] if cost went up significantly
|
|
74
|
-
console.log(report.recommendations) // priority-ranked actions
|
|
75
|
-
```
|
|
76
|
-
|
|
77
|
-
That's the **full deterministic flow** — no LLM, $0 cost, runs in ms.
|
|
78
|
-
|
|
79
|
-
Render in your dashboard or pipe to Slack:
|
|
80
|
-
|
|
81
|
-
```ts
|
|
82
|
-
for (const rec of report.recommendations) {
|
|
83
|
-
if (rec.priority === 'critical') {
|
|
84
|
-
await slack.post(`🔴 ${rec.title}\n${rec.detail}`)
|
|
85
|
-
}
|
|
86
|
-
}
|
|
87
|
-
```
|
|
88
|
-
|
|
89
|
-
## Step 3 — Wire the closed loop (real LLM cost — opt-in)
|
|
90
|
-
|
|
91
|
-
Pick the surface you want to optimize. For most customers this is the agent's system-prompt addendum:
|
|
92
|
-
|
|
93
|
-
```ts
|
|
94
|
-
import { selfImprove, gepaDriver } from '@tangle-network/agent-eval/contract'
|
|
95
|
-
|
|
96
|
-
const result = await selfImprove({
|
|
97
|
-
scenarios: yourScenarios, // 20-50 representative inputs
|
|
98
|
-
agent: async (surface, scenario) => {
|
|
99
|
-
// Your existing agent invocation, with the substrate-proposed surface
|
|
100
|
-
// injected as the system-prompt addendum.
|
|
101
|
-
return await runYourAgent({
|
|
102
|
-
...scenario,
|
|
103
|
-
systemPromptAddendum: surface as string,
|
|
104
|
-
})
|
|
105
|
-
},
|
|
106
|
-
judge: yourJudge, // function (artifact) → { composite, dimensions }
|
|
107
|
-
baselineSurface: currentAddendum, // the production string today
|
|
108
|
-
driver: gepaDriver({
|
|
109
|
-
llm: { apiKey: tcloudKey, baseUrl: 'https://router.tangle.tools/v1' },
|
|
110
|
-
model: 'anthropic/claude-sonnet-4.6',
|
|
111
|
-
target: 'agent system-prompt addendum',
|
|
112
|
-
}),
|
|
113
|
-
budget: {
|
|
114
|
-
generations: 3,
|
|
115
|
-
populationSize: 4,
|
|
116
|
-
holdoutFraction: 0.3,
|
|
117
|
-
maxUsd: 25, // hard ceiling — refuses to overspend
|
|
118
|
-
},
|
|
119
|
-
})
|
|
120
|
-
|
|
121
|
-
console.log(`gate: ${result.gateDecision.kind}`)
|
|
122
|
-
console.log(`lift: ${result.lift.delta.toFixed(3)} CI=[${result.lift.ci95.join(', ')}]`)
|
|
123
|
-
console.log(`cost spent: $${result.totalCostUsd.toFixed(2)}`)
|
|
124
|
-
```
|
|
125
|
-
|
|
126
|
-
`result.gateDecision` is one of:
|
|
127
|
-
- `ship-substrate` — winner statistically beats baseline; safe to deploy
|
|
128
|
-
- `inconclusive` — CI straddles zero; either run more rollouts or expand corpus
|
|
129
|
-
- `ship-harness` / `merge` — only when `driftPolicy: 'benchmark-branches'` is on (advanced)
|
|
130
|
-
|
|
131
|
-
## Step 4 — Auto-PR the winner
|
|
132
|
-
|
|
133
|
-
```ts
|
|
134
|
-
if (result.gateDecision.kind === 'ship-substrate') {
|
|
135
|
-
await openAutoPr({
|
|
136
|
-
title: `eval: auto-improve ${target} (composite +${result.lift.delta.toFixed(3)})`,
|
|
137
|
-
body: `${result.gateDecision.reason}\n\n${formatInsight(result.insight)}`,
|
|
138
|
-
filePath: 'src/lib/.server/production-loop/prompt-addendum.ts',
|
|
139
|
-
newContent: result.diff.kind === 'replace' ? result.diff.content : applyDiff(currentAddendum, result.diff),
|
|
140
|
-
})
|
|
141
|
-
}
|
|
142
|
-
```
|
|
143
|
-
|
|
144
|
-
We ship `openAutoPr` from `@tangle-network/agent-eval/contract`. It wraps the GitHub PR flow with your existing token.
|
|
145
|
-
|
|
146
|
-
## The full canonical flow (script you copy and run)
|
|
147
|
-
|
|
148
|
-
```ts
|
|
149
|
-
// scripts/weekly-improvement.ts — run from a cron / GitHub Action
|
|
150
|
-
|
|
151
|
-
import { fromTangleSandbox } from '@tangle-network/agent-eval/adapters/sandbox'
|
|
152
|
-
import {
|
|
153
|
-
analyzeRuns,
|
|
154
|
-
gepaDriver,
|
|
155
|
-
openAutoPr,
|
|
156
|
-
selfImprove,
|
|
157
|
-
} from '@tangle-network/agent-eval/contract'
|
|
158
|
-
import { scenarios } from './eval/scenarios'
|
|
159
|
-
import { judge } from './eval/judges'
|
|
160
|
-
import { runYourAgent } from './src/agent'
|
|
161
|
-
import { PRODUCTION_ADDENDUM } from './src/lib/.server/production-loop/prompt-addendum'
|
|
162
|
-
|
|
163
|
-
const lastWeek = Date.now() - 7 * 24 * 60 * 60 * 1000
|
|
164
|
-
const twoWeeksAgo = lastWeek - 7 * 24 * 60 * 60 * 1000
|
|
165
|
-
|
|
166
|
-
const thisWeekRuns = await fromTangleSandbox({ fromMs: lastWeek, toMs: Date.now() })
|
|
167
|
-
const lastWeekRuns = await fromTangleSandbox({ fromMs: twoWeeksAgo, toMs: lastWeek })
|
|
168
|
-
|
|
169
|
-
// 1. Deterministic packet — always
|
|
170
|
-
const report = await analyzeRuns({
|
|
171
|
-
runs: thisWeekRuns,
|
|
172
|
-
baselineRuns: lastWeekRuns,
|
|
173
|
-
baselineLabel: 'vs prior 7 days',
|
|
174
|
-
})
|
|
175
|
-
|
|
176
|
-
// 2. Closed loop — only if composite regressed OR we haven't tried in a while
|
|
177
|
-
const shouldRun =
|
|
178
|
-
report.priorPeriodComparison?.regressedMetrics.includes('composite') ||
|
|
179
|
-
daysSinceLastImprovement() > 7
|
|
180
|
-
|
|
181
|
-
if (!shouldRun) {
|
|
182
|
-
console.log('No regression + recent run; skipping.')
|
|
183
|
-
process.exit(0)
|
|
184
|
-
}
|
|
185
|
-
|
|
186
|
-
const result = await selfImprove({
|
|
187
|
-
scenarios,
|
|
188
|
-
agent: (surface, scenario) =>
|
|
189
|
-
runYourAgent({ ...scenario, systemPromptAddendum: surface as string }),
|
|
190
|
-
judge,
|
|
191
|
-
baselineSurface: PRODUCTION_ADDENDUM,
|
|
192
|
-
driver: gepaDriver({
|
|
193
|
-
llm: { apiKey: process.env.TANGLE_KEY!, baseUrl: 'https://router.tangle.tools/v1' },
|
|
194
|
-
model: 'anthropic/claude-sonnet-4.6',
|
|
195
|
-
target: 'production agent system-prompt addendum',
|
|
196
|
-
}),
|
|
197
|
-
budget: { generations: 3, populationSize: 4, holdoutFraction: 0.3, maxUsd: 50 },
|
|
198
|
-
})
|
|
199
|
-
|
|
200
|
-
if (result.gateDecision.kind === 'ship-substrate') {
|
|
201
|
-
await openAutoPr({
|
|
202
|
-
title: `eval: auto-improve addendum (composite +${result.lift.delta.toFixed(3)})`,
|
|
203
|
-
body: renderInsightAsPrBody(result.insight),
|
|
204
|
-
filePath: 'src/lib/.server/production-loop/prompt-addendum.ts',
|
|
205
|
-
newContent: result.diff.kind === 'replace' ? result.diff.content : '...',
|
|
206
|
-
})
|
|
207
|
-
}
|
|
208
|
-
```
|
|
209
|
-
|
|
210
|
-
## What we'll do together on the onboarding call
|
|
211
|
-
|
|
212
|
-
1. **Map your existing setup** — where do your traces emit? which sandbox sessions? which scenarios exist already?
|
|
213
|
-
2. **Stub the judge** — even a single dimension is enough to start
|
|
214
|
-
3. **Run a deterministic `analyzeRuns()` against your live data** — first decision packet rendered live
|
|
215
|
-
4. **Wire one selfImprove cycle** — small budget, single generation, see the loop fire
|
|
216
|
-
5. **Schedule the cron + auto-PR target** — the loop runs autonomously thereafter
|
|
217
|
-
|
|
218
|
-
Time budget: ~90 minutes. By the end you have a working pilot.
|
|
219
|
-
|
|
220
|
-
## What our hosted tier adds on top
|
|
221
|
-
|
|
222
|
-
- Decision packet rendered weekly in the Intelligence dashboard — no code changes needed
|
|
223
|
-
- Slack / email digest on `regressedMetrics`
|
|
224
|
-
- Pareto chart, judge calibration, failure-cluster drilldown in the UI
|
|
225
|
-
- Multi-week trend lines
|
|
226
|
-
- Stripe-billed usage tracking
|
|
227
|
-
|
|
228
|
-
If you want self-hosted only, every primitive above works locally. The hosted tier is a convenience.
|
|
229
|
-
|
|
230
|
-
## FAQ
|
|
231
|
-
|
|
232
|
-
**Q: What's the smallest scenario corpus that gives useful results?**
|
|
233
|
-
A: ~15 scenarios for the deterministic packet (you get distributional stats + recommendations). For `selfImprove`'s held-out gate you want ≥20 since `holdoutFraction: 0.3` reserves 6 for the gate. Below that, the gate often returns `inconclusive`.
|
|
234
|
-
|
|
235
|
-
**Q: What if my judge isn't reliable yet?**
|
|
236
|
-
A: That's normal. Use multi-rater intake (`fromFeedbackTable`) to get inter-rater agreement (κ) first, then iterate on the judge until raters agree. Substrate has an `interRater` block in InsightReport showing exactly which scenarios raters disagree on.
|
|
237
|
-
|
|
238
|
-
**Q: What if a `selfImprove` campaign returns `inconclusive`?**
|
|
239
|
-
A: It refused to claim improvement because the CI straddles zero. Either expand the corpus, raise `holdoutFraction`, or run more generations. Better than shipping noise.
|
|
240
|
-
|
|
241
|
-
**Q: Can I use a non-tcloud LLM provider?**
|
|
242
|
-
A: Yes — `gepaDriver` accepts any `LlmClientOptions` (any OpenAI-compatible endpoint). We default to tcloud because we already have your auth.
|
|
243
|
-
|
|
244
|
-
**Q: How do I see what changed when the gate ships?**
|
|
245
|
-
A: `result.diff` is a structured patch. We also ship `diffRuns()` separately if you want to compare two campaign outputs.
|
|
246
|
-
|
|
247
|
-
**Q: What if my agent self-modifies (Hermes / Claude Code skills)?**
|
|
248
|
-
A: This is the offline/online drift case. We have the architecture spec ready (`docs/specs/profile-versioning.md`) but the implementation is gated on a forcing-function experiment. For v0.5x pilots we assume the substrate is the only writer to your agent's optimizable surface.
|
package/docs/pilot/one-pager.md
DELETED
|
@@ -1,161 +0,0 @@
|
|
|
1
|
-
# Statistical self-improvement for your agent — one-pager
|
|
2
|
-
|
|
3
|
-
**For:** teams running an agent on the Tangle stack (sandbox + tcloud), OR any agent emitting OTel traces, OR LangChain / LlamaIndex / Anthropic SDK / OpenAI Assistants / OpenRouter / custom — we meet you where you are.
|
|
4
|
-
**The pitch:** every week, get a statistically-rigorous answer to *"did my last change help?"* + a closed loop that proposes the next improvement + a held-out gate that refuses to ship regressions.
|
|
5
|
-
|
|
6
|
-
## What you get
|
|
7
|
-
|
|
8
|
-
| Deliverable | Cadence | LLM cost |
|
|
9
|
-
|---|---|---|
|
|
10
|
-
| **Decision packet** — composite distribution, per-dimension judges, cost-quality Pareto, failure clusters, named worst-N runs, ranked recommendations | Whenever you want it. Hosted runs on a 15-min schedule by default. | $0 (deterministic) |
|
|
11
|
-
| **Prior-period comparison** — Welch CI on composite / cost / duration / per-dimension deltas vs your prior week, with regressed + improved metrics named | Same cadence | $0 |
|
|
12
|
-
| **Closed-loop improvement** — `selfImprove()` proposes prompt edits, runs scenarios, gates on paired-bootstrap CI, auto-PRs the winner | On-demand, opt-in | Real $; you set a `maxUsd` ceiling |
|
|
13
|
-
|
|
14
|
-
Every claim is falsifiable: `n=`, `CI95=[a, b]`, `p=`, `Cohen's d=`. No vibes, no "score went up." Where the data doesn't support a section, the report says so explicitly instead of inventing signal.
|
|
15
|
-
|
|
16
|
-
## Why this is different
|
|
17
|
-
|
|
18
|
-
| | LangSmith / Braintrust / Phoenix | Hermes / Claude Code skills | **Tangle** |
|
|
19
|
-
|---|---|---|---|
|
|
20
|
-
| Trace ingest | proprietary | own runtime | universal (sandbox + tcloud + OTel + any custom) |
|
|
21
|
-
| Decision packet | scorecards (no CI) | none | **paired-bootstrap CI on every claim** |
|
|
22
|
-
| Closed loop | none | heuristic, no gate | **statistically-gated; refuses regressions** |
|
|
23
|
-
| Prior-period delta | none | none | **Welch CI on every metric** |
|
|
24
|
-
| Sample-size guidance | none | none | **MDE-aware** |
|
|
25
|
-
| Auto-PR promotion | none | none | **opt-in, on green gate only** |
|
|
26
|
-
|
|
27
|
-
## Integration paths — pick your stack
|
|
28
|
-
|
|
29
|
-
| Your stack | Intake adapter | LLM provider for closed loop |
|
|
30
|
-
|---|---|---|
|
|
31
|
-
| **Tangle (sandbox + tcloud)** | `fromTangleSandbox` | tcloud (already wired) |
|
|
32
|
-
| Any OTel exporter (Datadog APM, Honeycomb, NewRelic, OpenInference) | `fromOtelSpans` | any OpenAI-compat |
|
|
33
|
-
| LangChain (LangSmith) | LangSmith → OTel export → `fromOtelSpans` today; `fromLangChain` queued 0.55.0 | OpenAI, Anthropic, OpenRouter, tcloud |
|
|
34
|
-
| LlamaIndex | `OpenInferenceCallbackHandler` → OTel → ingest | any OpenAI-compat |
|
|
35
|
-
| Anthropic SDK direct | OTel wrapping (~20 LOC) → `fromOtelSpans`; `fromAnthropicSDK` queued | Anthropic, OpenRouter |
|
|
36
|
-
| OpenAI Assistants API | Custom mapper (~20 LOC) today; `fromOpenAIAssistants` queued | OpenAI, OpenRouter |
|
|
37
|
-
| OpenRouter (any model on any path) | Whatever you already use for tracing | OpenRouter (OpenAI-compat baseUrl) |
|
|
38
|
-
| vLLM / Ollama / LMStudio / self-hosted | OTel wrap | Your local OpenAI-compat endpoint |
|
|
39
|
-
| Multi-rater human feedback (no automated judge yet) | `fromFeedbackTable` | n/a — gives you κ + disagreement triage |
|
|
40
|
-
| Custom logs / DB rows | ~20-line mapper to `RunRecord` | any OpenAI-compat |
|
|
41
|
-
|
|
42
|
-
Full integration walkthroughs:
|
|
43
|
-
- **Tangle stack** → [`integration-tangle-stack.md`](./integration-tangle-stack.md)
|
|
44
|
-
- **Everything else** → [`integration-foreign-stack.md`](./integration-foreign-stack.md)
|
|
45
|
-
|
|
46
|
-
## Zero-setup demo first — 30 seconds
|
|
47
|
-
|
|
48
|
-
Before any integration, run the demo against synthetic data so you see the output shape live:
|
|
49
|
-
|
|
50
|
-
```sh
|
|
51
|
-
npx @tangle-network/intelligence demo
|
|
52
|
-
```
|
|
53
|
-
|
|
54
|
-
No install, no key, no data. Synthetic agent runs through synthetic scenarios; the CLI prints a real `InsightReport` with composite distribution + Pareto + prior-period delta + ranked recommendations. Same output shape you'll get on your real data once we integrate.
|
|
55
|
-
|
|
56
|
-
When you're ready to integrate, the same CLI scaffolds your repo:
|
|
57
|
-
|
|
58
|
-
```sh
|
|
59
|
-
npx @tangle-network/intelligence init # creates eval/scenarios.json + judges.ts + pnpm scripts + .runs/
|
|
60
|
-
npx @tangle-network/intelligence report # renders InsightReport from your latest traces
|
|
61
|
-
npx @tangle-network/intelligence improve --max-usd 25 # runs selfImprove with cost ceiling, opens auto-PR on green gate
|
|
62
|
-
```
|
|
63
|
-
|
|
64
|
-
Hosted equivalent: **[staging-intelligence.tangle.tools](https://staging-intelligence.tangle.tools)** — open in your browser, ingest your traces, see the dashboard render the same packet your CLI produces.
|
|
65
|
-
|
|
66
|
-
## How you integrate (Tangle stack — 4 steps)
|
|
67
|
-
|
|
68
|
-
```ts
|
|
69
|
-
import { fromTangleSandbox } from '@tangle-network/agent-eval/adapters/sandbox'
|
|
70
|
-
import { analyzeRuns, selfImprove, gepaDriver } from '@tangle-network/agent-eval/contract'
|
|
71
|
-
|
|
72
|
-
// 1. You already emit traces via @tangle-network/sandbox + tcloud.
|
|
73
|
-
// Pull them into canonical RunRecord[]:
|
|
74
|
-
const runs = fromTangleSandbox({ sessionId, sinceMs: lastReportTime })
|
|
75
|
-
|
|
76
|
-
// 2. Get the decision packet — no LLM cost.
|
|
77
|
-
const report = await analyzeRuns({ runs, baselineRuns: priorWeekRuns })
|
|
78
|
-
// → report.composite + .priorPeriodComparison + .recommendations
|
|
79
|
-
|
|
80
|
-
// 3. When you want to actually improve, run the closed loop.
|
|
81
|
-
const result = await selfImprove({
|
|
82
|
-
scenarios: yourScenarios, // we help you build these
|
|
83
|
-
agent: (surface, scenario) => runYourAgent(scenario, surface),
|
|
84
|
-
judge: yourJudge, // any function (artifact) → JudgeScore
|
|
85
|
-
baselineSurface: currentSystemPrompt,
|
|
86
|
-
driver: gepaDriver({ llm: tcloud, model: 'claude-sonnet-4.6', target: 'agent prompt' }),
|
|
87
|
-
budget: { generations: 3, populationSize: 4, holdoutFraction: 0.3, maxUsd: 25 },
|
|
88
|
-
})
|
|
89
|
-
|
|
90
|
-
// 4. Result is a verifiable diff with statistical evidence.
|
|
91
|
-
// Auto-PR if result.gateDecision === 'ship-substrate'.
|
|
92
|
-
```
|
|
93
|
-
|
|
94
|
-
That's it. ~30 lines of integration code; the rest is your existing agent + tcloud setup.
|
|
95
|
-
|
|
96
|
-
## What we need from you
|
|
97
|
-
|
|
98
|
-
- API key for tcloud (you already have this)
|
|
99
|
-
- Read access to your sandbox session traces
|
|
100
|
-
- A list of 20-50 representative scenarios your agent should handle
|
|
101
|
-
- A judge function — even a simple LLM-as-judge gets you 80% of the value
|
|
102
|
-
- An LLM-cost budget for the closed loop (default: $25/campaign)
|
|
103
|
-
|
|
104
|
-
## What you ship back to your customers
|
|
105
|
-
|
|
106
|
-
The substrate produces a single JSON `InsightReport` your dashboard renders. Live demo embedded in the Tangle Intelligence dashboard. Example below — every section optional based on what your data supports.
|
|
107
|
-
|
|
108
|
-
```json
|
|
109
|
-
{
|
|
110
|
-
"n": 36,
|
|
111
|
-
"composite": {
|
|
112
|
-
"mean": 0.823, "p50": 0.85, "p95": 0.96, "stddev": 0.11,
|
|
113
|
-
"tailRuns": [
|
|
114
|
-
{ "runId": "scenario::checkout-bug", "score": 0.41 },
|
|
115
|
-
{ "runId": "scenario::refund-policy", "score": 0.48 }
|
|
116
|
-
]
|
|
117
|
-
},
|
|
118
|
-
"priorPeriodComparison": {
|
|
119
|
-
"baselineN": 34,
|
|
120
|
-
"currentN": 36,
|
|
121
|
-
"windowLabel": "vs prior 7 days",
|
|
122
|
-
"metrics": {
|
|
123
|
-
"composite": {
|
|
124
|
-
"current": 0.823, "baseline": 0.731, "delta": 0.092,
|
|
125
|
-
"ci95": [0.041, 0.143], "pValue": 0.0008,
|
|
126
|
-
"cohensD": 0.84, "significant": true
|
|
127
|
-
}
|
|
128
|
-
},
|
|
129
|
-
"improvedMetrics": ["composite"],
|
|
130
|
-
"regressedMetrics": []
|
|
131
|
-
},
|
|
132
|
-
"recommendations": [
|
|
133
|
-
{
|
|
134
|
-
"priority": "low",
|
|
135
|
-
"kind": "ship",
|
|
136
|
-
"title": "composite improved from 0.731 → 0.823 vs prior 7 days",
|
|
137
|
-
"detail": "Welch CI95=[0.041, 0.143], p=0.0008, Cohen's d=0.84 (n_current=36, n_baseline=34). Statistically significant improvement worth flagging."
|
|
138
|
-
},
|
|
139
|
-
{
|
|
140
|
-
"priority": "high",
|
|
141
|
-
"kind": "investigate",
|
|
142
|
-
"title": "Top failure cluster: refund-policy (12% of failures)",
|
|
143
|
-
"detail": "4 runs failed. Largest cluster groups by intent — agent missed compliance flag in 3 of 4."
|
|
144
|
-
}
|
|
145
|
-
]
|
|
146
|
-
}
|
|
147
|
-
```
|
|
148
|
-
|
|
149
|
-
## Pricing for the pilot
|
|
150
|
-
|
|
151
|
-
- Free for the first 30 days
|
|
152
|
-
- Hosted decision-packet generation: included
|
|
153
|
-
- LLM cost on closed-loop campaigns: pass-through to your tcloud account
|
|
154
|
-
- Post-pilot: per-campaign pricing tied to budget cap + per-decision-packet billed monthly
|
|
155
|
-
|
|
156
|
-
## Next step
|
|
157
|
-
|
|
158
|
-
Reply with: which agent + which week you want to start, and we'll set up the integration on a shared call. ~1 hour to first running report.
|
|
159
|
-
|
|
160
|
-
—
|
|
161
|
-
*Tangle Network · @tangle-network/agent-eval @0.53.0 · MIT · Self-hostable*
|
|
@@ -1,172 +0,0 @@
|
|
|
1
|
-
{
|
|
2
|
-
"n": 36,
|
|
3
|
-
"composite": {
|
|
4
|
-
"n": 36,
|
|
5
|
-
"mean": 0.823,
|
|
6
|
-
"p50": 0.85,
|
|
7
|
-
"p95": 0.96,
|
|
8
|
-
"stddev": 0.114,
|
|
9
|
-
"min": 0.41,
|
|
10
|
-
"max": 0.98,
|
|
11
|
-
"tailRuns": [
|
|
12
|
-
{ "runId": "scenario::checkout-bug", "score": 0.41 },
|
|
13
|
-
{ "runId": "scenario::refund-policy-edge", "score": 0.48 },
|
|
14
|
-
{ "runId": "scenario::multi-tenant-isolation", "score": 0.52 },
|
|
15
|
-
{ "runId": "scenario::stale-cache-invalidation", "score": 0.55 },
|
|
16
|
-
{ "runId": "scenario::partial-payment", "score": 0.61 }
|
|
17
|
-
],
|
|
18
|
-
"histogram": [
|
|
19
|
-
{ "lo": 0.40, "hi": 0.46, "count": 1 },
|
|
20
|
-
{ "lo": 0.46, "hi": 0.51, "count": 1 },
|
|
21
|
-
{ "lo": 0.51, "hi": 0.57, "count": 1 },
|
|
22
|
-
{ "lo": 0.57, "hi": 0.62, "count": 1 },
|
|
23
|
-
{ "lo": 0.62, "hi": 0.68, "count": 2 },
|
|
24
|
-
{ "lo": 0.68, "hi": 0.74, "count": 2 },
|
|
25
|
-
{ "lo": 0.74, "hi": 0.80, "count": 4 },
|
|
26
|
-
{ "lo": 0.80, "hi": 0.85, "count": 9 },
|
|
27
|
-
{ "lo": 0.85, "hi": 0.91, "count": 8 },
|
|
28
|
-
{ "lo": 0.91, "hi": 0.96, "count": 6 },
|
|
29
|
-
{ "lo": 0.96, "hi": 1.0, "count": 1 }
|
|
30
|
-
]
|
|
31
|
-
},
|
|
32
|
-
"perDimension": {
|
|
33
|
-
"intent-recognition": {
|
|
34
|
-
"n": 36, "mean": 0.89, "p50": 0.92, "p95": 0.98, "stddev": 0.08
|
|
35
|
-
},
|
|
36
|
-
"compliance-flagging": {
|
|
37
|
-
"n": 36, "mean": 0.71, "p50": 0.75, "p95": 0.92, "stddev": 0.18
|
|
38
|
-
},
|
|
39
|
-
"tone": {
|
|
40
|
-
"n": 36, "mean": 0.94, "p50": 0.95, "p95": 0.99, "stddev": 0.04
|
|
41
|
-
}
|
|
42
|
-
},
|
|
43
|
-
"costQuality": {
|
|
44
|
-
"cost": {
|
|
45
|
-
"n": 36, "mean": 0.087, "p50": 0.082, "p95": 0.124, "stddev": 0.021
|
|
46
|
-
},
|
|
47
|
-
"pareto": {
|
|
48
|
-
"kind": "pareto-cost-quality",
|
|
49
|
-
"split": "holdout",
|
|
50
|
-
"axes": { "x": "costUsd", "y": "score" },
|
|
51
|
-
"points": [
|
|
52
|
-
{ "candidateId": "baseline-v3.1", "cost": 0.087, "quality": 0.823, "n": 36, "onFrontier": true }
|
|
53
|
-
]
|
|
54
|
-
}
|
|
55
|
-
},
|
|
56
|
-
"judges": {
|
|
57
|
-
"claude-sonnet-4.6": {
|
|
58
|
-
"n": 36,
|
|
59
|
-
"meanScore": 0.831,
|
|
60
|
-
"calibration": null
|
|
61
|
-
}
|
|
62
|
-
},
|
|
63
|
-
"lift": null,
|
|
64
|
-
"failureClusters": {
|
|
65
|
-
"totalFailures": 4,
|
|
66
|
-
"clusters": [
|
|
67
|
-
{
|
|
68
|
-
"id": "cluster_refund_compliance",
|
|
69
|
-
"name": "refund-policy missed compliance flag",
|
|
70
|
-
"share": 0.75,
|
|
71
|
-
"exemplars": [
|
|
72
|
-
"scenario::refund-policy-edge",
|
|
73
|
-
"scenario::partial-payment",
|
|
74
|
-
"scenario::cross-border-refund"
|
|
75
|
-
],
|
|
76
|
-
"suggestedFix": "Add explicit step to the addendum: when refund amount > $100 OR cross-border, surface compliance flag before responding."
|
|
77
|
-
}
|
|
78
|
-
]
|
|
79
|
-
},
|
|
80
|
-
"priorPeriodComparison": {
|
|
81
|
-
"baselineN": 34,
|
|
82
|
-
"currentN": 36,
|
|
83
|
-
"windowLabel": "vs prior 7 days",
|
|
84
|
-
"metrics": {
|
|
85
|
-
"composite": {
|
|
86
|
-
"current": 0.823,
|
|
87
|
-
"baseline": 0.731,
|
|
88
|
-
"delta": 0.092,
|
|
89
|
-
"ci95": [0.041, 0.143],
|
|
90
|
-
"pValue": 0.0008,
|
|
91
|
-
"cohensD": 0.84,
|
|
92
|
-
"baselineN": 34,
|
|
93
|
-
"currentN": 36,
|
|
94
|
-
"significant": true
|
|
95
|
-
},
|
|
96
|
-
"cost": {
|
|
97
|
-
"current": 0.087,
|
|
98
|
-
"baseline": 0.082,
|
|
99
|
-
"delta": 0.005,
|
|
100
|
-
"ci95": [-0.003, 0.013],
|
|
101
|
-
"pValue": 0.21,
|
|
102
|
-
"cohensD": 0.23,
|
|
103
|
-
"baselineN": 34,
|
|
104
|
-
"currentN": 36,
|
|
105
|
-
"significant": false
|
|
106
|
-
},
|
|
107
|
-
"duration": {
|
|
108
|
-
"current": 4820,
|
|
109
|
-
"baseline": 5340,
|
|
110
|
-
"delta": -520,
|
|
111
|
-
"ci95": [-840, -200],
|
|
112
|
-
"pValue": 0.002,
|
|
113
|
-
"cohensD": -0.71,
|
|
114
|
-
"baselineN": 34,
|
|
115
|
-
"currentN": 36,
|
|
116
|
-
"significant": true
|
|
117
|
-
},
|
|
118
|
-
"dim.compliance-flagging": {
|
|
119
|
-
"current": 0.71,
|
|
120
|
-
"baseline": 0.58,
|
|
121
|
-
"delta": 0.13,
|
|
122
|
-
"ci95": [0.06, 0.20],
|
|
123
|
-
"pValue": 0.0004,
|
|
124
|
-
"cohensD": 0.79,
|
|
125
|
-
"baselineN": 34,
|
|
126
|
-
"currentN": 36,
|
|
127
|
-
"significant": true
|
|
128
|
-
}
|
|
129
|
-
},
|
|
130
|
-
"improvedMetrics": ["composite", "duration", "dim.compliance-flagging"],
|
|
131
|
-
"regressedMetrics": []
|
|
132
|
-
},
|
|
133
|
-
"release": {
|
|
134
|
-
"status": "pass",
|
|
135
|
-
"axes": [
|
|
136
|
-
{ "name": "quality-lift", "status": "pass", "detail": "no candidate/baseline pair within campaign; relying on priorPeriodComparison" },
|
|
137
|
-
{ "name": "contamination", "status": "pass", "detail": "no canaries supplied" },
|
|
138
|
-
{ "name": "composite-distribution", "status": "pass", "detail": "mean=0.823, p50=0.85, p95=0.96 over n=36" }
|
|
139
|
-
],
|
|
140
|
-
"issues": []
|
|
141
|
-
},
|
|
142
|
-
"recommendations": [
|
|
143
|
-
{
|
|
144
|
-
"priority": "low",
|
|
145
|
-
"kind": "ship",
|
|
146
|
-
"title": "composite improved from 0.731 → 0.823 vs prior 7 days",
|
|
147
|
-
"detail": "Welch CI95=[0.041, 0.143], p=0.0008, Cohen's d=0.84 (n_current=36, n_baseline=34). Statistically significant improvement worth flagging.",
|
|
148
|
-
"evidencePath": "priorPeriodComparison.metrics.composite"
|
|
149
|
-
},
|
|
150
|
-
{
|
|
151
|
-
"priority": "low",
|
|
152
|
-
"kind": "ship",
|
|
153
|
-
"title": "compliance-flagging dimension improved from 0.58 → 0.71",
|
|
154
|
-
"detail": "Welch CI95=[0.06, 0.20], p=0.0004, Cohen's d=0.79. The fix from last week's PR is statistically validated.",
|
|
155
|
-
"evidencePath": "priorPeriodComparison.metrics.dim.compliance-flagging"
|
|
156
|
-
},
|
|
157
|
-
{
|
|
158
|
-
"priority": "high",
|
|
159
|
-
"kind": "investigate",
|
|
160
|
-
"title": "Top failure cluster: refund-policy missed compliance flag (75% of failures)",
|
|
161
|
-
"detail": "3 of 4 failed runs cluster here. Suggested fix: add explicit step to addendum for refund > $100 OR cross-border → surface compliance flag.",
|
|
162
|
-
"evidencePath": "failureClusters.clusters[0]"
|
|
163
|
-
},
|
|
164
|
-
{
|
|
165
|
-
"priority": "low",
|
|
166
|
-
"kind": "ship",
|
|
167
|
-
"title": "duration improved from 5340ms → 4820ms vs prior 7 days",
|
|
168
|
-
"detail": "Welch CI95=[-840, -200]ms, p=0.002, Cohen's d=-0.71. Agent is meaningfully faster — worth keeping the optimization that drove this.",
|
|
169
|
-
"evidencePath": "priorPeriodComparison.metrics.duration"
|
|
170
|
-
}
|
|
171
|
-
]
|
|
172
|
-
}
|