@tangle-network/agent-bench 0.11.2 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. package/CHANGELOG.md +28 -0
  2. package/HARNESS.md +6 -2
  3. package/README.md +1 -4
  4. package/dist/benchmarks/swe-bench.js +4 -9
  5. package/dist/benchmarks/swe-bench.js.map +1 -1
  6. package/package.json +5 -5
  7. package/scripts/run-package-tests.mjs +2 -2
  8. package/src/benchmarks/swe-bench.test.mts +49 -0
  9. package/src/benchmarks/swe-bench.ts +4 -9
  10. package/src/quant-arena/README.md +0 -144
  11. package/src/quant-arena/backtest.test.mts +0 -135
  12. package/src/quant-arena/backtest.ts +0 -218
  13. package/src/quant-arena/data.test.mts +0 -44
  14. package/src/quant-arena/data.ts +0 -141
  15. package/src/quant-arena/driver.test.mts +0 -253
  16. package/src/quant-arena/driver.ts +0 -219
  17. package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
  18. package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
  19. package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
  20. package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
  21. package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
  22. package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
  23. package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
  24. package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
  25. package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
  26. package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
  27. package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
  28. package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
  29. package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
  30. package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
  31. package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
  32. package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
  33. package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
  34. package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
  35. package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
  36. package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
  37. package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
  38. package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
  39. package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
  40. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
  41. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
  42. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
  43. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
  44. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
  45. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
  46. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
  47. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
  48. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
  49. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
  50. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
  51. package/src/quant-arena/holdout-certify.mts +0 -206
  52. package/src/quant-arena/holdout-certify.test.mts +0 -82
  53. package/src/quant-arena/leak-audit.test.mts +0 -79
  54. package/src/quant-arena/leak-audit.ts +0 -95
  55. package/src/quant-arena/make-fixtures.mts +0 -161
  56. package/src/quant-arena/multiplicity.test.mts +0 -68
  57. package/src/quant-arena/multiplicity.ts +0 -87
  58. package/src/quant-arena/nautilus-certify.ts +0 -31
  59. package/src/quant-arena/oms.ts +0 -90
  60. package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
  61. package/src/quant-arena/python/pyproject.toml +0 -8
  62. package/src/quant-arena/python/uv.lock +0 -1297
  63. package/src/quant-arena/python/vbt-worker.py +0 -192
  64. package/src/quant-arena/quant-loop.mts +0 -840
  65. package/src/quant-arena/quant-loop.test.mts +0 -75
  66. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
  67. package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
  68. package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
  69. package/src/quant-arena/types.ts +0 -133
  70. package/src/quant-arena/vbt-client.ts +0 -321
  71. package/src/quant-arena/vbt-parity.test.mts +0 -183
  72. package/src/quant-arena/windows.test.mts +0 -45
  73. package/src/quant-arena/windows.ts +0 -54
  74. package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
  75. package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
  76. package/src/rollout-ledger/settle-capture.mts +0 -448
  77. package/src/rollout-ledger/settle-capture.test.mts +0 -270
  78. package/src/swe-arena/activation.mts +0 -225
  79. package/src/swe-arena/activation.test.mts +0 -300
  80. package/src/swe-arena/analyze.ts +0 -211
  81. package/src/swe-arena/arms.ts +0 -862
  82. package/src/swe-arena/bootstrap-meta.mts +0 -188
  83. package/src/swe-arena/bootstrap-meta.test.mts +0 -51
  84. package/src/swe-arena/briefing.mts +0 -217
  85. package/src/swe-arena/briefing.test.mts +0 -179
  86. package/src/swe-arena/calibrate.ts +0 -217
  87. package/src/swe-arena/capabilities.mts +0 -76
  88. package/src/swe-arena/capabilities.test.mts +0 -57
  89. package/src/swe-arena/capacity.ts +0 -198
  90. package/src/swe-arena/cell-evidence.mts +0 -437
  91. package/src/swe-arena/cell-evidence.test.mts +0 -248
  92. package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
  93. package/src/swe-arena/diagnosis-ensemble.ts +0 -523
  94. package/src/swe-arena/execution.test.mts +0 -1171
  95. package/src/swe-arena/factory-command-container.ts +0 -284
  96. package/src/swe-arena/factory-judge-child.mts +0 -228
  97. package/src/swe-arena/factory.test.mts +0 -645
  98. package/src/swe-arena/fixtures/analyze.py +0 -80
  99. package/src/swe-arena/fixtures/excludes.txt +0 -8
  100. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
  101. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
  102. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
  103. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
  104. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
  105. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
  106. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
  107. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
  108. package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
  109. package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
  110. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
  111. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
  112. package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
  113. package/src/swe-arena/fixtures/holdout.json +0 -44
  114. package/src/swe-arena/fixtures/instances.json +0 -146
  115. package/src/swe-arena/fixtures/ledger.jsonl +0 -12
  116. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
  117. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
  118. package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
  119. package/src/swe-arena/fixtures/rematch.jsonl +0 -3
  120. package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
  121. package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
  122. package/src/swe-arena/fixtures/run-report/README.md +0 -43
  123. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
  124. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
  125. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
  126. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
  127. package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
  128. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
  129. package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
  130. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
  131. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
  132. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
  133. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
  134. package/src/swe-arena/fixtures/worker-tokens.json +0 -42
  135. package/src/swe-arena/fixtures.ts +0 -237
  136. package/src/swe-arena/gepa-seat.mts +0 -886
  137. package/src/swe-arena/gepa-seat.test.mts +0 -1136
  138. package/src/swe-arena/holdout-certify.mts +0 -408
  139. package/src/swe-arena/holdout-certify.test.mts +0 -160
  140. package/src/swe-arena/implementation-ref.test.mts +0 -64
  141. package/src/swe-arena/implementation-ref.ts +0 -62
  142. package/src/swe-arena/judge-child.mts +0 -37
  143. package/src/swe-arena/ledger-orphans.mts +0 -77
  144. package/src/swe-arena/ledger-orphans.test.mts +0 -149
  145. package/src/swe-arena/manifest.mts +0 -293
  146. package/src/swe-arena/manifest.test.mts +0 -169
  147. package/src/swe-arena/materialize.ts +0 -142
  148. package/src/swe-arena/outer-loop.mts +0 -2854
  149. package/src/swe-arena/outer-loop.test.mts +0 -714
  150. package/src/swe-arena/parity.test.mts +0 -87
  151. package/src/swe-arena/premeasured-from-cells.mts +0 -296
  152. package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
  153. package/src/swe-arena/proc.test.mts +0 -172
  154. package/src/swe-arena/proc.ts +0 -260
  155. package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
  156. package/src/swe-arena/profiles/default-author.profile.json +0 -12
  157. package/src/swe-arena/proposer-fanout.mts +0 -736
  158. package/src/swe-arena/proposer-fanout.test.mts +0 -660
  159. package/src/swe-arena/proposer-provenance.mts +0 -176
  160. package/src/swe-arena/proposer-provenance.test.mts +0 -106
  161. package/src/swe-arena/reconcile.ts +0 -0
  162. package/src/swe-arena/replay.mts +0 -183
  163. package/src/swe-arena/replay.test.mts +0 -300
  164. package/src/swe-arena/run-experiment.mts +0 -729
  165. package/src/swe-arena/run-report.mts +0 -75
  166. package/src/swe-arena/run-supervisor.mjs +0 -297
  167. package/src/swe-arena/run-supervisor.test.mts +0 -539
  168. package/src/swe-arena/score-split.mts +0 -140
  169. package/src/swe-arena/score-split.test.mts +0 -123
  170. package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
  171. package/src/swe-arena/scratch-worktree.test.mts +0 -56
  172. package/src/swe-arena/scratch-worktree.ts +0 -64
  173. package/src/swe-arena/serialized-judge.ts +0 -414
  174. package/src/swe-arena/types.ts +0 -218
@@ -1,523 +0,0 @@
1
- /**
2
- * Diagnosis ensemble — N BLIND failure analysts over the SAME supervisor-run
3
- * artifacts (brain.jsonl, driver.log, workers/*.ndjson + *.patch, verify.log,
4
- * result.json, the delivered patch, and the official-judge outcome), each a
5
- * different model routed through router.tangle.tools, fused by union +
6
- * cross-analyst agreement ranking. Minority findings SURVIVE fusion marked as
7
- * competing hypotheses — the round-4 protocol treats them as candidate
8
- * proposal seeds, not noise.
9
- *
10
- * Design constraints (from supervisor-lab .evolve/state.json round4_design):
11
- * - the model list is CONFIG — glm-5.2 is the only model proven routed today;
12
- * gpt-5.5 / opus-4.8 slot in by editing the analyst spec list, never by a
13
- * hardcoded requirement on an unrouted model.
14
- * - analysts are blind: same bundle, no cross-talk, independent calls.
15
- * - the containing run is launched through dotenvx; every analyst enters Runtime through one
16
- * exact AgentProfile and its event record is retained with the response.
17
- */
18
-
19
- import { mkdir, readdir, readFile, writeFile } from 'node:fs/promises'
20
- import { join } from 'node:path'
21
- import { type AnalystFinding, makeFinding } from '@tangle-network/agent-eval'
22
- import { findSupervisorRunDir, type SecretsEnv } from './arms'
23
- import { ROUTER_ENDPOINT, sleepWithSignal } from './capacity'
24
- import { runBenchRouterTurn } from '../router-turn'
25
-
26
- // ---------------------------------------------------------------------------
27
- // Analyst specs.
28
- // ---------------------------------------------------------------------------
29
-
30
- export interface AnalystSpec {
31
- /** Stable label, e.g. 'glm-5.2#1'. Used in fusion attribution + filenames. */
32
- id: string
33
- model: string
34
- /** Provider identity stamped into the exact profile. Default: tangle-router. */
35
- provider?: string
36
- /** Chat-completions endpoint. Default: the Tangle router. */
37
- url?: string
38
- /** NAME of the env var holding the bearer key (resolved in the dotenvx child). */
39
- apiKeyEnv?: string
40
- /** glm-5.2 returns empty content when starved below ~8000, and at sampled
41
- * temperatures its REASONING alone can consume a full 8000 ceiling (both
42
- * measured — the calibration smoke saw out=8000 with empty content).
43
- * Default 16_000 so reasoning + the JSON answer always fit. */
44
- maxTokens?: number
45
- temperature?: number
46
- }
47
-
48
- /** N same-model analysts (the smoke default). Real rounds replace entries with
49
- * other routed models — diversity comes from the config, never a hardcode.
50
- * Same-model analysts at temperature 0 are byte-identical duplicates (measured
51
- * in the calibration smoke: #2 and #3 returned the same 2823 output tokens),
52
- * which silently inflates agreement — so only the FIRST same-model analyst
53
- * runs at 0; the rest sample at 0.7 to buy real diversity. */
54
- export function defaultAnalysts(n = 3, model = 'glm-5.2'): AnalystSpec[] {
55
- return Array.from({ length: n }, (_, i) => ({
56
- id: `${model}#${i + 1}`,
57
- model,
58
- temperature: i === 0 ? 0 : 0.7,
59
- }))
60
- }
61
-
62
- // ---------------------------------------------------------------------------
63
- // Artifact bundle — the ONE shared context every blind analyst reads.
64
- // ---------------------------------------------------------------------------
65
-
66
- /** One supervisor arm run to diagnose. `dir` is the run dir layout the arms
67
- * write: brain.jsonl / driver.log / result.json / verify.log / ws/. */
68
- export interface SupRunArtifacts {
69
- iid: string
70
- arm: string
71
- dir: string
72
- /** The delivered (extracted) arm patch, when it exists. */
73
- patchPath?: string
74
- /** Official-judge outcome for this run, when known. `resolved: null` =
75
- * inconclusive judge — shown to analysts as such, never coerced. */
76
- judge?: { resolved: boolean | null; score?: number; note?: string }
77
- }
78
-
79
- export interface BundleOptions {
80
- /** Total bundle ceiling (chars). Default 60_000 (~15-20k tokens). */
81
- maxChars?: number
82
- /** Max workers whose ndjson/patch are excerpted per run. Default 6. */
83
- maxWorkers?: number
84
- }
85
-
86
- const head = (s: string, n: number): string => (s.length <= n ? s : `${s.slice(0, n)}\n…[truncated head]`)
87
- const tail = (s: string, n: number): string => (s.length <= n ? s : `…[truncated tail]\n${s.slice(-n)}`)
88
-
89
- async function safeRead(path: string): Promise<string> {
90
- return readFile(path, 'utf8').catch(() => '')
91
- }
92
-
93
- function section(title: string, body: string): string {
94
- const trimmed = body.trim()
95
- if (trimmed.length === 0) return ''
96
- return `--- ${title} ---\n${trimmed}\n`
97
- }
98
-
99
- /** Build the bounded shared bundle. Every section is capped so one megabyte
100
- * brain log cannot crowd out the patch the analysts must actually read. */
101
- export async function buildArtifactBundle(
102
- runs: SupRunArtifacts[],
103
- opts: BundleOptions = {},
104
- ): Promise<string> {
105
- const maxChars = opts.maxChars ?? 60_000
106
- const maxWorkers = opts.maxWorkers ?? 6
107
- const parts: string[] = []
108
- for (const r of runs) {
109
- const chunks: string[] = [`### RUN ${r.iid} arm=${r.arm}\nrun dir: ${r.dir}`]
110
- if (r.judge) {
111
- chunks.push(
112
- `official judge: resolved=${r.judge.resolved === null ? 'INCONCLUSIVE' : r.judge.resolved}` +
113
- (r.judge.score !== undefined ? ` score=${r.judge.score}` : '') +
114
- (r.judge.note ? ` (${r.judge.note})` : ''),
115
- )
116
- }
117
- chunks.push(section('result.json', head(await safeRead(join(r.dir, 'result.json')), 1_500)))
118
- chunks.push(section('verify.log (tail)', tail(await safeRead(join(r.dir, 'verify.log')), 2_000)))
119
- const driver = await safeRead(join(r.dir, 'driver.log'))
120
- chunks.push(section('driver.log (head)', head(driver, 800)))
121
- chunks.push(section('driver.log (tail)', tail(driver, 4_000)))
122
- chunks.push(section('brain.jsonl (tail)', tail(await safeRead(join(r.dir, 'brain.jsonl')), 4_000)))
123
-
124
- const supRunDir = await findSupervisorRunDir(join(r.dir, 'ws'))
125
- if (supRunDir) {
126
- chunks.push(section('supervisor state.json', head(await safeRead(join(supRunDir, 'state.json')), 3_500)))
127
- chunks.push(section('supervisor journal.jsonl (tail)', tail(await safeRead(join(supRunDir, 'journal.jsonl')), 3_000)))
128
- const workerFiles = (await readdir(join(supRunDir, 'workers')).catch(() => [] as string[])).sort()
129
- const ndjson = workerFiles.filter((f) => f.endsWith('.ndjson')).slice(0, maxWorkers)
130
- const patches = workerFiles.filter((f) => f.endsWith('.patch')).slice(0, maxWorkers)
131
- for (const f of ndjson) {
132
- chunks.push(section(`workers/${f} (tail)`, tail(await safeRead(join(supRunDir, 'workers', f)), 2_500)))
133
- }
134
- for (const f of patches) {
135
- chunks.push(section(`workers/${f} (head)`, head(await safeRead(join(supRunDir, 'workers', f)), 3_000)))
136
- }
137
- }
138
- if (r.patchPath) {
139
- chunks.push(section('DELIVERED PATCH (head)', head(await safeRead(r.patchPath), 6_000)))
140
- }
141
- parts.push(chunks.filter(Boolean).join('\n'))
142
- }
143
- const bundle = parts.join('\n\n')
144
- if (bundle.length <= maxChars) return bundle
145
- // Keep the head (earliest runs) and the tail (latest run's patch) — the
146
- // middle is the least diagnostic. Marked loudly so analysts know.
147
- const keep = Math.floor(maxChars / 2)
148
- return `${bundle.slice(0, keep)}\n\n…[BUNDLE TRUNCATED: ${bundle.length - maxChars} chars removed]…\n\n${bundle.slice(-keep)}`
149
- }
150
-
151
- // ---------------------------------------------------------------------------
152
- // One blind analyst call.
153
- // ---------------------------------------------------------------------------
154
-
155
- export interface AnalystRawFinding {
156
- failure_class: string
157
- evidence_quote: string
158
- proposed_direction: string
159
- /** 0..1, clamped. Defaults to 0.5 when the model omits it. */
160
- confidence: number
161
- }
162
-
163
- export interface AnalystReport {
164
- analystId: string
165
- model: string
166
- ok: boolean
167
- findings: AnalystRawFinding[]
168
- error?: string
169
- /** Raw model text, kept for the audit trail. */
170
- rawText?: string
171
- tokens?: { input: number; output: number }
172
- }
173
-
174
- /** The blind-analyst instruction. Exported so the calibration smoke and tests
175
- * pin the exact contract the parser expects. */
176
- export function analystPrompt(bundle: string): string {
177
- return [
178
- 'You are one independent, BLIND failure analyst (other analysts see the same artifacts; you cannot see them).',
179
- 'The artifacts below come from runs of an automated software-engineering SUPERVISOR agent on SWE-bench Verified instances:',
180
- '- a supervisor "brain" plans, spawns sandboxed workers (each in a clone of the instance repo), and settles a delivered patch;',
181
- '- each run has a self-authored verify script (worker-visible reproduction check);',
182
- '- after the run locks, the OFFICIAL hidden maintainer test suite grades the delivered patch (worker-blind);',
183
- '- "verify_pass=true but official resolved=false" means the self-check passed while the maintainers\' tests did not.',
184
- '',
185
- 'Diagnose the DOMINANT reasons these runs failed to resolve. Ground every finding in a verbatim quote from the artifacts.',
186
- 'Respond with STRICT JSON only (no markdown fences, no commentary):',
187
- '{"findings":[{"failure_class":"2-6 word category","evidence_quote":"verbatim from artifacts","proposed_direction":"concrete change direction","confidence":0.0}]}',
188
- 'Return 1 to 5 findings, most important first.',
189
- '',
190
- '===== ARTIFACTS =====',
191
- bundle,
192
- ].join('\n')
193
- }
194
-
195
- /** Extract + validate the strict-JSON findings contract from model text.
196
- * Tolerates code fences and leading/trailing prose; throws on anything that
197
- * does not contain one parseable findings object (the caller records the
198
- * analyst as failed — never a silently-empty diagnosis). */
199
- export function parseAnalystFindings(text: string): AnalystRawFinding[] {
200
- const stripped = text.replace(/```(?:json)?/gi, '').trim()
201
- const start = stripped.indexOf('{')
202
- if (start === -1) throw new Error('analyst output contains no JSON object')
203
- // Balanced-brace scan from the first '{' — models love trailing prose.
204
- let depth = 0
205
- let end = -1
206
- let inString = false
207
- let escaped = false
208
- for (let i = start; i < stripped.length; i++) {
209
- const ch = stripped[i]
210
- if (inString) {
211
- if (escaped) escaped = false
212
- else if (ch === '\\') escaped = true
213
- else if (ch === '"') inString = false
214
- continue
215
- }
216
- if (ch === '"') inString = true
217
- else if (ch === '{') depth += 1
218
- else if (ch === '}') {
219
- depth -= 1
220
- if (depth === 0) {
221
- end = i
222
- break
223
- }
224
- }
225
- }
226
- if (end === -1) throw new Error('analyst output JSON object never closes')
227
- let parsed: unknown
228
- try {
229
- parsed = JSON.parse(stripped.slice(start, end + 1))
230
- } catch (cause) {
231
- throw new Error(`analyst output is not valid JSON: ${(cause as Error).message}`)
232
- }
233
- const findings = (parsed as { findings?: unknown }).findings
234
- if (!Array.isArray(findings) || findings.length === 0) {
235
- throw new Error('analyst output has no findings array')
236
- }
237
- return findings.map((f, i) => {
238
- const o = f as Record<string, unknown>
239
- const failureClass = typeof o.failure_class === 'string' ? o.failure_class.trim() : ''
240
- if (!failureClass) throw new Error(`finding ${i} missing failure_class`)
241
- const conf = typeof o.confidence === 'number' && Number.isFinite(o.confidence) ? o.confidence : 0.5
242
- return {
243
- failure_class: failureClass,
244
- evidence_quote: typeof o.evidence_quote === 'string' ? o.evidence_quote : '',
245
- proposed_direction: typeof o.proposed_direction === 'string' ? o.proposed_direction : '',
246
- confidence: Math.min(1, Math.max(0, conf)),
247
- }
248
- })
249
- }
250
-
251
- /** Run one blind analyst through Runtime's exact-profile Router boundary.
252
- * ONE retry on transport failure (5xx/524/timeout) — an edge flake was
253
- * observed live in the calibration smoke; a parse failure is NOT retried
254
- * (same prompt, same model ⇒ same bad shape). */
255
- export async function runAnalyst(
256
- spec: AnalystSpec,
257
- bundle: string,
258
- secrets: SecretsEnv,
259
- scratchDir: string,
260
- opts: { timeoutMs?: number; retries?: number; retryDelayMs?: number; signal?: AbortSignal } = {},
261
- ): Promise<AnalystReport> {
262
- opts.signal?.throwIfAborted()
263
- const apiKeyEnv = spec.apiKeyEnv ?? 'TANGLE_API_KEY'
264
- if (!/^[A-Z_][A-Z0-9_]*$/.test(apiKeyEnv)) {
265
- return { analystId: spec.id, model: spec.model, ok: false, findings: [], error: `invalid apiKeyEnv name: ${apiKeyEnv}` }
266
- }
267
- const url = spec.url ?? ROUTER_ENDPOINT
268
- const routerKey = process.env[apiKeyEnv]
269
- if (!routerKey) {
270
- return {
271
- analystId: spec.id,
272
- model: spec.model,
273
- ok: false,
274
- findings: [],
275
- error: `${apiKeyEnv} is required; launch through dotenvx`,
276
- }
277
- }
278
- void secrets
279
- opts.signal?.throwIfAborted()
280
- await mkdir(scratchDir, { recursive: true })
281
- const outFile = join(scratchDir, `analyst-${spec.id.replace(/[^a-zA-Z0-9._-]/g, '_')}.response.json`)
282
- const timeoutMs = opts.timeoutMs ?? 600_000
283
- const attempts = 1 + (opts.retries ?? 1)
284
- let transportError = ''
285
- let content = ''
286
- let tokens: AnalystReport['tokens']
287
- for (let attempt = 0; attempt < attempts && content.length === 0; attempt++) {
288
- opts.signal?.throwIfAborted()
289
- if (attempt > 0) await sleepWithSignal(opts.retryDelayMs ?? 5_000, opts.signal)
290
- try {
291
- const turn = await runBenchRouterTurn(
292
- {
293
- routerBaseUrl: url.replace(/\/chat\/completions\/?$/u, ''),
294
- routerKey,
295
- profile: {
296
- name: `diagnosis-${spec.id}`,
297
- harness: 'cli-base',
298
- model: {
299
- provider: spec.provider ?? 'tangle-router',
300
- default: spec.model,
301
- metadata: {
302
- temperature: spec.temperature ?? 0,
303
- maxTokens: spec.maxTokens ?? 16_000,
304
- },
305
- },
306
- prompt: { systemPrompt: 'Diagnose the supplied run evidence as an independent analyst.' },
307
- },
308
- timeoutMs,
309
- signal: opts.signal,
310
- },
311
- analystPrompt(bundle),
312
- )
313
- content = turn.finalText
314
- if (turn.usage.tokensKnown !== false) {
315
- tokens = { input: turn.usage.input, output: turn.usage.output }
316
- }
317
- await writeFile(outFile, JSON.stringify(turn, null, 2))
318
- } catch (cause) {
319
- transportError = `router call failed (${(cause as Error).message}, attempt=${attempt + 1}/${attempts})`
320
- }
321
- }
322
- if (content.length === 0 && transportError) {
323
- return { analystId: spec.id, model: spec.model, ok: false, findings: [], error: transportError }
324
- }
325
- if (content.trim().length === 0) {
326
- return { analystId: spec.id, model: spec.model, ok: false, findings: [], error: 'empty content (max_tokens starvation?)', tokens }
327
- }
328
- try {
329
- const findings = parseAnalystFindings(content)
330
- return { analystId: spec.id, model: spec.model, ok: true, findings, rawText: content, tokens }
331
- } catch (cause) {
332
- return { analystId: spec.id, model: spec.model, ok: false, findings: [], error: (cause as Error).message, rawText: content, tokens }
333
- }
334
- }
335
-
336
- // ---------------------------------------------------------------------------
337
- // Fusion — union + agreement rank; minority findings survive flagged.
338
- // ---------------------------------------------------------------------------
339
-
340
- export interface FusedFinding {
341
- /** Representative label (the first-seen member's failure_class). */
342
- failure_class: string
343
- /** Distinct analysts whose findings landed in this cluster. */
344
- analysts: string[]
345
- agreement: number
346
- /** True when only ONE analyst surfaced it while others reported ok — a
347
- * surviving minority view, kept as an explicit competing hypothesis. */
348
- competingHypothesis: boolean
349
- meanConfidence: number
350
- evidence: Array<{ analyst: string; quote: string }>
351
- directions: Array<{ analyst: string; direction: string }>
352
- }
353
-
354
- const CLASS_STOPWORDS = new Set([
355
- 'the', 'a', 'an', 'of', 'in', 'on', 'to', 'and', 'or', 'is', 'for', 'with', 'by', 'at', 'not', 'no',
356
- ])
357
-
358
- /** Normalized token set of a failure-class label. */
359
- export function classTokens(label: string): string[] {
360
- return [
361
- ...new Set(
362
- label
363
- .toLowerCase()
364
- .split(/[^a-z0-9]+/)
365
- .filter((t) => t.length > 1 && !CLASS_STOPWORDS.has(t)),
366
- ),
367
- ].sort()
368
- }
369
-
370
- /** Jaccard similarity of two class labels' token sets. */
371
- export function classSimilarity(a: string, b: string): number {
372
- const ta = classTokens(a)
373
- const tb = new Set(classTokens(b))
374
- if (ta.length === 0 || tb.size === 0) return 0
375
- let inter = 0
376
- for (const t of ta) if (tb.has(t)) inter += 1
377
- return inter / (ta.length + tb.size - inter)
378
- }
379
-
380
- /**
381
- * Fuse per-analyst findings: greedy clustering on failure-class similarity
382
- * (>= threshold joins the first matching cluster), then rank by agreement
383
- * (desc) and mean confidence (desc). Every input finding survives — union, not
384
- * intersection; a single-analyst cluster is marked `competingHypothesis` when
385
- * at least one OTHER analyst produced a successful report.
386
- */
387
- export function fuseFindings(
388
- reports: AnalystReport[],
389
- opts: { similarityThreshold?: number } = {},
390
- ): FusedFinding[] {
391
- const threshold = opts.similarityThreshold ?? 0.5
392
- const okReports = reports.filter((r) => r.ok)
393
- interface Cluster {
394
- representative: string
395
- members: Array<{ analyst: string; finding: AnalystRawFinding }>
396
- }
397
- const clusters: Cluster[] = []
398
- for (const report of okReports) {
399
- for (const finding of report.findings) {
400
- const match = clusters.find((c) => classSimilarity(c.representative, finding.failure_class) >= threshold)
401
- if (match) {
402
- match.members.push({ analyst: report.analystId, finding })
403
- } else {
404
- clusters.push({ representative: finding.failure_class, members: [{ analyst: report.analystId, finding }] })
405
- }
406
- }
407
- }
408
- const fused = clusters.map((c): FusedFinding => {
409
- const analysts = [...new Set(c.members.map((m) => m.analyst))]
410
- return {
411
- failure_class: c.representative,
412
- analysts,
413
- agreement: analysts.length,
414
- competingHypothesis: analysts.length === 1 && okReports.length > 1,
415
- meanConfidence:
416
- c.members.reduce((s, m) => s + m.finding.confidence, 0) / c.members.length,
417
- evidence: c.members
418
- .filter((m) => m.finding.evidence_quote.length > 0)
419
- .map((m) => ({ analyst: m.analyst, quote: m.finding.evidence_quote })),
420
- directions: c.members
421
- .filter((m) => m.finding.proposed_direction.length > 0)
422
- .map((m) => ({ analyst: m.analyst, direction: m.finding.proposed_direction })),
423
- }
424
- })
425
- return fused.sort((a, b) => b.agreement - a.agreement || b.meanConfidence - a.meanConfidence)
426
- }
427
-
428
- // ---------------------------------------------------------------------------
429
- // The ensemble.
430
- // ---------------------------------------------------------------------------
431
-
432
- export interface DiagnosisEnsembleResult {
433
- bundleChars: number
434
- reports: AnalystReport[]
435
- fused: FusedFinding[]
436
- }
437
-
438
- /** Run every analyst (sequentially — gentle on the shared router key) over the
439
- * SAME bundle and fuse. A failed analyst is recorded, never fabricated. */
440
- export async function runDiagnosisEnsemble(input: {
441
- analysts: AnalystSpec[]
442
- runs: SupRunArtifacts[]
443
- secrets: SecretsEnv
444
- scratchDir: string
445
- maxBundleChars?: number
446
- timeoutMsPerAnalyst?: number
447
- /** Transport retries per analyst (router 524 storms are a measured, known
448
- * infra class — the loops brain itself runs with LOOPS_BRAIN_RETRIES=30). */
449
- retriesPerAnalyst?: number
450
- retryDelayMs?: number
451
- onStatus?: (msg: string) => void
452
- signal?: AbortSignal
453
- }): Promise<DiagnosisEnsembleResult> {
454
- input.signal?.throwIfAborted()
455
- if (input.analysts.length === 0) throw new Error('diagnosis ensemble: no analysts configured')
456
- if (input.runs.length === 0) throw new Error('diagnosis ensemble: no run artifacts to diagnose')
457
- const bundle = await buildArtifactBundle(
458
- input.runs,
459
- input.maxBundleChars !== undefined ? { maxChars: input.maxBundleChars } : {},
460
- )
461
- input.signal?.throwIfAborted()
462
- await mkdir(input.scratchDir, { recursive: true })
463
- await writeFile(join(input.scratchDir, 'bundle.txt'), bundle)
464
- input.signal?.throwIfAborted()
465
- const reports: AnalystReport[] = []
466
- for (const spec of input.analysts) {
467
- input.signal?.throwIfAborted()
468
- input.onStatus?.(`analyst ${spec.id} (${spec.model}) reading ${bundle.length} chars…`)
469
- const report = await runAnalyst(spec, bundle, input.secrets, input.scratchDir, {
470
- ...(input.timeoutMsPerAnalyst !== undefined ? { timeoutMs: input.timeoutMsPerAnalyst } : {}),
471
- ...(input.retriesPerAnalyst !== undefined ? { retries: input.retriesPerAnalyst } : {}),
472
- ...(input.retryDelayMs !== undefined ? { retryDelayMs: input.retryDelayMs } : {}),
473
- ...(input.signal ? { signal: input.signal } : {}),
474
- })
475
- input.signal?.throwIfAborted()
476
- input.onStatus?.(
477
- `analyst ${spec.id}: ${report.ok ? `${report.findings.length} finding(s)` : `FAILED (${report.error})`}` +
478
- (report.tokens ? ` [tokens in=${report.tokens.input} out=${report.tokens.output}]` : ''),
479
- )
480
- reports.push(report)
481
- }
482
- return { bundleChars: bundle.length, reports, fused: fuseFindings(reports) }
483
- }
484
-
485
- /** Calibration grading: does a finding surface FIX PLACEMENT (the known truth
486
- * of the round-2 django run — a locally-authored helper where the gold fix
487
- * extends django/utils/encoding.py)? Assistive keyword net; the smoke always
488
- * prints the raw findings so a miss here is auditable, never load-bearing. */
489
- export function surfacesPlacementRegex(): RegExp {
490
- return /placement|misplac|wrong\s+(file|location|module|place)|different\s+(file|location|module)|expected\s+(location|file|module)|local\s+(helper|copy|implementation)|duplicat|encoding\.py|django\.utils\.encoding|utils\/encoding|belongs\s+in|should\s+(live|be\s+placed|be\s+located|go)\s+in|reimplement|instead\s+of\s+(the\s+)?(existing|shared|upstream)/i
491
- }
492
-
493
- /** Map fused findings onto the substrate's `AnalystFinding` envelope so they
494
- * drop into the same `analyzeGeneration` slot the raw-trace distiller uses. */
495
- export function fusedToAnalystFindings(
496
- fused: FusedFinding[],
497
- evidence: { dirs: string[]; totalAnalysts: number },
498
- ): AnalystFinding[] {
499
- return fused.map((f) => {
500
- const directions = f.directions.map((d) => `[${d.analyst}] ${d.direction}`).join(' | ')
501
- const quotes = f.evidence
502
- .slice(0, 3)
503
- .map((e) => `[${e.analyst}] "${e.quote.slice(0, 240)}"`)
504
- .join('; ')
505
- return makeFinding({
506
- analyst_id: 'diagnosis-ensemble',
507
- severity: f.agreement >= 2 ? 'high' : 'medium',
508
- area: 'failure-diagnosis',
509
- confidence: Math.min(1, Math.max(0, f.meanConfidence)),
510
- claim:
511
- `${f.competingHypothesis ? '[competing hypothesis] ' : ''}` +
512
- `(${f.agreement}/${evidence.totalAnalysts} analysts) ${f.failure_class}` +
513
- (quotes ? ` — evidence: ${quotes}` : ''),
514
- ...(directions ? { recommended_action: directions } : {}),
515
- evidence_refs: evidence.dirs.map((d) => ({ kind: 'artifact' as const, uri: d })),
516
- metadata: {
517
- agreement: f.agreement,
518
- competingHypothesis: f.competingHypothesis,
519
- analysts: f.analysts,
520
- },
521
- })
522
- })
523
- }