@tangle-network/agent-bench 0.11.3 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. package/CHANGELOG.md +22 -0
  2. package/HARNESS.md +6 -2
  3. package/README.md +1 -4
  4. package/package.json +5 -5
  5. package/scripts/run-package-tests.mjs +2 -2
  6. package/src/quant-arena/README.md +0 -144
  7. package/src/quant-arena/backtest.test.mts +0 -135
  8. package/src/quant-arena/backtest.ts +0 -218
  9. package/src/quant-arena/data.test.mts +0 -44
  10. package/src/quant-arena/data.ts +0 -141
  11. package/src/quant-arena/driver.test.mts +0 -253
  12. package/src/quant-arena/driver.ts +0 -219
  13. package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
  14. package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
  15. package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
  16. package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
  17. package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
  18. package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
  19. package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
  20. package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
  21. package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
  22. package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
  23. package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
  24. package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
  25. package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
  26. package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
  27. package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
  28. package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
  29. package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
  30. package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
  31. package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
  32. package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
  33. package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
  34. package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
  35. package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
  36. package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
  37. package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
  38. package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
  39. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
  40. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
  41. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
  42. package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
  43. package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
  44. package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
  45. package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
  46. package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
  47. package/src/quant-arena/holdout-certify.mts +0 -206
  48. package/src/quant-arena/holdout-certify.test.mts +0 -82
  49. package/src/quant-arena/leak-audit.test.mts +0 -79
  50. package/src/quant-arena/leak-audit.ts +0 -95
  51. package/src/quant-arena/make-fixtures.mts +0 -161
  52. package/src/quant-arena/multiplicity.test.mts +0 -68
  53. package/src/quant-arena/multiplicity.ts +0 -87
  54. package/src/quant-arena/nautilus-certify.ts +0 -31
  55. package/src/quant-arena/oms.ts +0 -90
  56. package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
  57. package/src/quant-arena/python/pyproject.toml +0 -8
  58. package/src/quant-arena/python/uv.lock +0 -1297
  59. package/src/quant-arena/python/vbt-worker.py +0 -192
  60. package/src/quant-arena/quant-loop.mts +0 -840
  61. package/src/quant-arena/quant-loop.test.mts +0 -75
  62. package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
  63. package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
  64. package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
  65. package/src/quant-arena/types.ts +0 -133
  66. package/src/quant-arena/vbt-client.ts +0 -321
  67. package/src/quant-arena/vbt-parity.test.mts +0 -183
  68. package/src/quant-arena/windows.test.mts +0 -45
  69. package/src/quant-arena/windows.ts +0 -54
  70. package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
  71. package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
  72. package/src/rollout-ledger/settle-capture.mts +0 -448
  73. package/src/rollout-ledger/settle-capture.test.mts +0 -270
  74. package/src/swe-arena/activation.mts +0 -225
  75. package/src/swe-arena/activation.test.mts +0 -300
  76. package/src/swe-arena/analyze.ts +0 -211
  77. package/src/swe-arena/arms.ts +0 -862
  78. package/src/swe-arena/bootstrap-meta.mts +0 -188
  79. package/src/swe-arena/bootstrap-meta.test.mts +0 -51
  80. package/src/swe-arena/briefing.mts +0 -217
  81. package/src/swe-arena/briefing.test.mts +0 -179
  82. package/src/swe-arena/calibrate.ts +0 -217
  83. package/src/swe-arena/capabilities.mts +0 -76
  84. package/src/swe-arena/capabilities.test.mts +0 -57
  85. package/src/swe-arena/capacity.ts +0 -198
  86. package/src/swe-arena/cell-evidence.mts +0 -437
  87. package/src/swe-arena/cell-evidence.test.mts +0 -248
  88. package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
  89. package/src/swe-arena/diagnosis-ensemble.ts +0 -523
  90. package/src/swe-arena/execution.test.mts +0 -1171
  91. package/src/swe-arena/factory-command-container.ts +0 -284
  92. package/src/swe-arena/factory-judge-child.mts +0 -228
  93. package/src/swe-arena/factory.test.mts +0 -645
  94. package/src/swe-arena/fixtures/analyze.py +0 -80
  95. package/src/swe-arena/fixtures/excludes.txt +0 -8
  96. package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
  97. package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
  98. package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
  99. package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
  100. package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
  101. package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
  102. package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
  103. package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
  104. package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
  105. package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
  106. package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
  107. package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
  108. package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
  109. package/src/swe-arena/fixtures/holdout.json +0 -44
  110. package/src/swe-arena/fixtures/instances.json +0 -146
  111. package/src/swe-arena/fixtures/ledger.jsonl +0 -12
  112. package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
  113. package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
  114. package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
  115. package/src/swe-arena/fixtures/rematch.jsonl +0 -3
  116. package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
  117. package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
  118. package/src/swe-arena/fixtures/run-report/README.md +0 -43
  119. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
  120. package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
  121. package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
  122. package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
  123. package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
  124. package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
  125. package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
  126. package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
  127. package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
  128. package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
  129. package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
  130. package/src/swe-arena/fixtures/worker-tokens.json +0 -42
  131. package/src/swe-arena/fixtures.ts +0 -237
  132. package/src/swe-arena/gepa-seat.mts +0 -886
  133. package/src/swe-arena/gepa-seat.test.mts +0 -1136
  134. package/src/swe-arena/holdout-certify.mts +0 -408
  135. package/src/swe-arena/holdout-certify.test.mts +0 -160
  136. package/src/swe-arena/implementation-ref.test.mts +0 -64
  137. package/src/swe-arena/implementation-ref.ts +0 -62
  138. package/src/swe-arena/judge-child.mts +0 -37
  139. package/src/swe-arena/ledger-orphans.mts +0 -77
  140. package/src/swe-arena/ledger-orphans.test.mts +0 -149
  141. package/src/swe-arena/manifest.mts +0 -293
  142. package/src/swe-arena/manifest.test.mts +0 -169
  143. package/src/swe-arena/materialize.ts +0 -142
  144. package/src/swe-arena/outer-loop.mts +0 -2854
  145. package/src/swe-arena/outer-loop.test.mts +0 -714
  146. package/src/swe-arena/parity.test.mts +0 -87
  147. package/src/swe-arena/premeasured-from-cells.mts +0 -296
  148. package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
  149. package/src/swe-arena/proc.test.mts +0 -172
  150. package/src/swe-arena/proc.ts +0 -260
  151. package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
  152. package/src/swe-arena/profiles/default-author.profile.json +0 -12
  153. package/src/swe-arena/proposer-fanout.mts +0 -736
  154. package/src/swe-arena/proposer-fanout.test.mts +0 -660
  155. package/src/swe-arena/proposer-provenance.mts +0 -176
  156. package/src/swe-arena/proposer-provenance.test.mts +0 -106
  157. package/src/swe-arena/reconcile.ts +0 -0
  158. package/src/swe-arena/replay.mts +0 -183
  159. package/src/swe-arena/replay.test.mts +0 -300
  160. package/src/swe-arena/run-experiment.mts +0 -729
  161. package/src/swe-arena/run-report.mts +0 -75
  162. package/src/swe-arena/run-supervisor.mjs +0 -297
  163. package/src/swe-arena/run-supervisor.test.mts +0 -539
  164. package/src/swe-arena/score-split.mts +0 -140
  165. package/src/swe-arena/score-split.test.mts +0 -123
  166. package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
  167. package/src/swe-arena/scratch-worktree.test.mts +0 -56
  168. package/src/swe-arena/scratch-worktree.ts +0 -64
  169. package/src/swe-arena/serialized-judge.ts +0 -414
  170. package/src/swe-arena/types.ts +0 -218
@@ -1,523 +0,0 @@
1
- /**
2
- * Diagnosis ensemble — N BLIND failure analysts over the SAME supervisor-run
3
- * artifacts (brain.jsonl, driver.log, workers/*.ndjson + *.patch, verify.log,
4
- * result.json, the delivered patch, and the official-judge outcome), each a
5
- * different model routed through router.tangle.tools, fused by union +
6
- * cross-analyst agreement ranking. Minority findings SURVIVE fusion marked as
7
- * competing hypotheses — the round-4 protocol treats them as candidate
8
- * proposal seeds, not noise.
9
- *
10
- * Design constraints (from supervisor-lab .evolve/state.json round4_design):
11
- * - the model list is CONFIG — glm-5.2 is the only model proven routed today;
12
- * gpt-5.5 / opus-4.8 slot in by editing the analyst spec list, never by a
13
- * hardcoded requirement on an unrouted model.
14
- * - analysts are blind: same bundle, no cross-talk, independent calls.
15
- * - the containing run is launched through dotenvx; every analyst enters Runtime through one
16
- * exact AgentProfile and its event record is retained with the response.
17
- */
18
-
19
- import { mkdir, readdir, readFile, writeFile } from 'node:fs/promises'
20
- import { join } from 'node:path'
21
- import { type AnalystFinding, makeFinding } from '@tangle-network/agent-eval'
22
- import { findSupervisorRunDir, type SecretsEnv } from './arms'
23
- import { ROUTER_ENDPOINT, sleepWithSignal } from './capacity'
24
- import { runBenchRouterTurn } from '../router-turn'
25
-
26
- // ---------------------------------------------------------------------------
27
- // Analyst specs.
28
- // ---------------------------------------------------------------------------
29
-
30
- export interface AnalystSpec {
31
- /** Stable label, e.g. 'glm-5.2#1'. Used in fusion attribution + filenames. */
32
- id: string
33
- model: string
34
- /** Provider identity stamped into the exact profile. Default: tangle-router. */
35
- provider?: string
36
- /** Chat-completions endpoint. Default: the Tangle router. */
37
- url?: string
38
- /** NAME of the env var holding the bearer key (resolved in the dotenvx child). */
39
- apiKeyEnv?: string
40
- /** glm-5.2 returns empty content when starved below ~8000, and at sampled
41
- * temperatures its REASONING alone can consume a full 8000 ceiling (both
42
- * measured — the calibration smoke saw out=8000 with empty content).
43
- * Default 16_000 so reasoning + the JSON answer always fit. */
44
- maxTokens?: number
45
- temperature?: number
46
- }
47
-
48
- /** N same-model analysts (the smoke default). Real rounds replace entries with
49
- * other routed models — diversity comes from the config, never a hardcode.
50
- * Same-model analysts at temperature 0 are byte-identical duplicates (measured
51
- * in the calibration smoke: #2 and #3 returned the same 2823 output tokens),
52
- * which silently inflates agreement — so only the FIRST same-model analyst
53
- * runs at 0; the rest sample at 0.7 to buy real diversity. */
54
- export function defaultAnalysts(n = 3, model = 'glm-5.2'): AnalystSpec[] {
55
- return Array.from({ length: n }, (_, i) => ({
56
- id: `${model}#${i + 1}`,
57
- model,
58
- temperature: i === 0 ? 0 : 0.7,
59
- }))
60
- }
61
-
62
- // ---------------------------------------------------------------------------
63
- // Artifact bundle — the ONE shared context every blind analyst reads.
64
- // ---------------------------------------------------------------------------
65
-
66
- /** One supervisor arm run to diagnose. `dir` is the run dir layout the arms
67
- * write: brain.jsonl / driver.log / result.json / verify.log / ws/. */
68
- export interface SupRunArtifacts {
69
- iid: string
70
- arm: string
71
- dir: string
72
- /** The delivered (extracted) arm patch, when it exists. */
73
- patchPath?: string
74
- /** Official-judge outcome for this run, when known. `resolved: null` =
75
- * inconclusive judge — shown to analysts as such, never coerced. */
76
- judge?: { resolved: boolean | null; score?: number; note?: string }
77
- }
78
-
79
- export interface BundleOptions {
80
- /** Total bundle ceiling (chars). Default 60_000 (~15-20k tokens). */
81
- maxChars?: number
82
- /** Max workers whose ndjson/patch are excerpted per run. Default 6. */
83
- maxWorkers?: number
84
- }
85
-
86
- const head = (s: string, n: number): string => (s.length <= n ? s : `${s.slice(0, n)}\n…[truncated head]`)
87
- const tail = (s: string, n: number): string => (s.length <= n ? s : `…[truncated tail]\n${s.slice(-n)}`)
88
-
89
- async function safeRead(path: string): Promise<string> {
90
- return readFile(path, 'utf8').catch(() => '')
91
- }
92
-
93
- function section(title: string, body: string): string {
94
- const trimmed = body.trim()
95
- if (trimmed.length === 0) return ''
96
- return `--- ${title} ---\n${trimmed}\n`
97
- }
98
-
99
- /** Build the bounded shared bundle. Every section is capped so one megabyte
100
- * brain log cannot crowd out the patch the analysts must actually read. */
101
- export async function buildArtifactBundle(
102
- runs: SupRunArtifacts[],
103
- opts: BundleOptions = {},
104
- ): Promise<string> {
105
- const maxChars = opts.maxChars ?? 60_000
106
- const maxWorkers = opts.maxWorkers ?? 6
107
- const parts: string[] = []
108
- for (const r of runs) {
109
- const chunks: string[] = [`### RUN ${r.iid} arm=${r.arm}\nrun dir: ${r.dir}`]
110
- if (r.judge) {
111
- chunks.push(
112
- `official judge: resolved=${r.judge.resolved === null ? 'INCONCLUSIVE' : r.judge.resolved}` +
113
- (r.judge.score !== undefined ? ` score=${r.judge.score}` : '') +
114
- (r.judge.note ? ` (${r.judge.note})` : ''),
115
- )
116
- }
117
- chunks.push(section('result.json', head(await safeRead(join(r.dir, 'result.json')), 1_500)))
118
- chunks.push(section('verify.log (tail)', tail(await safeRead(join(r.dir, 'verify.log')), 2_000)))
119
- const driver = await safeRead(join(r.dir, 'driver.log'))
120
- chunks.push(section('driver.log (head)', head(driver, 800)))
121
- chunks.push(section('driver.log (tail)', tail(driver, 4_000)))
122
- chunks.push(section('brain.jsonl (tail)', tail(await safeRead(join(r.dir, 'brain.jsonl')), 4_000)))
123
-
124
- const supRunDir = await findSupervisorRunDir(join(r.dir, 'ws'))
125
- if (supRunDir) {
126
- chunks.push(section('supervisor state.json', head(await safeRead(join(supRunDir, 'state.json')), 3_500)))
127
- chunks.push(section('supervisor journal.jsonl (tail)', tail(await safeRead(join(supRunDir, 'journal.jsonl')), 3_000)))
128
- const workerFiles = (await readdir(join(supRunDir, 'workers')).catch(() => [] as string[])).sort()
129
- const ndjson = workerFiles.filter((f) => f.endsWith('.ndjson')).slice(0, maxWorkers)
130
- const patches = workerFiles.filter((f) => f.endsWith('.patch')).slice(0, maxWorkers)
131
- for (const f of ndjson) {
132
- chunks.push(section(`workers/${f} (tail)`, tail(await safeRead(join(supRunDir, 'workers', f)), 2_500)))
133
- }
134
- for (const f of patches) {
135
- chunks.push(section(`workers/${f} (head)`, head(await safeRead(join(supRunDir, 'workers', f)), 3_000)))
136
- }
137
- }
138
- if (r.patchPath) {
139
- chunks.push(section('DELIVERED PATCH (head)', head(await safeRead(r.patchPath), 6_000)))
140
- }
141
- parts.push(chunks.filter(Boolean).join('\n'))
142
- }
143
- const bundle = parts.join('\n\n')
144
- if (bundle.length <= maxChars) return bundle
145
- // Keep the head (earliest runs) and the tail (latest run's patch) — the
146
- // middle is the least diagnostic. Marked loudly so analysts know.
147
- const keep = Math.floor(maxChars / 2)
148
- return `${bundle.slice(0, keep)}\n\n…[BUNDLE TRUNCATED: ${bundle.length - maxChars} chars removed]…\n\n${bundle.slice(-keep)}`
149
- }
150
-
151
- // ---------------------------------------------------------------------------
152
- // One blind analyst call.
153
- // ---------------------------------------------------------------------------
154
-
155
- export interface AnalystRawFinding {
156
- failure_class: string
157
- evidence_quote: string
158
- proposed_direction: string
159
- /** 0..1, clamped. Defaults to 0.5 when the model omits it. */
160
- confidence: number
161
- }
162
-
163
- export interface AnalystReport {
164
- analystId: string
165
- model: string
166
- ok: boolean
167
- findings: AnalystRawFinding[]
168
- error?: string
169
- /** Raw model text, kept for the audit trail. */
170
- rawText?: string
171
- tokens?: { input: number; output: number }
172
- }
173
-
174
- /** The blind-analyst instruction. Exported so the calibration smoke and tests
175
- * pin the exact contract the parser expects. */
176
- export function analystPrompt(bundle: string): string {
177
- return [
178
- 'You are one independent, BLIND failure analyst (other analysts see the same artifacts; you cannot see them).',
179
- 'The artifacts below come from runs of an automated software-engineering SUPERVISOR agent on SWE-bench Verified instances:',
180
- '- a supervisor "brain" plans, spawns sandboxed workers (each in a clone of the instance repo), and settles a delivered patch;',
181
- '- each run has a self-authored verify script (worker-visible reproduction check);',
182
- '- after the run locks, the OFFICIAL hidden maintainer test suite grades the delivered patch (worker-blind);',
183
- '- "verify_pass=true but official resolved=false" means the self-check passed while the maintainers\' tests did not.',
184
- '',
185
- 'Diagnose the DOMINANT reasons these runs failed to resolve. Ground every finding in a verbatim quote from the artifacts.',
186
- 'Respond with STRICT JSON only (no markdown fences, no commentary):',
187
- '{"findings":[{"failure_class":"2-6 word category","evidence_quote":"verbatim from artifacts","proposed_direction":"concrete change direction","confidence":0.0}]}',
188
- 'Return 1 to 5 findings, most important first.',
189
- '',
190
- '===== ARTIFACTS =====',
191
- bundle,
192
- ].join('\n')
193
- }
194
-
195
- /** Extract + validate the strict-JSON findings contract from model text.
196
- * Tolerates code fences and leading/trailing prose; throws on anything that
197
- * does not contain one parseable findings object (the caller records the
198
- * analyst as failed — never a silently-empty diagnosis). */
199
- export function parseAnalystFindings(text: string): AnalystRawFinding[] {
200
- const stripped = text.replace(/```(?:json)?/gi, '').trim()
201
- const start = stripped.indexOf('{')
202
- if (start === -1) throw new Error('analyst output contains no JSON object')
203
- // Balanced-brace scan from the first '{' — models love trailing prose.
204
- let depth = 0
205
- let end = -1
206
- let inString = false
207
- let escaped = false
208
- for (let i = start; i < stripped.length; i++) {
209
- const ch = stripped[i]
210
- if (inString) {
211
- if (escaped) escaped = false
212
- else if (ch === '\\') escaped = true
213
- else if (ch === '"') inString = false
214
- continue
215
- }
216
- if (ch === '"') inString = true
217
- else if (ch === '{') depth += 1
218
- else if (ch === '}') {
219
- depth -= 1
220
- if (depth === 0) {
221
- end = i
222
- break
223
- }
224
- }
225
- }
226
- if (end === -1) throw new Error('analyst output JSON object never closes')
227
- let parsed: unknown
228
- try {
229
- parsed = JSON.parse(stripped.slice(start, end + 1))
230
- } catch (cause) {
231
- throw new Error(`analyst output is not valid JSON: ${(cause as Error).message}`)
232
- }
233
- const findings = (parsed as { findings?: unknown }).findings
234
- if (!Array.isArray(findings) || findings.length === 0) {
235
- throw new Error('analyst output has no findings array')
236
- }
237
- return findings.map((f, i) => {
238
- const o = f as Record<string, unknown>
239
- const failureClass = typeof o.failure_class === 'string' ? o.failure_class.trim() : ''
240
- if (!failureClass) throw new Error(`finding ${i} missing failure_class`)
241
- const conf = typeof o.confidence === 'number' && Number.isFinite(o.confidence) ? o.confidence : 0.5
242
- return {
243
- failure_class: failureClass,
244
- evidence_quote: typeof o.evidence_quote === 'string' ? o.evidence_quote : '',
245
- proposed_direction: typeof o.proposed_direction === 'string' ? o.proposed_direction : '',
246
- confidence: Math.min(1, Math.max(0, conf)),
247
- }
248
- })
249
- }
250
-
251
- /** Run one blind analyst through Runtime's exact-profile Router boundary.
252
- * ONE retry on transport failure (5xx/524/timeout) — an edge flake was
253
- * observed live in the calibration smoke; a parse failure is NOT retried
254
- * (same prompt, same model ⇒ same bad shape). */
255
- export async function runAnalyst(
256
- spec: AnalystSpec,
257
- bundle: string,
258
- secrets: SecretsEnv,
259
- scratchDir: string,
260
- opts: { timeoutMs?: number; retries?: number; retryDelayMs?: number; signal?: AbortSignal } = {},
261
- ): Promise<AnalystReport> {
262
- opts.signal?.throwIfAborted()
263
- const apiKeyEnv = spec.apiKeyEnv ?? 'TANGLE_API_KEY'
264
- if (!/^[A-Z_][A-Z0-9_]*$/.test(apiKeyEnv)) {
265
- return { analystId: spec.id, model: spec.model, ok: false, findings: [], error: `invalid apiKeyEnv name: ${apiKeyEnv}` }
266
- }
267
- const url = spec.url ?? ROUTER_ENDPOINT
268
- const routerKey = process.env[apiKeyEnv]
269
- if (!routerKey) {
270
- return {
271
- analystId: spec.id,
272
- model: spec.model,
273
- ok: false,
274
- findings: [],
275
- error: `${apiKeyEnv} is required; launch through dotenvx`,
276
- }
277
- }
278
- void secrets
279
- opts.signal?.throwIfAborted()
280
- await mkdir(scratchDir, { recursive: true })
281
- const outFile = join(scratchDir, `analyst-${spec.id.replace(/[^a-zA-Z0-9._-]/g, '_')}.response.json`)
282
- const timeoutMs = opts.timeoutMs ?? 600_000
283
- const attempts = 1 + (opts.retries ?? 1)
284
- let transportError = ''
285
- let content = ''
286
- let tokens: AnalystReport['tokens']
287
- for (let attempt = 0; attempt < attempts && content.length === 0; attempt++) {
288
- opts.signal?.throwIfAborted()
289
- if (attempt > 0) await sleepWithSignal(opts.retryDelayMs ?? 5_000, opts.signal)
290
- try {
291
- const turn = await runBenchRouterTurn(
292
- {
293
- routerBaseUrl: url.replace(/\/chat\/completions\/?$/u, ''),
294
- routerKey,
295
- profile: {
296
- name: `diagnosis-${spec.id}`,
297
- harness: 'cli-base',
298
- model: {
299
- provider: spec.provider ?? 'tangle-router',
300
- default: spec.model,
301
- metadata: {
302
- temperature: spec.temperature ?? 0,
303
- maxTokens: spec.maxTokens ?? 16_000,
304
- },
305
- },
306
- prompt: { systemPrompt: 'Diagnose the supplied run evidence as an independent analyst.' },
307
- },
308
- timeoutMs,
309
- signal: opts.signal,
310
- },
311
- analystPrompt(bundle),
312
- )
313
- content = turn.finalText
314
- if (turn.usage.tokensKnown !== false) {
315
- tokens = { input: turn.usage.input, output: turn.usage.output }
316
- }
317
- await writeFile(outFile, JSON.stringify(turn, null, 2))
318
- } catch (cause) {
319
- transportError = `router call failed (${(cause as Error).message}, attempt=${attempt + 1}/${attempts})`
320
- }
321
- }
322
- if (content.length === 0 && transportError) {
323
- return { analystId: spec.id, model: spec.model, ok: false, findings: [], error: transportError }
324
- }
325
- if (content.trim().length === 0) {
326
- return { analystId: spec.id, model: spec.model, ok: false, findings: [], error: 'empty content (max_tokens starvation?)', tokens }
327
- }
328
- try {
329
- const findings = parseAnalystFindings(content)
330
- return { analystId: spec.id, model: spec.model, ok: true, findings, rawText: content, tokens }
331
- } catch (cause) {
332
- return { analystId: spec.id, model: spec.model, ok: false, findings: [], error: (cause as Error).message, rawText: content, tokens }
333
- }
334
- }
335
-
336
- // ---------------------------------------------------------------------------
337
- // Fusion — union + agreement rank; minority findings survive flagged.
338
- // ---------------------------------------------------------------------------
339
-
340
- export interface FusedFinding {
341
- /** Representative label (the first-seen member's failure_class). */
342
- failure_class: string
343
- /** Distinct analysts whose findings landed in this cluster. */
344
- analysts: string[]
345
- agreement: number
346
- /** True when only ONE analyst surfaced it while others reported ok — a
347
- * surviving minority view, kept as an explicit competing hypothesis. */
348
- competingHypothesis: boolean
349
- meanConfidence: number
350
- evidence: Array<{ analyst: string; quote: string }>
351
- directions: Array<{ analyst: string; direction: string }>
352
- }
353
-
354
- const CLASS_STOPWORDS = new Set([
355
- 'the', 'a', 'an', 'of', 'in', 'on', 'to', 'and', 'or', 'is', 'for', 'with', 'by', 'at', 'not', 'no',
356
- ])
357
-
358
- /** Normalized token set of a failure-class label. */
359
- export function classTokens(label: string): string[] {
360
- return [
361
- ...new Set(
362
- label
363
- .toLowerCase()
364
- .split(/[^a-z0-9]+/)
365
- .filter((t) => t.length > 1 && !CLASS_STOPWORDS.has(t)),
366
- ),
367
- ].sort()
368
- }
369
-
370
- /** Jaccard similarity of two class labels' token sets. */
371
- export function classSimilarity(a: string, b: string): number {
372
- const ta = classTokens(a)
373
- const tb = new Set(classTokens(b))
374
- if (ta.length === 0 || tb.size === 0) return 0
375
- let inter = 0
376
- for (const t of ta) if (tb.has(t)) inter += 1
377
- return inter / (ta.length + tb.size - inter)
378
- }
379
-
380
- /**
381
- * Fuse per-analyst findings: greedy clustering on failure-class similarity
382
- * (>= threshold joins the first matching cluster), then rank by agreement
383
- * (desc) and mean confidence (desc). Every input finding survives — union, not
384
- * intersection; a single-analyst cluster is marked `competingHypothesis` when
385
- * at least one OTHER analyst produced a successful report.
386
- */
387
- export function fuseFindings(
388
- reports: AnalystReport[],
389
- opts: { similarityThreshold?: number } = {},
390
- ): FusedFinding[] {
391
- const threshold = opts.similarityThreshold ?? 0.5
392
- const okReports = reports.filter((r) => r.ok)
393
- interface Cluster {
394
- representative: string
395
- members: Array<{ analyst: string; finding: AnalystRawFinding }>
396
- }
397
- const clusters: Cluster[] = []
398
- for (const report of okReports) {
399
- for (const finding of report.findings) {
400
- const match = clusters.find((c) => classSimilarity(c.representative, finding.failure_class) >= threshold)
401
- if (match) {
402
- match.members.push({ analyst: report.analystId, finding })
403
- } else {
404
- clusters.push({ representative: finding.failure_class, members: [{ analyst: report.analystId, finding }] })
405
- }
406
- }
407
- }
408
- const fused = clusters.map((c): FusedFinding => {
409
- const analysts = [...new Set(c.members.map((m) => m.analyst))]
410
- return {
411
- failure_class: c.representative,
412
- analysts,
413
- agreement: analysts.length,
414
- competingHypothesis: analysts.length === 1 && okReports.length > 1,
415
- meanConfidence:
416
- c.members.reduce((s, m) => s + m.finding.confidence, 0) / c.members.length,
417
- evidence: c.members
418
- .filter((m) => m.finding.evidence_quote.length > 0)
419
- .map((m) => ({ analyst: m.analyst, quote: m.finding.evidence_quote })),
420
- directions: c.members
421
- .filter((m) => m.finding.proposed_direction.length > 0)
422
- .map((m) => ({ analyst: m.analyst, direction: m.finding.proposed_direction })),
423
- }
424
- })
425
- return fused.sort((a, b) => b.agreement - a.agreement || b.meanConfidence - a.meanConfidence)
426
- }
427
-
428
- // ---------------------------------------------------------------------------
429
- // The ensemble.
430
- // ---------------------------------------------------------------------------
431
-
432
- export interface DiagnosisEnsembleResult {
433
- bundleChars: number
434
- reports: AnalystReport[]
435
- fused: FusedFinding[]
436
- }
437
-
438
- /** Run every analyst (sequentially — gentle on the shared router key) over the
439
- * SAME bundle and fuse. A failed analyst is recorded, never fabricated. */
440
- export async function runDiagnosisEnsemble(input: {
441
- analysts: AnalystSpec[]
442
- runs: SupRunArtifacts[]
443
- secrets: SecretsEnv
444
- scratchDir: string
445
- maxBundleChars?: number
446
- timeoutMsPerAnalyst?: number
447
- /** Transport retries per analyst (router 524 storms are a measured, known
448
- * infra class — the loops brain itself runs with LOOPS_BRAIN_RETRIES=30). */
449
- retriesPerAnalyst?: number
450
- retryDelayMs?: number
451
- onStatus?: (msg: string) => void
452
- signal?: AbortSignal
453
- }): Promise<DiagnosisEnsembleResult> {
454
- input.signal?.throwIfAborted()
455
- if (input.analysts.length === 0) throw new Error('diagnosis ensemble: no analysts configured')
456
- if (input.runs.length === 0) throw new Error('diagnosis ensemble: no run artifacts to diagnose')
457
- const bundle = await buildArtifactBundle(
458
- input.runs,
459
- input.maxBundleChars !== undefined ? { maxChars: input.maxBundleChars } : {},
460
- )
461
- input.signal?.throwIfAborted()
462
- await mkdir(input.scratchDir, { recursive: true })
463
- await writeFile(join(input.scratchDir, 'bundle.txt'), bundle)
464
- input.signal?.throwIfAborted()
465
- const reports: AnalystReport[] = []
466
- for (const spec of input.analysts) {
467
- input.signal?.throwIfAborted()
468
- input.onStatus?.(`analyst ${spec.id} (${spec.model}) reading ${bundle.length} chars…`)
469
- const report = await runAnalyst(spec, bundle, input.secrets, input.scratchDir, {
470
- ...(input.timeoutMsPerAnalyst !== undefined ? { timeoutMs: input.timeoutMsPerAnalyst } : {}),
471
- ...(input.retriesPerAnalyst !== undefined ? { retries: input.retriesPerAnalyst } : {}),
472
- ...(input.retryDelayMs !== undefined ? { retryDelayMs: input.retryDelayMs } : {}),
473
- ...(input.signal ? { signal: input.signal } : {}),
474
- })
475
- input.signal?.throwIfAborted()
476
- input.onStatus?.(
477
- `analyst ${spec.id}: ${report.ok ? `${report.findings.length} finding(s)` : `FAILED (${report.error})`}` +
478
- (report.tokens ? ` [tokens in=${report.tokens.input} out=${report.tokens.output}]` : ''),
479
- )
480
- reports.push(report)
481
- }
482
- return { bundleChars: bundle.length, reports, fused: fuseFindings(reports) }
483
- }
484
-
485
- /** Calibration grading: does a finding surface FIX PLACEMENT (the known truth
486
- * of the round-2 django run — a locally-authored helper where the gold fix
487
- * extends django/utils/encoding.py)? Assistive keyword net; the smoke always
488
- * prints the raw findings so a miss here is auditable, never load-bearing. */
489
- export function surfacesPlacementRegex(): RegExp {
490
- return /placement|misplac|wrong\s+(file|location|module|place)|different\s+(file|location|module)|expected\s+(location|file|module)|local\s+(helper|copy|implementation)|duplicat|encoding\.py|django\.utils\.encoding|utils\/encoding|belongs\s+in|should\s+(live|be\s+placed|be\s+located|go)\s+in|reimplement|instead\s+of\s+(the\s+)?(existing|shared|upstream)/i
491
- }
492
-
493
- /** Map fused findings onto the substrate's `AnalystFinding` envelope so they
494
- * drop into the same `analyzeGeneration` slot the raw-trace distiller uses. */
495
- export function fusedToAnalystFindings(
496
- fused: FusedFinding[],
497
- evidence: { dirs: string[]; totalAnalysts: number },
498
- ): AnalystFinding[] {
499
- return fused.map((f) => {
500
- const directions = f.directions.map((d) => `[${d.analyst}] ${d.direction}`).join(' | ')
501
- const quotes = f.evidence
502
- .slice(0, 3)
503
- .map((e) => `[${e.analyst}] "${e.quote.slice(0, 240)}"`)
504
- .join('; ')
505
- return makeFinding({
506
- analyst_id: 'diagnosis-ensemble',
507
- severity: f.agreement >= 2 ? 'high' : 'medium',
508
- area: 'failure-diagnosis',
509
- confidence: Math.min(1, Math.max(0, f.meanConfidence)),
510
- claim:
511
- `${f.competingHypothesis ? '[competing hypothesis] ' : ''}` +
512
- `(${f.agreement}/${evidence.totalAnalysts} analysts) ${f.failure_class}` +
513
- (quotes ? ` — evidence: ${quotes}` : ''),
514
- ...(directions ? { recommended_action: directions } : {}),
515
- evidence_refs: evidence.dirs.map((d) => ({ kind: 'artifact' as const, uri: d })),
516
- metadata: {
517
- agreement: f.agreement,
518
- competingHypothesis: f.competingHypothesis,
519
- analysts: f.analysts,
520
- },
521
- })
522
- })
523
- }