@tangle-network/agent-bench 0.11.3 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +22 -0
- package/HARNESS.md +6 -2
- package/README.md +1 -4
- package/package.json +5 -5
- package/scripts/run-package-tests.mjs +2 -2
- package/src/quant-arena/README.md +0 -144
- package/src/quant-arena/backtest.test.mts +0 -135
- package/src/quant-arena/backtest.ts +0 -218
- package/src/quant-arena/data.test.mts +0 -44
- package/src/quant-arena/data.ts +0 -141
- package/src/quant-arena/driver.test.mts +0 -253
- package/src/quant-arena/driver.ts +0 -219
- package/src/quant-arena/fixtures/data/PROVENANCE.md +0 -26
- package/src/quant-arena/fixtures/data/holdout/IDX.csv +0 -523
- package/src/quant-arena/fixtures/data/holdout/S01.csv +0 -523
- package/src/quant-arena/fixtures/data/holdout/S02.csv +0 -523
- package/src/quant-arena/fixtures/data/holdout/S03.csv +0 -523
- package/src/quant-arena/fixtures/data/holdout/S04.csv +0 -523
- package/src/quant-arena/fixtures/data/holdout/S05.csv +0 -523
- package/src/quant-arena/fixtures/data/holdout/S06.csv +0 -523
- package/src/quant-arena/fixtures/data/holdout/S07.csv +0 -523
- package/src/quant-arena/fixtures/data/holdout/S08.csv +0 -523
- package/src/quant-arena/fixtures/data/holdout/S09.csv +0 -523
- package/src/quant-arena/fixtures/data/holdout/S10.csv +0 -523
- package/src/quant-arena/fixtures/data/insample/IDX.csv +0 -2087
- package/src/quant-arena/fixtures/data/insample/S01.csv +0 -2087
- package/src/quant-arena/fixtures/data/insample/S02.csv +0 -2087
- package/src/quant-arena/fixtures/data/insample/S03.csv +0 -2087
- package/src/quant-arena/fixtures/data/insample/S04.csv +0 -2087
- package/src/quant-arena/fixtures/data/insample/S05.csv +0 -2087
- package/src/quant-arena/fixtures/data/insample/S06.csv +0 -2087
- package/src/quant-arena/fixtures/data/insample/S07.csv +0 -2087
- package/src/quant-arena/fixtures/data/insample/S08.csv +0 -2087
- package/src/quant-arena/fixtures/data/insample/S09.csv +0 -2087
- package/src/quant-arena/fixtures/data/insample/S10.csv +0 -2087
- package/src/quant-arena/fixtures/demo-campaign/cost-ledger.jsonl +0 -16
- package/src/quant-arena/fixtures/demo-campaign/notebook.jsonl +0 -5
- package/src/quant-arena/fixtures/demo-campaign/rollout-manifest.json +0 -171
- package/src/quant-arena/fixtures/demo-campaign/strategies/cand-001-default-author/strategy.ts +0 -119
- package/src/quant-arena/fixtures/demo-campaign/strategies/cand-002-default-author/strategy.ts +0 -119
- package/src/quant-arena/fixtures/demo-campaign/strategies/cand-003-quant-researcher/strategy.ts +0 -105
- package/src/quant-arena/fixtures/demo-campaign/strategies/cand-004-quant-researcher/strategy.ts +0 -102
- package/src/quant-arena/fixtures/demo-campaign-v2/cost-ledger.jsonl +0 -4
- package/src/quant-arena/fixtures/demo-campaign-v2/notebook.jsonl +0 -2
- package/src/quant-arena/fixtures/demo-campaign-v2/rollout-manifest.json +0 -84
- package/src/quant-arena/fixtures/demo-campaign-v2/strategies/cand-001-quant-researcher/strategy.ts +0 -117
- package/src/quant-arena/holdout-certify.mts +0 -206
- package/src/quant-arena/holdout-certify.test.mts +0 -82
- package/src/quant-arena/leak-audit.test.mts +0 -79
- package/src/quant-arena/leak-audit.ts +0 -95
- package/src/quant-arena/make-fixtures.mts +0 -161
- package/src/quant-arena/multiplicity.test.mts +0 -68
- package/src/quant-arena/multiplicity.ts +0 -87
- package/src/quant-arena/nautilus-certify.ts +0 -31
- package/src/quant-arena/oms.ts +0 -90
- package/src/quant-arena/profiles/quant-researcher.profile.json +0 -12
- package/src/quant-arena/python/pyproject.toml +0 -8
- package/src/quant-arena/python/uv.lock +0 -1297
- package/src/quant-arena/python/vbt-worker.py +0 -192
- package/src/quant-arena/quant-loop.mts +0 -840
- package/src/quant-arena/quant-loop.test.mts +0 -75
- package/src/quant-arena/strategies/buy-hold-index/strategy.ts +0 -11
- package/src/quant-arena/strategies/equal-weight/strategy.ts +0 -20
- package/src/quant-arena/strategies/sma-crossover/strategy.ts +0 -42
- package/src/quant-arena/types.ts +0 -133
- package/src/quant-arena/vbt-client.ts +0 -321
- package/src/quant-arena/vbt-parity.test.mts +0 -183
- package/src/quant-arena/windows.test.mts +0 -45
- package/src/quant-arena/windows.ts +0 -54
- package/src/rollout-ledger/backfill-swe-arena.mts +0 -610
- package/src/rollout-ledger/backfill-swe-arena.test.mts +0 -347
- package/src/rollout-ledger/settle-capture.mts +0 -448
- package/src/rollout-ledger/settle-capture.test.mts +0 -270
- package/src/swe-arena/activation.mts +0 -225
- package/src/swe-arena/activation.test.mts +0 -300
- package/src/swe-arena/analyze.ts +0 -211
- package/src/swe-arena/arms.ts +0 -862
- package/src/swe-arena/bootstrap-meta.mts +0 -188
- package/src/swe-arena/bootstrap-meta.test.mts +0 -51
- package/src/swe-arena/briefing.mts +0 -217
- package/src/swe-arena/briefing.test.mts +0 -179
- package/src/swe-arena/calibrate.ts +0 -217
- package/src/swe-arena/capabilities.mts +0 -76
- package/src/swe-arena/capabilities.test.mts +0 -57
- package/src/swe-arena/capacity.ts +0 -198
- package/src/swe-arena/cell-evidence.mts +0 -437
- package/src/swe-arena/cell-evidence.test.mts +0 -248
- package/src/swe-arena/diagnosis-ensemble.test.mts +0 -210
- package/src/swe-arena/diagnosis-ensemble.ts +0 -523
- package/src/swe-arena/execution.test.mts +0 -1171
- package/src/swe-arena/factory-command-container.ts +0 -284
- package/src/swe-arena/factory-judge-child.mts +0 -228
- package/src/swe-arena/factory.test.mts +0 -645
- package/src/swe-arena/fixtures/analyze.py +0 -80
- package/src/swe-arena/fixtures/excludes.txt +0 -8
- package/src/swe-arena/fixtures/factory/agent-eval-309/calibration.md +0 -51
- package/src/swe-arena/fixtures/factory/agent-eval-309/manifest.json +0 -29
- package/src/swe-arena/fixtures/factory/agent-eval-309/spec.md +0 -64
- package/src/swe-arena/fixtures/factory/agent-runtime-232/calibration.md +0 -48
- package/src/swe-arena/fixtures/factory/agent-runtime-232/manifest.json +0 -29
- package/src/swe-arena/fixtures/factory/agent-runtime-232/spec.md +0 -48
- package/src/swe-arena/fixtures/factory/loops-28/calibration.md +0 -47
- package/src/swe-arena/fixtures/factory/loops-28/manifest.json +0 -30
- package/src/swe-arena/fixtures/factory/loops-28/spec.md +0 -50
- package/src/swe-arena/fixtures/gen1-salvage/README.md +0 -45
- package/src/swe-arena/fixtures/gen1-salvage/cand0-e6d7361.diff +0 -116
- package/src/swe-arena/fixtures/gen1-salvage/cand1-76a8590.diff +0 -293
- package/src/swe-arena/fixtures/holdout-preregister.log +0 -12
- package/src/swe-arena/fixtures/holdout.json +0 -44
- package/src/swe-arena/fixtures/instances.json +0 -146
- package/src/swe-arena/fixtures/ledger.jsonl +0 -12
- package/src/swe-arena/fixtures/patches/pallets__flask-5014.solo.patch +0 -36
- package/src/swe-arena/fixtures/patches/pydata__xarray-4687.sup.patch +0 -33
- package/src/swe-arena/fixtures/rejudge.jsonl +0 -15
- package/src/swe-arena/fixtures/rematch.jsonl +0 -3
- package/src/swe-arena/fixtures/rematch2.jsonl +0 -3
- package/src/swe-arena/fixtures/rematch3.jsonl +0 -3
- package/src/swe-arena/fixtures/run-report/README.md +0 -43
- package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.json +0 -173
- package/src/swe-arena/fixtures/run-report/factory-agent-eval-309-FSUP0.md +0 -100
- package/src/swe-arena/fixtures/run-report/gen3-rollup.json +0 -551
- package/src/swe-arena/fixtures/run-report/gen3-rollup.md +0 -64
- package/src/swe-arena/fixtures/sup-journal-true.json +0 -19
- package/src/swe-arena/fixtures/verify/astropy__astropy-13033.sh +0 -48
- package/src/swe-arena/fixtures/verify/django__django-11532.sh +0 -50
- package/src/swe-arena/fixtures/verify/matplotlib__matplotlib-20826.sh +0 -76
- package/src/swe-arena/fixtures/verify/pydata__xarray-4687.sh +0 -44
- package/src/swe-arena/fixtures/verify/pytest-dev__pytest-6197.sh +0 -32
- package/src/swe-arena/fixtures/verify/sphinx-doc__sphinx-9658.sh +0 -51
- package/src/swe-arena/fixtures/worker-tokens.json +0 -42
- package/src/swe-arena/fixtures.ts +0 -237
- package/src/swe-arena/gepa-seat.mts +0 -886
- package/src/swe-arena/gepa-seat.test.mts +0 -1136
- package/src/swe-arena/holdout-certify.mts +0 -408
- package/src/swe-arena/holdout-certify.test.mts +0 -160
- package/src/swe-arena/implementation-ref.test.mts +0 -64
- package/src/swe-arena/implementation-ref.ts +0 -62
- package/src/swe-arena/judge-child.mts +0 -37
- package/src/swe-arena/ledger-orphans.mts +0 -77
- package/src/swe-arena/ledger-orphans.test.mts +0 -149
- package/src/swe-arena/manifest.mts +0 -293
- package/src/swe-arena/manifest.test.mts +0 -169
- package/src/swe-arena/materialize.ts +0 -142
- package/src/swe-arena/outer-loop.mts +0 -2854
- package/src/swe-arena/outer-loop.test.mts +0 -714
- package/src/swe-arena/parity.test.mts +0 -87
- package/src/swe-arena/premeasured-from-cells.mts +0 -296
- package/src/swe-arena/premeasured-from-cells.test.mts +0 -201
- package/src/swe-arena/proc.test.mts +0 -172
- package/src/swe-arena/proc.ts +0 -260
- package/src/swe-arena/profiles/deepseek-author.profile.json +0 -12
- package/src/swe-arena/profiles/default-author.profile.json +0 -12
- package/src/swe-arena/proposer-fanout.mts +0 -736
- package/src/swe-arena/proposer-fanout.test.mts +0 -660
- package/src/swe-arena/proposer-provenance.mts +0 -176
- package/src/swe-arena/proposer-provenance.test.mts +0 -106
- package/src/swe-arena/reconcile.ts +0 -0
- package/src/swe-arena/replay.mts +0 -183
- package/src/swe-arena/replay.test.mts +0 -300
- package/src/swe-arena/run-experiment.mts +0 -729
- package/src/swe-arena/run-report.mts +0 -75
- package/src/swe-arena/run-supervisor.mjs +0 -297
- package/src/swe-arena/run-supervisor.test.mts +0 -539
- package/src/swe-arena/score-split.mts +0 -140
- package/src/swe-arena/score-split.test.mts +0 -123
- package/src/swe-arena/scratch-worktree-serialization.test.mts +0 -72
- package/src/swe-arena/scratch-worktree.test.mts +0 -56
- package/src/swe-arena/scratch-worktree.ts +0 -64
- package/src/swe-arena/serialized-judge.ts +0 -414
- package/src/swe-arena/types.ts +0 -218
|
@@ -1,62 +0,0 @@
|
|
|
1
|
-
import { createHash } from 'node:crypto'
|
|
2
|
-
import { lstat, readFile, readdir, readlink, realpath } from 'node:fs/promises'
|
|
3
|
-
import { join, relative } from 'node:path'
|
|
4
|
-
import { canonicalCandidateDigest } from '@tangle-network/agent-interface'
|
|
5
|
-
|
|
6
|
-
const sha256Content = (value: Uint8Array): string =>
|
|
7
|
-
`sha256:${createHash('sha256').update(value).digest('hex')}`
|
|
8
|
-
|
|
9
|
-
export async function fileTreeImplementationRef(root: string): Promise<string> {
|
|
10
|
-
const canonicalRoot = await realpath(root)
|
|
11
|
-
const files: Array<{ path: string; sha256?: string; symlink?: string }> = []
|
|
12
|
-
|
|
13
|
-
const walk = async (directory: string): Promise<void> => {
|
|
14
|
-
const entries = await readdir(directory, { withFileTypes: true })
|
|
15
|
-
entries.sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0))
|
|
16
|
-
for (const entry of entries) {
|
|
17
|
-
const path = join(directory, entry.name)
|
|
18
|
-
const relativePath = relative(canonicalRoot, path)
|
|
19
|
-
if (entry.isDirectory()) {
|
|
20
|
-
await walk(path)
|
|
21
|
-
} else if (entry.isSymbolicLink()) {
|
|
22
|
-
files.push({ path: relativePath, symlink: await readlink(path) })
|
|
23
|
-
} else if (entry.isFile() || (await lstat(path)).isFile()) {
|
|
24
|
-
files.push({ path: relativePath, sha256: sha256Content(await readFile(path)) })
|
|
25
|
-
}
|
|
26
|
-
}
|
|
27
|
-
}
|
|
28
|
-
|
|
29
|
-
await walk(canonicalRoot)
|
|
30
|
-
return canonicalCandidateDigest(files)
|
|
31
|
-
}
|
|
32
|
-
|
|
33
|
-
export async function pythonDistributionImplementationRef(
|
|
34
|
-
distribution: string,
|
|
35
|
-
runPython: (script: string, args: string[]) => Promise<string>,
|
|
36
|
-
): Promise<string> {
|
|
37
|
-
const output = await runPython(
|
|
38
|
-
[
|
|
39
|
-
'import hashlib, importlib, importlib.metadata, json, pathlib, sys',
|
|
40
|
-
'name = sys.argv[1]',
|
|
41
|
-
'module = importlib.import_module(name)',
|
|
42
|
-
'root = pathlib.Path(module.__file__).resolve().parent',
|
|
43
|
-
'files = []',
|
|
44
|
-
'for path in sorted(p for p in root.rglob("*") if p.is_file()):',
|
|
45
|
-
' if "__pycache__" in path.parts or path.suffix == ".pyc":',
|
|
46
|
-
' continue',
|
|
47
|
-
' files.append({"path": str(path.relative_to(root)), "sha256": hashlib.sha256(path.read_bytes()).hexdigest()})',
|
|
48
|
-
'try:',
|
|
49
|
-
' version = importlib.metadata.version(name)',
|
|
50
|
-
'except importlib.metadata.PackageNotFoundError:',
|
|
51
|
-
' version = None',
|
|
52
|
-
'print(json.dumps({"distribution": name, "version": version, "python": sys.version, "files": files}, sort_keys=True))',
|
|
53
|
-
].join('\n'),
|
|
54
|
-
[distribution],
|
|
55
|
-
)
|
|
56
|
-
const line = output.trim().split('\n').at(-1)
|
|
57
|
-
if (line === undefined) {
|
|
58
|
-
throw new Error(`python distribution ${distribution} returned no implementation manifest`)
|
|
59
|
-
}
|
|
60
|
-
const manifest = JSON.parse(line) as unknown
|
|
61
|
-
return canonicalCandidateDigest(manifest)
|
|
62
|
-
}
|
|
@@ -1,37 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Judge child entrypoint — the experiment's `judge.mts`, tracked. One patch,
|
|
3
|
-
* one official swebench verdict, printed as a single `JUDGE_RESULT {...}` line
|
|
4
|
-
* that serialized-judge.ts parses. Runs as a CHILD process (not in-process) so
|
|
5
|
-
* the judge ceiling can SIGKILL a hung swebench run without taking the
|
|
6
|
-
* experiment down, exactly like the bash `timeout N node judge.mts`.
|
|
7
|
-
*
|
|
8
|
-
* usage: node --import tsx src/swe-arena/judge-child.mts <instance_id> <patchPath>
|
|
9
|
-
*/
|
|
10
|
-
|
|
11
|
-
import { readFileSync } from 'node:fs'
|
|
12
|
-
import { createSweBenchAdapter } from '../benchmarks/swe-bench.ts'
|
|
13
|
-
|
|
14
|
-
const [, , iid, patchPath] = process.argv
|
|
15
|
-
if (!iid || !patchPath) {
|
|
16
|
-
console.error('usage: node --import tsx judge-child.mts <instance_id> <patchPath>')
|
|
17
|
-
process.exit(2)
|
|
18
|
-
}
|
|
19
|
-
const patch = readFileSync(patchPath, 'utf8')
|
|
20
|
-
const adapter = createSweBenchAdapter()
|
|
21
|
-
const [task] = await adapter.loadTasks({ ids: [iid], split: 'test' })
|
|
22
|
-
if (!task) {
|
|
23
|
-
console.log(JSON.stringify({ iid, error: 'no-task' }))
|
|
24
|
-
process.exit(2)
|
|
25
|
-
}
|
|
26
|
-
const t0 = Date.now()
|
|
27
|
-
const score = await adapter.judge(task, patch)
|
|
28
|
-
console.log(
|
|
29
|
-
'JUDGE_RESULT ' +
|
|
30
|
-
JSON.stringify({
|
|
31
|
-
iid,
|
|
32
|
-
resolved: score.resolved,
|
|
33
|
-
score: score.score,
|
|
34
|
-
secs: Math.round((Date.now() - t0) / 1000),
|
|
35
|
-
patch_bytes: patch.length,
|
|
36
|
-
}),
|
|
37
|
-
)
|
|
@@ -1,77 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Crash-orphan reconciliation for the improve-run CostLedger.
|
|
3
|
-
*
|
|
4
|
-
* A killed outer-loop process leaves its in-flight paid call as a durable
|
|
5
|
-
* 'pending' record with no receipt. On resume the ledger's fail-closed
|
|
6
|
-
* admission guard ("N unresolved call(s) must be reconciled before new paid
|
|
7
|
-
* work") then refuses EVERY new paid call, so a single crash permanently
|
|
8
|
-
* blocks the run. This module settles those orphans at startup through the
|
|
9
|
-
* ledger's dedicated crash-recovery verb, `CostLedger.reconcile()`.
|
|
10
|
-
*
|
|
11
|
-
* Safety contract — never settle a live run's call:
|
|
12
|
-
* - The caller must hold the outDir single-instance pid lock
|
|
13
|
-
* (`acquireInstanceLock` in outer-loop.mts) before invoking this, so no
|
|
14
|
-
* concurrent outer-loop shares the ledger file. Every pending record
|
|
15
|
-
* restored from disk is therefore from a dead process.
|
|
16
|
-
* - Belt and braces, only `state: 'interrupted'` pendings are touched:
|
|
17
|
-
* restored-from-persistence calls this process's ledger instance never
|
|
18
|
-
* started. An 'active'/'late' pending belongs to an in-flight
|
|
19
|
-
* `runPaidCall` and settles through its own path.
|
|
20
|
-
* - The ledger's persistence append is compare-and-swap on the file
|
|
21
|
-
* revision, so even an unexpected concurrent writer makes reconciliation
|
|
22
|
-
* fail loud (`CostCallConflictError`) instead of corrupting the log.
|
|
23
|
-
*
|
|
24
|
-
* The orphan settles as a FAILED receipt with zero usage and a fully-known
|
|
25
|
-
* $0 actual cost (`actualCostUsd: 0`), reason 'process-crash-orphan'. A
|
|
26
|
-
* `costUnknown`/`usageUnknown` receipt would be marginally more honest about
|
|
27
|
-
* the crashed call's true spend, but an unknown-cost receipt marks the
|
|
28
|
-
* ledger's accounting incomplete forever, which blocks all paid work in
|
|
29
|
-
* capped runs — the same permanent lockout this module exists to fix.
|
|
30
|
-
*/
|
|
31
|
-
|
|
32
|
-
import { existsSync } from 'node:fs'
|
|
33
|
-
import { join } from 'node:path'
|
|
34
|
-
import type { CostReceipt, PendingCostCallView } from '@tangle-network/agent-eval'
|
|
35
|
-
import { createRunCostLedger, fsCampaignStorage } from '@tangle-network/agent-eval/campaign'
|
|
36
|
-
|
|
37
|
-
export const CRASH_ORPHAN_REASON = 'process-crash-orphan'
|
|
38
|
-
|
|
39
|
-
/** The two ledger verbs reconciliation needs, structural for testability. */
|
|
40
|
-
export interface OrphanReconcilableLedger {
|
|
41
|
-
listPending(): PendingCostCallView[]
|
|
42
|
-
reconcile(
|
|
43
|
-
callId: string,
|
|
44
|
-
observed: {
|
|
45
|
-
model: string
|
|
46
|
-
inputTokens: number
|
|
47
|
-
outputTokens: number
|
|
48
|
-
actualCostUsd?: number
|
|
49
|
-
},
|
|
50
|
-
options?: { error?: string },
|
|
51
|
-
): CostReceipt
|
|
52
|
-
}
|
|
53
|
-
|
|
54
|
-
/** Settle every crash-orphaned ('interrupted') pending call as a $0 failure. */
|
|
55
|
-
export function reconcileCrashOrphanCalls(ledger: OrphanReconcilableLedger): CostReceipt[] {
|
|
56
|
-
return ledger
|
|
57
|
-
.listPending()
|
|
58
|
-
.filter((pending) => pending.state === 'interrupted')
|
|
59
|
-
.map((pending) =>
|
|
60
|
-
ledger.reconcile(
|
|
61
|
-
pending.callId,
|
|
62
|
-
{ model: pending.model, inputTokens: 0, outputTokens: 0, actualCostUsd: 0 },
|
|
63
|
-
{ error: CRASH_ORPHAN_REASON },
|
|
64
|
-
),
|
|
65
|
-
)
|
|
66
|
-
}
|
|
67
|
-
|
|
68
|
-
/**
|
|
69
|
-
* Open the durable ledger under `improveRunDir` and settle crash orphans.
|
|
70
|
-
* No-op (empty result) when no ledger file exists yet — a fresh run.
|
|
71
|
-
* PRECONDITION: the caller holds the outDir instance lock (see module doc).
|
|
72
|
-
*/
|
|
73
|
-
export function reconcileCrashOrphansOnDisk(improveRunDir: string): CostReceipt[] {
|
|
74
|
-
if (!existsSync(join(improveRunDir, 'cost-ledger.jsonl'))) return []
|
|
75
|
-
const ledger = createRunCostLedger({ storage: fsCampaignStorage(), runDir: improveRunDir })
|
|
76
|
-
return reconcileCrashOrphanCalls(ledger)
|
|
77
|
-
}
|
|
@@ -1,149 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Crash-orphan ledger reconciliation: a synthetic durable ledger with one
|
|
3
|
-
* orphaned pending call (crashed process — call opened, no receipt) and one
|
|
4
|
-
* resolved call. Reconciliation must settle ONLY the orphan, as a $0
|
|
5
|
-
* failure receipt tagged 'process-crash-orphan', after which the ledger's
|
|
6
|
-
* fail-closed admission guard passes new paid work again — the exact resume
|
|
7
|
-
* path improve() rides.
|
|
8
|
-
*/
|
|
9
|
-
|
|
10
|
-
import { execFileSync } from 'node:child_process'
|
|
11
|
-
import { mkdtempSync, readFileSync } from 'node:fs'
|
|
12
|
-
import { tmpdir } from 'node:os'
|
|
13
|
-
import { join } from 'node:path'
|
|
14
|
-
import { describe, expect, it } from 'vitest'
|
|
15
|
-
import type { CostReceipt, PendingCostCallView } from '@tangle-network/agent-eval'
|
|
16
|
-
import { createRunCostLedger, fsCampaignStorage } from '@tangle-network/agent-eval/campaign'
|
|
17
|
-
import {
|
|
18
|
-
CRASH_ORPHAN_REASON,
|
|
19
|
-
reconcileCrashOrphanCalls,
|
|
20
|
-
reconcileCrashOrphansOnDisk,
|
|
21
|
-
type OrphanReconcilableLedger,
|
|
22
|
-
} from './ledger-orphans.mts'
|
|
23
|
-
|
|
24
|
-
const ATTRIBUTION = { channel: 'agent', phase: 'search.baseline', model: 'm' }
|
|
25
|
-
|
|
26
|
-
/**
|
|
27
|
-
* Write one settled receipt and then terminate a process with a real pending
|
|
28
|
-
* call. The child uses the public ledger API so this fixture cannot shadow its
|
|
29
|
-
* private JSONL event format.
|
|
30
|
-
*/
|
|
31
|
-
function writeCrashedLedger(dir: string): void {
|
|
32
|
-
const child = `
|
|
33
|
-
import { createRunCostLedger, fsCampaignStorage } from '@tangle-network/agent-eval/campaign'
|
|
34
|
-
const runDir = process.env.COST_LEDGER_RUN_DIR
|
|
35
|
-
if (typeof runDir !== 'string' || runDir.length === 0) throw new Error('missing COST_LEDGER_RUN_DIR')
|
|
36
|
-
const attribution = { channel: 'agent', phase: 'search.baseline', model: 'm' }
|
|
37
|
-
const ledger = createRunCostLedger({ storage: fsCampaignStorage(), runDir })
|
|
38
|
-
await ledger.runPaidCall({
|
|
39
|
-
...attribution,
|
|
40
|
-
callId: 'settled-1',
|
|
41
|
-
actor: 'worker:astropy#r0',
|
|
42
|
-
execute: async () => 'settled',
|
|
43
|
-
receipt: () => ({ model: 'm', inputTokens: 10, outputTokens: 5, actualCostUsd: 0.01 }),
|
|
44
|
-
})
|
|
45
|
-
let started
|
|
46
|
-
const pending = new Promise((resolve) => { started = resolve })
|
|
47
|
-
void ledger.runPaidCall({
|
|
48
|
-
...attribution,
|
|
49
|
-
callId: 'orphan-1',
|
|
50
|
-
actor: 'worker:xarray#r0',
|
|
51
|
-
tags: { cellId: 'pydata__xarray-4687:0' },
|
|
52
|
-
execute: async () => { started(); return await new Promise(() => {}) },
|
|
53
|
-
receipt: () => ({ model: 'm', inputTokens: 0, outputTokens: 0, actualCostUsd: 0 }),
|
|
54
|
-
})
|
|
55
|
-
await pending
|
|
56
|
-
`
|
|
57
|
-
execFileSync(process.execPath, ['--input-type=module', '--eval', child], {
|
|
58
|
-
cwd: process.cwd(),
|
|
59
|
-
env: { ...process.env, COST_LEDGER_RUN_DIR: dir },
|
|
60
|
-
})
|
|
61
|
-
}
|
|
62
|
-
|
|
63
|
-
const openLedger = (dir: string) =>
|
|
64
|
-
createRunCostLedger({ storage: fsCampaignStorage(), runDir: dir })
|
|
65
|
-
|
|
66
|
-
describe('reconcileCrashOrphansOnDisk', () => {
|
|
67
|
-
it('settles only the orphan, and the admission guard passes afterwards', async () => {
|
|
68
|
-
const dir = mkdtempSync(join(tmpdir(), 'ledger-orphans-'))
|
|
69
|
-
writeCrashedLedger(dir)
|
|
70
|
-
|
|
71
|
-
// The bug being fixed: with the orphan unresolved, ALL new paid work is refused.
|
|
72
|
-
const blocked = await openLedger(dir).runPaidCall({
|
|
73
|
-
...ATTRIBUTION,
|
|
74
|
-
actor: 'worker:next',
|
|
75
|
-
execute: async () => 'ok',
|
|
76
|
-
receipt: () => ({ model: 'm', inputTokens: 1, outputTokens: 1, actualCostUsd: 0 }),
|
|
77
|
-
})
|
|
78
|
-
expect(blocked.succeeded).toBe(false)
|
|
79
|
-
if (!blocked.succeeded) {
|
|
80
|
-
expect(blocked.error.message).toContain('unresolved call(s) must be reconciled')
|
|
81
|
-
}
|
|
82
|
-
|
|
83
|
-
const receipts = reconcileCrashOrphansOnDisk(dir)
|
|
84
|
-
expect(receipts.map((r) => r.callId)).toEqual(['orphan-1'])
|
|
85
|
-
expect(receipts[0]).toMatchObject({
|
|
86
|
-
status: 'settled',
|
|
87
|
-
error: CRASH_ORPHAN_REASON,
|
|
88
|
-
inputTokens: 0,
|
|
89
|
-
outputTokens: 0,
|
|
90
|
-
costUsd: 0,
|
|
91
|
-
costUnknown: false,
|
|
92
|
-
})
|
|
93
|
-
|
|
94
|
-
// Durable: a fresh ledger instance sees zero unresolved calls, the
|
|
95
|
-
// settled call untouched, and admits new paid work again.
|
|
96
|
-
const reopened = openLedger(dir)
|
|
97
|
-
const summary = reopened.summary()
|
|
98
|
-
expect(summary.unresolvedCalls).toBe(0)
|
|
99
|
-
expect(summary.pendingCalls).toBe(0)
|
|
100
|
-
expect(summary.accountingComplete).toBe(true)
|
|
101
|
-
const settled = reopened.list().find((r) => r.callId === 'settled-1')
|
|
102
|
-
expect(settled?.costUsd).toBe(0.01)
|
|
103
|
-
expect(settled?.error).toBeUndefined()
|
|
104
|
-
const admitted = await reopened.runPaidCall({
|
|
105
|
-
...ATTRIBUTION,
|
|
106
|
-
actor: 'worker:next',
|
|
107
|
-
execute: async () => 'ok',
|
|
108
|
-
receipt: () => ({ model: 'm', inputTokens: 1, outputTokens: 1, actualCostUsd: 0 }),
|
|
109
|
-
})
|
|
110
|
-
expect(admitted.succeeded).toBe(true)
|
|
111
|
-
|
|
112
|
-
// The durable log carries the orphan's failure receipt verbatim.
|
|
113
|
-
const persisted = readFileSync(join(dir, 'cost-ledger.jsonl'), 'utf8')
|
|
114
|
-
expect(persisted).toContain(CRASH_ORPHAN_REASON)
|
|
115
|
-
})
|
|
116
|
-
|
|
117
|
-
it('is idempotent — a second startup pass finds nothing to settle', () => {
|
|
118
|
-
const dir = mkdtempSync(join(tmpdir(), 'ledger-orphans-'))
|
|
119
|
-
writeCrashedLedger(dir)
|
|
120
|
-
expect(reconcileCrashOrphansOnDisk(dir)).toHaveLength(1)
|
|
121
|
-
expect(reconcileCrashOrphansOnDisk(dir)).toHaveLength(0)
|
|
122
|
-
})
|
|
123
|
-
|
|
124
|
-
it('is a no-op on a fresh run dir with no ledger file', () => {
|
|
125
|
-
const dir = mkdtempSync(join(tmpdir(), 'ledger-orphans-'))
|
|
126
|
-
expect(reconcileCrashOrphansOnDisk(dir)).toEqual([])
|
|
127
|
-
})
|
|
128
|
-
})
|
|
129
|
-
|
|
130
|
-
describe('reconcileCrashOrphanCalls', () => {
|
|
131
|
-
it("never settles a live in-process call — only 'interrupted' pendings", () => {
|
|
132
|
-
const base = { ...ATTRIBUTION, status: 'pending' as const, timestamp: 1 }
|
|
133
|
-
const pendings: PendingCostCallView[] = [
|
|
134
|
-
{ ...base, callId: 'live-1', actor: 'a', state: 'active' },
|
|
135
|
-
{ ...base, callId: 'late-1', actor: 'b', state: 'late' },
|
|
136
|
-
{ ...base, callId: 'orphan-1', actor: 'c', state: 'interrupted' },
|
|
137
|
-
]
|
|
138
|
-
const settled: string[] = []
|
|
139
|
-
const ledger: OrphanReconcilableLedger = {
|
|
140
|
-
listPending: () => pendings,
|
|
141
|
-
reconcile: (callId) => {
|
|
142
|
-
settled.push(callId)
|
|
143
|
-
return { callId } as CostReceipt
|
|
144
|
-
},
|
|
145
|
-
}
|
|
146
|
-
reconcileCrashOrphanCalls(ledger)
|
|
147
|
-
expect(settled).toEqual(['orphan-1'])
|
|
148
|
-
})
|
|
149
|
-
})
|
|
@@ -1,293 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Rollout manifest — a pure READER/JOIN over what the round already writes.
|
|
3
|
-
* No new capture pipeline: the lib's improve() run emits loop-provenance.json,
|
|
4
|
-
* per-cell cached results, and the durable cost log; the swe-arena dispatch
|
|
5
|
-
* emits arm result/judge files, candidate patches, and proposer-shot receipts.
|
|
6
|
-
* This module joins them into one versioned per-generation manifest of
|
|
7
|
-
* (state, action, outcome, cost, artifactPaths) triples.
|
|
8
|
-
*
|
|
9
|
-
* tsx src/swe-arena/manifest.mts <outDir> # writes <outDir>/rollout-manifest.json
|
|
10
|
-
*
|
|
11
|
-
* Sources (all optional — a missing file is a labeled gap, never a crash):
|
|
12
|
-
* - <outDir>/improve-run/loop-provenance.json (lib)
|
|
13
|
-
* - <outDir>/improve-run/cost-ledger.jsonl (lib, durable receipts)
|
|
14
|
-
* - <outDir>/improve-run/baseline and gen-N/candidate-K (lib, per-cell caches)
|
|
15
|
-
* - <outDir>/candidates/<commit10>.patch (ours, diff writing)
|
|
16
|
-
* - <outDir>/proposer-shots/genN-candK-shotS.json (ours, shot receipts)
|
|
17
|
-
* - per-cell artifact.runDir result.json + judge.json (ours, arm + judge)
|
|
18
|
-
*/
|
|
19
|
-
|
|
20
|
-
import { readdir, readFile, writeFile } from 'node:fs/promises'
|
|
21
|
-
import { existsSync } from 'node:fs'
|
|
22
|
-
import { join } from 'node:path'
|
|
23
|
-
import { pathToFileURL } from 'node:url'
|
|
24
|
-
import {
|
|
25
|
-
loadCampaignCells,
|
|
26
|
-
loadCandidateCellGroups,
|
|
27
|
-
perInstanceFromCells,
|
|
28
|
-
replicateRunsFromCells,
|
|
29
|
-
resolvedInstanceCount,
|
|
30
|
-
sumWallSFromCells,
|
|
31
|
-
type EvidenceCell,
|
|
32
|
-
type StaircasePerInstance,
|
|
33
|
-
} from './cell-evidence.mts'
|
|
34
|
-
|
|
35
|
-
export const ROLLOUT_SCHEMA = 'swe-arena.rollout.v1'
|
|
36
|
-
|
|
37
|
-
export interface RolloutCost {
|
|
38
|
-
/** Σ campaign-cell CostLedger spend (worker receipts) across the cells. */
|
|
39
|
-
cellCostUsd: number
|
|
40
|
-
cellTokensIn: number
|
|
41
|
-
cellTokensOut: number
|
|
42
|
-
/** Σ runtime spend-tree usd/tokens off the artifacts (telemetry-gap nulls skipped). */
|
|
43
|
-
armSpentUsd: number
|
|
44
|
-
armSpentTokens: number
|
|
45
|
-
wallS: number
|
|
46
|
-
judgeWallS: number
|
|
47
|
-
/** Durable cost-ledger receipts whose tags name this campaign dir. */
|
|
48
|
-
receiptCallIds: string[]
|
|
49
|
-
}
|
|
50
|
-
|
|
51
|
-
export interface RolloutEntry {
|
|
52
|
-
state: {
|
|
53
|
-
generation: number
|
|
54
|
-
/** -1 candidateIndex = the baseline campaign. */
|
|
55
|
-
candidateIndex: number
|
|
56
|
-
campaignDir: string
|
|
57
|
-
commit: string | null
|
|
58
|
-
}
|
|
59
|
-
action: {
|
|
60
|
-
/** Candidate diff written at dispatch time (null for the baseline / a
|
|
61
|
-
* resumed-only run whose diff was not rewritten). */
|
|
62
|
-
diffPath: string | null
|
|
63
|
-
/** Persisted proposer-shot receipts for this (generation, candidate). */
|
|
64
|
-
shotReceiptPaths: string[]
|
|
65
|
-
}
|
|
66
|
-
outcome: {
|
|
67
|
-
perInstance: StaircasePerInstance[]
|
|
68
|
-
resolvedCount: number
|
|
69
|
-
cellsCached: number
|
|
70
|
-
}
|
|
71
|
-
cost: RolloutCost
|
|
72
|
-
artifactPaths: {
|
|
73
|
-
cells: string[]
|
|
74
|
-
armResults: string[]
|
|
75
|
-
judgeVerdicts: string[]
|
|
76
|
-
patches: string[]
|
|
77
|
-
}
|
|
78
|
-
}
|
|
79
|
-
|
|
80
|
-
export interface RolloutGeneration {
|
|
81
|
-
generation: number
|
|
82
|
-
rollouts: RolloutEntry[]
|
|
83
|
-
}
|
|
84
|
-
|
|
85
|
-
export interface RolloutManifest {
|
|
86
|
-
schema: typeof ROLLOUT_SCHEMA
|
|
87
|
-
outDir: string
|
|
88
|
-
at: string
|
|
89
|
-
/** The lib's durable provenance record, verbatim (null = not emitted). */
|
|
90
|
-
provenance: unknown | null
|
|
91
|
-
/** Improvement-set ids inferred from the baseline cells. */
|
|
92
|
-
instances: string[]
|
|
93
|
-
generations: RolloutGeneration[]
|
|
94
|
-
}
|
|
95
|
-
|
|
96
|
-
interface LedgerReceipt {
|
|
97
|
-
callId: string
|
|
98
|
-
tags?: Record<string, string>
|
|
99
|
-
[k: string]: unknown
|
|
100
|
-
}
|
|
101
|
-
|
|
102
|
-
/** Settled receipts out of the lib's append-only cost-ledger.jsonl. */
|
|
103
|
-
export async function loadLedgerReceipts(improveRunDir: string): Promise<LedgerReceipt[]> {
|
|
104
|
-
const raw = await readFile(join(improveRunDir, 'cost-ledger.jsonl'), 'utf8').catch(() => '')
|
|
105
|
-
const receipts: LedgerReceipt[] = []
|
|
106
|
-
for (const line of raw.split('\n')) {
|
|
107
|
-
if (!line.trim()) continue
|
|
108
|
-
let event: Record<string, unknown>
|
|
109
|
-
try {
|
|
110
|
-
event = JSON.parse(line) as Record<string, unknown>
|
|
111
|
-
} catch {
|
|
112
|
-
continue
|
|
113
|
-
}
|
|
114
|
-
const record = event.record as Record<string, unknown> | undefined
|
|
115
|
-
if (record && record.status === 'settled' && typeof record.callId === 'string') {
|
|
116
|
-
receipts.push(record as unknown as LedgerReceipt)
|
|
117
|
-
}
|
|
118
|
-
}
|
|
119
|
-
return receipts
|
|
120
|
-
}
|
|
121
|
-
|
|
122
|
-
function costFromCells(
|
|
123
|
-
cells: EvidenceCell[],
|
|
124
|
-
campaignDir: string,
|
|
125
|
-
receipts: LedgerReceipt[],
|
|
126
|
-
): RolloutCost {
|
|
127
|
-
let cellCostUsd = 0
|
|
128
|
-
let cellTokensIn = 0
|
|
129
|
-
let cellTokensOut = 0
|
|
130
|
-
let armSpentUsd = 0
|
|
131
|
-
let armSpentTokens = 0
|
|
132
|
-
let judgeWallS = 0
|
|
133
|
-
for (const cell of cells) {
|
|
134
|
-
cellCostUsd += cell.costUsd ?? 0
|
|
135
|
-
cellTokensIn += cell.tokenUsage?.input ?? 0
|
|
136
|
-
cellTokensOut += cell.tokenUsage?.output ?? 0
|
|
137
|
-
const a = cell.artifact
|
|
138
|
-
if (a !== null && a.kind === 'swe-arm') {
|
|
139
|
-
armSpentUsd += a.spentUsd ?? 0
|
|
140
|
-
armSpentTokens += a.spentTokens ?? 0
|
|
141
|
-
judgeWallS += a.judgeWallS ?? 0
|
|
142
|
-
}
|
|
143
|
-
}
|
|
144
|
-
return {
|
|
145
|
-
cellCostUsd: Number(cellCostUsd.toFixed(6)),
|
|
146
|
-
cellTokensIn,
|
|
147
|
-
cellTokensOut,
|
|
148
|
-
armSpentUsd: Number(armSpentUsd.toFixed(6)),
|
|
149
|
-
armSpentTokens,
|
|
150
|
-
wallS: sumWallSFromCells(cells),
|
|
151
|
-
judgeWallS,
|
|
152
|
-
receiptCallIds: receipts
|
|
153
|
-
.filter((r) => r.tags?.runDir === campaignDir)
|
|
154
|
-
.map((r) => r.callId)
|
|
155
|
-
.sort(),
|
|
156
|
-
}
|
|
157
|
-
}
|
|
158
|
-
|
|
159
|
-
function artifactPathsFromCells(cells: EvidenceCell[], campaignDir: string): RolloutEntry['artifactPaths'] {
|
|
160
|
-
const armResults: string[] = []
|
|
161
|
-
const judgeVerdicts: string[] = []
|
|
162
|
-
const patches: string[] = []
|
|
163
|
-
for (const cell of cells) {
|
|
164
|
-
const a = cell.artifact
|
|
165
|
-
if (a === null || a.kind !== 'swe-arm') continue
|
|
166
|
-
if (a.runDir) {
|
|
167
|
-
const resultPath = join(a.runDir, 'result.json')
|
|
168
|
-
const judgePath = join(a.runDir, 'judge.json')
|
|
169
|
-
if (existsSync(resultPath)) armResults.push(resultPath)
|
|
170
|
-
if (existsSync(judgePath)) judgeVerdicts.push(judgePath)
|
|
171
|
-
}
|
|
172
|
-
if (a.patchPath && existsSync(a.patchPath)) patches.push(a.patchPath)
|
|
173
|
-
}
|
|
174
|
-
return {
|
|
175
|
-
// The campaign dir holds the per-cell cached-result.json score records.
|
|
176
|
-
cells: cells.length > 0 ? [campaignDir] : [],
|
|
177
|
-
armResults: [...new Set(armResults)].sort(),
|
|
178
|
-
judgeVerdicts: [...new Set(judgeVerdicts)].sort(),
|
|
179
|
-
patches: [...new Set(patches)].sort(),
|
|
180
|
-
}
|
|
181
|
-
}
|
|
182
|
-
|
|
183
|
-
async function shotReceiptPathsFor(
|
|
184
|
-
outDir: string,
|
|
185
|
-
generation: number,
|
|
186
|
-
candidateIndex: number,
|
|
187
|
-
): Promise<string[]> {
|
|
188
|
-
const shotDir = join(outDir, 'proposer-shots')
|
|
189
|
-
const names = await readdir(shotDir).catch(() => [])
|
|
190
|
-
return names
|
|
191
|
-
.filter((n) => n.startsWith(`gen${generation}-cand${candidateIndex}-`) && n.endsWith('.json'))
|
|
192
|
-
.sort()
|
|
193
|
-
.map((n) => join(shotDir, n))
|
|
194
|
-
}
|
|
195
|
-
|
|
196
|
-
/** Build the rollout manifest for one round outDir. Pure join — reads only. */
|
|
197
|
-
export async function buildRolloutManifest(outDir: string): Promise<RolloutManifest> {
|
|
198
|
-
const improveRunDir = join(outDir, 'improve-run')
|
|
199
|
-
const provenanceRaw = await readFile(join(improveRunDir, 'loop-provenance.json'), 'utf8').catch(() => null)
|
|
200
|
-
const provenance = provenanceRaw === null ? null : (JSON.parse(provenanceRaw) as unknown)
|
|
201
|
-
const receipts = await loadLedgerReceipts(improveRunDir)
|
|
202
|
-
|
|
203
|
-
const baselineDir = join(improveRunDir, 'baseline')
|
|
204
|
-
const baselineCells = await loadCampaignCells(baselineDir)
|
|
205
|
-
const instances = [...new Set(baselineCells.map((c) => c.scenarioId))].sort()
|
|
206
|
-
|
|
207
|
-
const generations = new Map<number, RolloutGeneration>()
|
|
208
|
-
const genOf = (g: number): RolloutGeneration => {
|
|
209
|
-
const existing = generations.get(g)
|
|
210
|
-
if (existing) return existing
|
|
211
|
-
const fresh: RolloutGeneration = { generation: g, rollouts: [] }
|
|
212
|
-
generations.set(g, fresh)
|
|
213
|
-
return fresh
|
|
214
|
-
}
|
|
215
|
-
|
|
216
|
-
if (baselineCells.length > 0) {
|
|
217
|
-
genOf(-1).rollouts.push({
|
|
218
|
-
state: { generation: -1, candidateIndex: -1, campaignDir: baselineDir, commit: baselineCells[0]?.artifact?.commit ?? null },
|
|
219
|
-
action: { diffPath: null, shotReceiptPaths: [] },
|
|
220
|
-
outcome: {
|
|
221
|
-
perInstance: perInstanceFromCells(baselineCells),
|
|
222
|
-
resolvedCount: resolvedInstanceCount(
|
|
223
|
-
replicateRunsFromCells(baselineCells),
|
|
224
|
-
instances,
|
|
225
|
-
maxRep(baselineCells) + 1,
|
|
226
|
-
),
|
|
227
|
-
cellsCached: baselineCells.length,
|
|
228
|
-
},
|
|
229
|
-
cost: costFromCells(baselineCells, baselineDir, receipts),
|
|
230
|
-
artifactPaths: artifactPathsFromCells(baselineCells, baselineDir),
|
|
231
|
-
})
|
|
232
|
-
}
|
|
233
|
-
|
|
234
|
-
for (const group of await loadCandidateCellGroups(improveRunDir)) {
|
|
235
|
-
const diffPath = group.commit ? join(outDir, 'candidates', `${group.commit.slice(0, 10)}.patch`) : null
|
|
236
|
-
genOf(group.generation).rollouts.push({
|
|
237
|
-
state: {
|
|
238
|
-
generation: group.generation,
|
|
239
|
-
candidateIndex: group.candidateIndex,
|
|
240
|
-
campaignDir: group.dir,
|
|
241
|
-
commit: group.commit,
|
|
242
|
-
},
|
|
243
|
-
action: {
|
|
244
|
-
diffPath: diffPath !== null && existsSync(diffPath) ? diffPath : null,
|
|
245
|
-
shotReceiptPaths: await shotReceiptPathsFor(outDir, group.generation, group.candidateIndex),
|
|
246
|
-
},
|
|
247
|
-
outcome: {
|
|
248
|
-
perInstance: perInstanceFromCells(group.cells),
|
|
249
|
-
resolvedCount: resolvedInstanceCount(
|
|
250
|
-
replicateRunsFromCells(group.cells),
|
|
251
|
-
instances,
|
|
252
|
-
maxRep(group.cells) + 1,
|
|
253
|
-
),
|
|
254
|
-
cellsCached: group.cells.length,
|
|
255
|
-
},
|
|
256
|
-
cost: costFromCells(group.cells, group.dir, receipts),
|
|
257
|
-
artifactPaths: artifactPathsFromCells(group.cells, group.dir),
|
|
258
|
-
})
|
|
259
|
-
}
|
|
260
|
-
|
|
261
|
-
return {
|
|
262
|
-
schema: ROLLOUT_SCHEMA,
|
|
263
|
-
outDir,
|
|
264
|
-
at: new Date().toISOString(),
|
|
265
|
-
provenance,
|
|
266
|
-
instances,
|
|
267
|
-
generations: [...generations.values()].sort((a, b) => a.generation - b.generation),
|
|
268
|
-
}
|
|
269
|
-
}
|
|
270
|
-
|
|
271
|
-
/** Replicate count inferred from the cells themselves (reps = max rep index
|
|
272
|
-
* + 1). The manifest is a reader with no config in scope; an incomplete
|
|
273
|
-
* candidate under-infers reps and its resolvedCount stays fail-closed. */
|
|
274
|
-
const maxRep = (cells: EvidenceCell[]): number => cells.reduce((m, c) => Math.max(m, c.rep), 0)
|
|
275
|
-
|
|
276
|
-
export async function writeRolloutManifest(outDir: string): Promise<string> {
|
|
277
|
-
const manifest = await buildRolloutManifest(outDir)
|
|
278
|
-
const path = join(outDir, 'rollout-manifest.json')
|
|
279
|
-
await writeFile(path, JSON.stringify(manifest, null, 2))
|
|
280
|
-
return path
|
|
281
|
-
}
|
|
282
|
-
|
|
283
|
-
const isMain = process.argv[1] !== undefined && import.meta.url === pathToFileURL(process.argv[1]).href
|
|
284
|
-
|
|
285
|
-
if (isMain) {
|
|
286
|
-
const outDir = process.argv[2]
|
|
287
|
-
if (!outDir) {
|
|
288
|
-
console.error('usage: tsx src/swe-arena/manifest.mts <outDir>')
|
|
289
|
-
process.exit(2)
|
|
290
|
-
}
|
|
291
|
-
const path = await writeRolloutManifest(outDir)
|
|
292
|
-
console.log(`rollout manifest → ${path}`)
|
|
293
|
-
}
|