@tangle-network/agent-eval 0.124.0 → 0.126.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +60 -35
- package/README.md +270 -189
- package/dist/analyst/index.d.ts +15 -145
- package/dist/analyst/index.js +33 -47
- package/dist/analyst/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +45 -162
- package/dist/benchmarks/index.js +8 -9
- package/dist/campaign/index.d.ts +3655 -5365
- package/dist/campaign/index.js +21 -95
- package/dist/{chunk-R226UZOI.js → chunk-474LBSOX.js} +2 -2
- package/dist/{chunk-W5B3ZGP3.js → chunk-4B7ZZHPX.js} +8 -6
- package/dist/{chunk-W5B3ZGP3.js.map → chunk-4B7ZZHPX.js.map} +1 -1
- package/dist/{chunk-DT7OXY3C.js → chunk-CM4OILD2.js} +535 -846
- package/dist/chunk-CM4OILD2.js.map +1 -0
- package/dist/{chunk-HM6V7F3M.js → chunk-FO7HEH76.js} +3 -3
- package/dist/chunk-IILEIWGW.js +635 -0
- package/dist/chunk-IILEIWGW.js.map +1 -0
- package/dist/{chunk-EQUK3RFS.js → chunk-J5SQWP6Y.js} +8 -5
- package/dist/chunk-J5SQWP6Y.js.map +1 -0
- package/dist/chunk-KO2PZOGP.js +4637 -0
- package/dist/chunk-KO2PZOGP.js.map +1 -0
- package/dist/{chunk-4Y7AAATF.js → chunk-LKKT3IVV.js} +574 -81
- package/dist/chunk-LKKT3IVV.js.map +1 -0
- package/dist/chunk-M7AH34KV.js +155 -0
- package/dist/chunk-M7AH34KV.js.map +1 -0
- package/dist/chunk-NTOV7RU5.js +7152 -0
- package/dist/chunk-NTOV7RU5.js.map +1 -0
- package/dist/{chunk-QFQZ3U3X.js → chunk-OCFJACJU.js} +2 -2
- package/dist/{chunk-GID26AN4.js → chunk-P22LJ3Y2.js} +4 -6
- package/dist/{chunk-GID26AN4.js.map → chunk-P22LJ3Y2.js.map} +1 -1
- package/dist/{chunk-SJT4OBVL.js → chunk-SDPM6554.js} +3 -3
- package/dist/{chunk-D5JZ7UDZ.js → chunk-UCLVDLCH.js} +136 -50
- package/dist/chunk-UCLVDLCH.js.map +1 -0
- package/dist/chunk-UI4YMIN2.js +105 -0
- package/dist/chunk-UI4YMIN2.js.map +1 -0
- package/dist/chunk-VBQ3CRKH.js +291 -0
- package/dist/chunk-VBQ3CRKH.js.map +1 -0
- package/dist/{chunk-JKDNAOF5.js → chunk-W4L6C2XT.js} +2 -2
- package/dist/{chunk-GRCDRKII.js → chunk-WS3NZZQQ.js} +58 -20
- package/dist/chunk-WS3NZZQQ.js.map +1 -0
- package/dist/cli.js +3 -3
- package/dist/contract/index.d.ts +3221 -3094
- package/dist/contract/index.js +173 -42
- package/dist/contract/index.js.map +1 -1
- package/dist/control.js +2 -3
- package/dist/fuzz.d.ts +14 -1
- package/dist/fuzz.js +1 -1
- package/dist/hosted/index.d.ts +8 -1
- package/dist/index.d.ts +208 -690
- package/dist/index.js +185 -500
- package/dist/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/rl.d.ts +5 -100
- package/dist/rl.js +4 -5
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +9 -1
- package/dist/rollout/index.js +6 -6
- package/dist/{run-campaign-I3JXKVAK.js → run-campaign-LVFKZCEU.js} +3 -3
- package/dist/supervisor-run/index.d.ts +156 -4
- package/dist/supervisor-run/index.js +14 -2
- package/dist/traces.js +2 -3
- package/dist/wire/index.d.ts +14 -1
- package/dist/wire/index.js +3 -3
- package/docs/campaign-proposers.md +363 -168
- package/docs/design/loop-taxonomy.md +142 -190
- package/docs/design.md +1 -1
- package/docs/distributed-driver.md +8 -11
- package/docs/feature-guide.md +20 -19
- package/docs/knowledge-readiness.md +2 -5
- package/docs/multi-shot-optimization.md +35 -27
- package/docs/rollout.md +5 -5
- package/package.json +4 -4
- package/dist/chunk-4Y7AAATF.js.map +0 -1
- package/dist/chunk-5PVZVCZB.js +0 -9190
- package/dist/chunk-5PVZVCZB.js.map +0 -1
- package/dist/chunk-A6GT67HT.js +0 -550
- package/dist/chunk-A6GT67HT.js.map +0 -1
- package/dist/chunk-D5JZ7UDZ.js.map +0 -1
- package/dist/chunk-DT7OXY3C.js.map +0 -1
- package/dist/chunk-EQUK3RFS.js.map +0 -1
- package/dist/chunk-GC4ATIKK.js +0 -317
- package/dist/chunk-GC4ATIKK.js.map +0 -1
- package/dist/chunk-GRCDRKII.js.map +0 -1
- package/dist/chunk-LOW3U7JZ.js +0 -328
- package/dist/chunk-LOW3U7JZ.js.map +0 -1
- package/dist/chunk-MGGFVCJ7.js +0 -288
- package/dist/chunk-MGGFVCJ7.js.map +0 -1
- package/dist/chunk-PMITBABE.js +0 -3841
- package/dist/chunk-PMITBABE.js.map +0 -1
- package/dist/chunk-R7ZRE2KV.js +0 -138
- package/dist/chunk-R7ZRE2KV.js.map +0 -1
- /package/dist/{chunk-R226UZOI.js.map → chunk-474LBSOX.js.map} +0 -0
- /package/dist/{chunk-HM6V7F3M.js.map → chunk-FO7HEH76.js.map} +0 -0
- /package/dist/{chunk-QFQZ3U3X.js.map → chunk-OCFJACJU.js.map} +0 -0
- /package/dist/{chunk-SJT4OBVL.js.map → chunk-SDPM6554.js.map} +0 -0
- /package/dist/{chunk-JKDNAOF5.js.map → chunk-W4L6C2XT.js.map} +0 -0
- /package/dist/{run-campaign-I3JXKVAK.js.map → run-campaign-LVFKZCEU.js.map} +0 -0
package/dist/rollout/index.d.ts
CHANGED
|
@@ -876,8 +876,16 @@ interface ClaudeTranscript {
|
|
|
876
876
|
endedAt: string | null;
|
|
877
877
|
model: string | null;
|
|
878
878
|
}
|
|
879
|
+
interface ReadClaudeTranscriptOptions {
|
|
880
|
+
/**
|
|
881
|
+
* Read the sidechain (subagent) thread instead of skipping it. Subagent
|
|
882
|
+
* transcripts under `<session>/subagents/agent-<id>.jsonl` are sidechain
|
|
883
|
+
* lines end to end, so their usage is invisible without this.
|
|
884
|
+
*/
|
|
885
|
+
readonly includeSidechain?: boolean;
|
|
886
|
+
}
|
|
879
887
|
/** Parse one transcript jsonl into canonical messages + usage totals. */
|
|
880
|
-
declare function readClaudeTranscript(path: string): Promise<ClaudeTranscript>;
|
|
888
|
+
declare function readClaudeTranscript(path: string, options?: ReadClaudeTranscriptOptions): Promise<ClaudeTranscript>;
|
|
881
889
|
|
|
882
890
|
/**
|
|
883
891
|
* Read-only backfill reader over the opencode sqlite store
|
package/dist/rollout/index.js
CHANGED
|
@@ -1,18 +1,18 @@
|
|
|
1
1
|
import {
|
|
2
|
-
DEFAULT_CLAUDE_PROJECTS_DIR,
|
|
3
|
-
claudeProjectSlug,
|
|
4
|
-
findClaudeTranscripts,
|
|
5
2
|
mintRolloutRows,
|
|
6
|
-
readClaudeTranscript,
|
|
7
3
|
rolloutReward
|
|
8
|
-
} from "../chunk-
|
|
4
|
+
} from "../chunk-M7AH34KV.js";
|
|
9
5
|
import {
|
|
6
|
+
DEFAULT_CLAUDE_PROJECTS_DIR,
|
|
10
7
|
DEFAULT_OPENCODE_DB,
|
|
8
|
+
claudeProjectSlug,
|
|
9
|
+
findClaudeTranscripts,
|
|
11
10
|
findOpencodeSessionById,
|
|
12
11
|
findOpencodeSessionsByDirectory,
|
|
13
12
|
openOpencodeDb,
|
|
13
|
+
readClaudeTranscript,
|
|
14
14
|
readOpencodeSessionMessages
|
|
15
|
-
} from "../chunk-
|
|
15
|
+
} from "../chunk-VBQ3CRKH.js";
|
|
16
16
|
import {
|
|
17
17
|
FORMAT_FILES,
|
|
18
18
|
RELEASE_FORMATS,
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
import {
|
|
2
2
|
planCampaignRun,
|
|
3
3
|
runCampaign
|
|
4
|
-
} from "./chunk-
|
|
4
|
+
} from "./chunk-UCLVDLCH.js";
|
|
5
5
|
import "./chunk-PJQFMIOX.js";
|
|
6
|
-
import "./chunk-
|
|
6
|
+
import "./chunk-WS3NZZQQ.js";
|
|
7
7
|
import "./chunk-VI2UW6B6.js";
|
|
8
8
|
import "./chunk-ONWEPEDO.js";
|
|
9
9
|
import "./chunk-PZ5AY32C.js";
|
|
@@ -11,4 +11,4 @@ export {
|
|
|
11
11
|
planCampaignRun,
|
|
12
12
|
runCampaign
|
|
13
13
|
};
|
|
14
|
-
//# sourceMappingURL=run-campaign-
|
|
14
|
+
//# sourceMappingURL=run-campaign-LVFKZCEU.js.map
|
|
@@ -225,7 +225,37 @@ interface WorkerLogSource {
|
|
|
225
225
|
readonly inbox: string | null;
|
|
226
226
|
/** Worker patch byte length, or null when absent. */
|
|
227
227
|
readonly patchBytes: number | null;
|
|
228
|
+
/** Where this worker's transcript lives, for the rollout row. Null = no such artifact. */
|
|
229
|
+
readonly transcriptRef?: string | null;
|
|
230
|
+
/** Where this worker's delivered patch lives. Null = the store keeps no patch per worker. */
|
|
231
|
+
readonly patchPath?: string | null;
|
|
232
|
+
/** This worker's own inference tokens, when the store records them per worker. */
|
|
233
|
+
readonly tokensIn?: number | null;
|
|
234
|
+
readonly tokensOut?: number | null;
|
|
235
|
+
readonly cacheRead?: number | null;
|
|
236
|
+
readonly cacheWrite?: number | null;
|
|
228
237
|
}
|
|
238
|
+
/**
|
|
239
|
+
* Facts a SOURCE structurally cannot express, each with the reason.
|
|
240
|
+
*
|
|
241
|
+
* The difference between "the artifact is missing" and "this store never
|
|
242
|
+
* records that fact" is the difference between a run that spent $0 and a
|
|
243
|
+
* harness that does not price inference — and the second harness is where a
|
|
244
|
+
* loops-shaped assumption becomes a fabricated zero. A reader declares its
|
|
245
|
+
* limits once; the analyzer reports `unavailable` for everything downstream.
|
|
246
|
+
*
|
|
247
|
+
* `null` on a field means the source DOES carry that fact.
|
|
248
|
+
*/
|
|
249
|
+
interface SourceLimits {
|
|
250
|
+
/** Reason inference spend has no price in this store (null = the store prices it). */
|
|
251
|
+
readonly spendUsd: string | null;
|
|
252
|
+
/** Reason workers carry no pass/fail verdict (null = verdicts are recorded). */
|
|
253
|
+
readonly workerVerdicts: string | null;
|
|
254
|
+
/** Reason no delivered artifact (patch/diff) is retained per worker (null = retained). */
|
|
255
|
+
readonly deliverables: string | null;
|
|
256
|
+
}
|
|
257
|
+
/** A source that carries every fact the analyzer can use. */
|
|
258
|
+
declare const NO_SOURCE_LIMITS: SourceLimits;
|
|
229
259
|
/**
|
|
230
260
|
* Everything the pure analyzer reads — already-read bytes, never paths. Each
|
|
231
261
|
* field is `null` when its artifact was absent, which is what turns the
|
|
@@ -279,8 +309,24 @@ interface SupervisorRunSources {
|
|
|
279
309
|
sessions: number;
|
|
280
310
|
input: number;
|
|
281
311
|
output: number;
|
|
312
|
+
/** Cached prompt tokens, when the store counts them separately. */
|
|
313
|
+
cacheRead?: number;
|
|
314
|
+
cacheWrite?: number;
|
|
282
315
|
} | null;
|
|
283
316
|
readonly harnessMissingReason: string | null;
|
|
317
|
+
/** What this store structurally cannot record. See `SourceLimits`. */
|
|
318
|
+
readonly limits: SourceLimits;
|
|
319
|
+
/**
|
|
320
|
+
* Where the ROOT invocation's transcript lives. Undefined lets the rollout
|
|
321
|
+
* minter fall back to the loops layout (`<supRunDir>/journal.jsonl`); any
|
|
322
|
+
* other store must say, or the row points at a path that never existed.
|
|
323
|
+
*/
|
|
324
|
+
readonly rootTranscriptRef?: string | null;
|
|
325
|
+
/**
|
|
326
|
+
* The `traces` CLI command that covers this run's harness-session layer.
|
|
327
|
+
* Null falls back to the analyzer's default (an opencode worker fleet).
|
|
328
|
+
*/
|
|
329
|
+
readonly traceCommand: string | null;
|
|
284
330
|
}
|
|
285
331
|
/**
|
|
286
332
|
* A source of supervisor-run bytes. Implementations own their storage layout;
|
|
@@ -352,15 +398,23 @@ interface DecisionMetrics {
|
|
|
352
398
|
interface RoleSpend {
|
|
353
399
|
readonly tokensIn: Measured<number>;
|
|
354
400
|
readonly tokensOut: Measured<number>;
|
|
401
|
+
/**
|
|
402
|
+
* Cached prompt tokens read/written. On a harness that caches aggressively
|
|
403
|
+
* these dwarf `tokensIn`, so a report that omits them understates the context
|
|
404
|
+
* each invocation actually consumed. `unavailable` = the store has no such counter.
|
|
405
|
+
*/
|
|
406
|
+
readonly cacheRead: Measured<number>;
|
|
407
|
+
readonly cacheWrite: Measured<number>;
|
|
355
408
|
readonly usd: Measured<number>;
|
|
356
409
|
readonly source: string;
|
|
357
410
|
}
|
|
358
411
|
interface PerWorkerRow {
|
|
359
412
|
readonly worker: string;
|
|
360
413
|
readonly wallMs: number | null;
|
|
361
|
-
|
|
362
|
-
readonly
|
|
363
|
-
readonly
|
|
414
|
+
/** `null` = this store does not attribute tokens per worker (NOT "zero tokens"). */
|
|
415
|
+
readonly tokensIn: number | null;
|
|
416
|
+
readonly tokensOut: number | null;
|
|
417
|
+
readonly usd: number | null;
|
|
364
418
|
readonly patchBytes: number | null;
|
|
365
419
|
readonly passed: boolean | null;
|
|
366
420
|
}
|
|
@@ -482,6 +536,10 @@ interface SupervisorRunTree {
|
|
|
482
536
|
interface Tokens {
|
|
483
537
|
input: number;
|
|
484
538
|
output: number;
|
|
539
|
+
cacheRead: number;
|
|
540
|
+
cacheWrite: number;
|
|
541
|
+
/** False when the event carried no cache counters at all — not "zero cached". */
|
|
542
|
+
hasCache: boolean;
|
|
485
543
|
}
|
|
486
544
|
interface SpendLike {
|
|
487
545
|
tokens: Tokens;
|
|
@@ -500,6 +558,8 @@ interface CloseRow {
|
|
|
500
558
|
verdict: string | null;
|
|
501
559
|
at: number | null;
|
|
502
560
|
spend: SpendLike;
|
|
561
|
+
/** False when the close event carried no spend object — not "spent nothing". */
|
|
562
|
+
hasSpend: boolean;
|
|
503
563
|
}
|
|
504
564
|
interface WorkerLogFacts {
|
|
505
565
|
started: number | null;
|
|
@@ -527,6 +587,10 @@ interface SupervisorTreeFacts {
|
|
|
527
587
|
readonly brain: {
|
|
528
588
|
tokensIn: number;
|
|
529
589
|
tokensOut: number;
|
|
590
|
+
cacheRead: number;
|
|
591
|
+
cacheWrite: number;
|
|
592
|
+
/** False when no metered event carried cache counters — not "nothing cached". */
|
|
593
|
+
hasCache: boolean;
|
|
530
594
|
usd: number;
|
|
531
595
|
meteredCount: number;
|
|
532
596
|
};
|
|
@@ -549,6 +613,94 @@ declare function parsePatch(text: string): PatchStats;
|
|
|
549
613
|
*/
|
|
550
614
|
declare function rollupSupervisorRuns(reports: readonly SupervisorRunReport[]): SupervisorRunRollup;
|
|
551
615
|
|
|
616
|
+
/**
|
|
617
|
+
* Supervision-tree reader over a THIRD-PARTY harness: Claude Code.
|
|
618
|
+
*
|
|
619
|
+
* `loops-reader.ts` reads a supervisor we wrote, whose journal was designed
|
|
620
|
+
* for this analysis. This reader reads a harness we do not control, whose
|
|
621
|
+
* transcript was designed for replaying a chat — and recovers the same tree
|
|
622
|
+
* from it. If both produce a `SupervisorRunSources`, the tree model is a
|
|
623
|
+
* property of multi-agent runs, not of our journal format.
|
|
624
|
+
*
|
|
625
|
+
* ## Where the tree hides in a Claude Code transcript
|
|
626
|
+
*
|
|
627
|
+
* | Tree fact | Claude Code evidence |
|
|
628
|
+
* |---|---|
|
|
629
|
+
* | spawn | assistant `tool_use` (`Agent` / `Task`), answered by a `tool_result` whose `toolUseResult.agentId` names the child |
|
|
630
|
+
* | settle | a `<task-notification>` block in a later user line: `<task-id>` = agentId, `<status>` |
|
|
631
|
+
* | steer | assistant `tool_use` (`SendMessage`) with `input.to` = agentId — mid-task, to a LIVE child |
|
|
632
|
+
* | delivered | that steer's `tool_result` carrying `success` / `resumedAgentId` |
|
|
633
|
+
* | cancel | assistant `tool_use` (`TaskStop`) targeting an agentId |
|
|
634
|
+
* | brain spend| `message.usage` on the main thread's assistant lines |
|
|
635
|
+
* | worker spend| `message.usage` inside `<session>/subagents/agent-<id>.jsonl` |
|
|
636
|
+
* | depth | a child transcript that itself contains `Agent` tool_use lines |
|
|
637
|
+
*
|
|
638
|
+
* Every one of those is read through `parseClaudeEntries` — the SAME line
|
|
639
|
+
* parser `src/rollout/readers/claude-jsonl.ts` uses for solo rollouts. There
|
|
640
|
+
* is no second transcript parser.
|
|
641
|
+
*
|
|
642
|
+
* ## What Claude Code cannot say
|
|
643
|
+
*
|
|
644
|
+
* It records tokens but never a price, runs no per-worker verify, and keeps no
|
|
645
|
+
* per-worker patch. Those are declared once in `limits`, so the analyzer
|
|
646
|
+
* reports `unavailable — <reason>` instead of the $0 / 0-accepted that summing
|
|
647
|
+
* an empty field would produce. See `SourceLimits`.
|
|
648
|
+
*
|
|
649
|
+
* ## Metric coverage vs the loops journal
|
|
650
|
+
*
|
|
651
|
+
* Measured on a real 52-agent session (fixture:
|
|
652
|
+
* `tests/fixtures/supervisor-run/claude-code-session-*`).
|
|
653
|
+
*
|
|
654
|
+
* | Metric | loops | Claude Code | Why |
|
|
655
|
+
* |---|---|---|---|
|
|
656
|
+
* | workersSpawned / Settled / Cancelled | full | full | spawn tool_use + task-notification + TaskStop |
|
|
657
|
+
* | steers / steersDelivered / steersByWorker | full | full | `SendMessage`; delivery from its tool_result |
|
|
658
|
+
* | waves / waveSizes / maxConcurrency | full | full | derived from spawn/settle instants |
|
|
659
|
+
* | respawns / repeatedLabels | full | full | same derivation |
|
|
660
|
+
* | delegationDepth | full | full | a child transcript's own spawn calls |
|
|
661
|
+
* | timeToFirstSpawn / supervisorWall | full | full | transcript instants |
|
|
662
|
+
* | idleMs / idlePct / workerUtilization | full | PARTIAL | an agent that never notifies is counted live to the end of the transcript |
|
|
663
|
+
* | observeThenRespawn / respawnWithoutEvidence | full | full | ordering of spawn vs settle instants |
|
|
664
|
+
* | workerEvidenceBytes | full | PARTIAL | the child's closing message; 0 for pruned transcripts |
|
|
665
|
+
* | brain tokens in/out + cache | full | full | main-thread `message.usage` |
|
|
666
|
+
* | worker tokens in/out + cache | via harness join | PARTIAL | only for retained subagent transcripts |
|
|
667
|
+
* | perWorker wall | full | full | spawn → settle instants |
|
|
668
|
+
* | accepted / rejected / emptyPass / settledVerdicts | full | NONE | no per-worker verify step exists |
|
|
669
|
+
* | brain/worker/total usd, costPerAcceptedPatch | full | NONE | transcripts carry no price |
|
|
670
|
+
* | patch stats, delivered, verifyPass/Rc | full | NONE | no diff is handed back |
|
|
671
|
+
* | judgeResolved / Score / Passed / Total | full | NONE | no judge in the loop |
|
|
672
|
+
* | driverSteerCalls, brainTruncations | full | NONE | no outer driver log, no per-call finish_reason tap |
|
|
673
|
+
*/
|
|
674
|
+
|
|
675
|
+
/** Tool names that spawn a child agent. `Task` is the older name for `Agent`. */
|
|
676
|
+
declare const DEFAULT_SPAWN_TOOLS: readonly ["Agent", "Task"];
|
|
677
|
+
/** Tool names that deliver a message to an ALREADY-RUNNING child agent. */
|
|
678
|
+
declare const DEFAULT_STEER_TOOLS: readonly ["SendMessage"];
|
|
679
|
+
/** Tool names that stop a running child agent. */
|
|
680
|
+
declare const DEFAULT_CANCEL_TOOLS: readonly ["TaskStop", "KillAgent"];
|
|
681
|
+
interface ClaudeCodeReaderOptions {
|
|
682
|
+
/** The main session transcript: `~/.claude/projects/<slug>/<sessionId>.jsonl`. */
|
|
683
|
+
readonly transcriptPath: string;
|
|
684
|
+
/**
|
|
685
|
+
* Directory of child transcripts. Defaults to `<transcript-dir>/<sessionId>/subagents`.
|
|
686
|
+
* `null` skips the join, and every per-worker token count becomes unavailable.
|
|
687
|
+
*/
|
|
688
|
+
readonly subagentsDir?: string | null;
|
|
689
|
+
readonly runRef?: string;
|
|
690
|
+
readonly instanceId?: string | null;
|
|
691
|
+
readonly arm?: string | null;
|
|
692
|
+
readonly spawnTools?: readonly string[];
|
|
693
|
+
readonly steerTools?: readonly string[];
|
|
694
|
+
readonly cancelTools?: readonly string[];
|
|
695
|
+
}
|
|
696
|
+
/**
|
|
697
|
+
* Read a Claude Code session (plus its subagent transcripts) as supervision-tree
|
|
698
|
+
* source bytes. Never throws on a missing artifact.
|
|
699
|
+
*/
|
|
700
|
+
declare function readClaudeCodeSupervisorRun(opts: ClaudeCodeReaderOptions): Promise<SupervisorRunSources>;
|
|
701
|
+
/** A `SupervisorRunReader` over a Claude Code session — the same contract loops implements. */
|
|
702
|
+
declare function claudeCodeSupervisorRunReader(opts: ClaudeCodeReaderOptions): SupervisorRunReader;
|
|
703
|
+
|
|
552
704
|
/**
|
|
553
705
|
* ONE implementation of `SupervisorRunReader`: the on-disk layout the loops
|
|
554
706
|
* supervisor writes — `<runDir>/ws/.loops/supervisor/<id>/{journal.jsonl,
|
|
@@ -702,4 +854,4 @@ interface SupervisorRolloutOptions {
|
|
|
702
854
|
*/
|
|
703
855
|
declare function supervisorRunRolloutLines(src: SupervisorRunSources, opts?: SupervisorRolloutOptions): SupervisorRunTree;
|
|
704
856
|
|
|
705
|
-
export { type CloseRow, type DecisionMetrics, type EconomicsMetrics, type LoopsReaderOptions, type Measured, type OrchestrationMetrics, type OutcomeMetrics, type PatchStats, type PerWorkerRow, type RoleSpend, type RollupCellRow, SUPERVISOR_RUN_ROLLUP_SCHEMA, SUPERVISOR_RUN_SCHEMA, type SpawnRow, type SteerBreakdown, type SupervisorRolloutOptions, type SupervisorRunReader, type SupervisorRunReport, type SupervisorRunRollup, type SupervisorRunSources, type SupervisorRunTree, type SupervisorTreeFacts, type Unavailable, type WallDistribution, type WorkerLogFacts, type WorkerLogSource, type WriteSupervisorRunOptions, analyzeSupervisorRun, analyzeSupervisorRunSources, findSupervisorRunDirIn, findSupervisorRunDirs, isUnavailable, loopsSupervisorRunReader, parsePatch, parseSupervisorTree, readLoopsSupervisorRun, renderSupervisorRollupMarkdown, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, reportSupervisorRound, rollupSupervisorRuns, showMeasured, supervisorReportStem, supervisorRunRolloutLines, unavailable, writeSupervisorRunReport, writeSupervisorRunReportSafe };
|
|
857
|
+
export { type ClaudeCodeReaderOptions, type CloseRow, DEFAULT_CANCEL_TOOLS, DEFAULT_SPAWN_TOOLS, DEFAULT_STEER_TOOLS, type DecisionMetrics, type EconomicsMetrics, type LoopsReaderOptions, type Measured, NO_SOURCE_LIMITS, type OrchestrationMetrics, type OutcomeMetrics, type PatchStats, type PerWorkerRow, type RoleSpend, type RollupCellRow, SUPERVISOR_RUN_ROLLUP_SCHEMA, SUPERVISOR_RUN_SCHEMA, type SourceLimits, type SpawnRow, type SteerBreakdown, type SupervisorRolloutOptions, type SupervisorRunReader, type SupervisorRunReport, type SupervisorRunRollup, type SupervisorRunSources, type SupervisorRunTree, type SupervisorTreeFacts, type Unavailable, type WallDistribution, type WorkerLogFacts, type WorkerLogSource, type WriteSupervisorRunOptions, analyzeSupervisorRun, analyzeSupervisorRunSources, claudeCodeSupervisorRunReader, findSupervisorRunDirIn, findSupervisorRunDirs, isUnavailable, loopsSupervisorRunReader, parsePatch, parseSupervisorTree, readClaudeCodeSupervisorRun, readLoopsSupervisorRun, renderSupervisorRollupMarkdown, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, reportSupervisorRound, rollupSupervisorRuns, showMeasured, supervisorReportStem, supervisorRunRolloutLines, unavailable, writeSupervisorRunReport, writeSupervisorRunReportSafe };
|
|
@@ -1,14 +1,20 @@
|
|
|
1
1
|
import {
|
|
2
|
+
DEFAULT_CANCEL_TOOLS,
|
|
3
|
+
DEFAULT_SPAWN_TOOLS,
|
|
4
|
+
DEFAULT_STEER_TOOLS,
|
|
5
|
+
NO_SOURCE_LIMITS,
|
|
2
6
|
SUPERVISOR_RUN_ROLLUP_SCHEMA,
|
|
3
7
|
SUPERVISOR_RUN_SCHEMA,
|
|
4
8
|
analyzeSupervisorRun,
|
|
5
9
|
analyzeSupervisorRunSources,
|
|
10
|
+
claudeCodeSupervisorRunReader,
|
|
6
11
|
findSupervisorRunDirIn,
|
|
7
12
|
findSupervisorRunDirs,
|
|
8
13
|
isUnavailable,
|
|
9
14
|
loopsSupervisorRunReader,
|
|
10
15
|
parsePatch,
|
|
11
16
|
parseSupervisorTree,
|
|
17
|
+
readClaudeCodeSupervisorRun,
|
|
12
18
|
readLoopsSupervisorRun,
|
|
13
19
|
renderSupervisorRollupMarkdown,
|
|
14
20
|
renderSupervisorRunHeadline,
|
|
@@ -21,21 +27,27 @@ import {
|
|
|
21
27
|
unavailable,
|
|
22
28
|
writeSupervisorRunReport,
|
|
23
29
|
writeSupervisorRunReportSafe
|
|
24
|
-
} from "../chunk-
|
|
25
|
-
import "../chunk-
|
|
30
|
+
} from "../chunk-LKKT3IVV.js";
|
|
31
|
+
import "../chunk-VBQ3CRKH.js";
|
|
26
32
|
import "../chunk-MAX3TN3C.js";
|
|
27
33
|
import "../chunk-PZ5AY32C.js";
|
|
28
34
|
export {
|
|
35
|
+
DEFAULT_CANCEL_TOOLS,
|
|
36
|
+
DEFAULT_SPAWN_TOOLS,
|
|
37
|
+
DEFAULT_STEER_TOOLS,
|
|
38
|
+
NO_SOURCE_LIMITS,
|
|
29
39
|
SUPERVISOR_RUN_ROLLUP_SCHEMA,
|
|
30
40
|
SUPERVISOR_RUN_SCHEMA,
|
|
31
41
|
analyzeSupervisorRun,
|
|
32
42
|
analyzeSupervisorRunSources,
|
|
43
|
+
claudeCodeSupervisorRunReader,
|
|
33
44
|
findSupervisorRunDirIn,
|
|
34
45
|
findSupervisorRunDirs,
|
|
35
46
|
isUnavailable,
|
|
36
47
|
loopsSupervisorRunReader,
|
|
37
48
|
parsePatch,
|
|
38
49
|
parseSupervisorTree,
|
|
50
|
+
readClaudeCodeSupervisorRun,
|
|
39
51
|
readLoopsSupervisorRun,
|
|
40
52
|
renderSupervisorRollupMarkdown,
|
|
41
53
|
renderSupervisorRunHeadline,
|
package/dist/traces.js
CHANGED
|
@@ -27,7 +27,7 @@ import {
|
|
|
27
27
|
scoreTraceInsightReadiness,
|
|
28
28
|
tokenizeDomainWords,
|
|
29
29
|
traceAnalystOnRunComplete
|
|
30
|
-
} from "./chunk-
|
|
30
|
+
} from "./chunk-OCFJACJU.js";
|
|
31
31
|
import "./chunk-H5UD2323.js";
|
|
32
32
|
import {
|
|
33
33
|
extractUsage,
|
|
@@ -91,7 +91,7 @@ import {
|
|
|
91
91
|
TraceEmitter,
|
|
92
92
|
llmSpanFromProvider
|
|
93
93
|
} from "./chunk-VQMK5FMP.js";
|
|
94
|
-
import "./chunk-
|
|
94
|
+
import "./chunk-IILEIWGW.js";
|
|
95
95
|
import {
|
|
96
96
|
FAILURE_CLASSES,
|
|
97
97
|
TRACE_SCHEMA_VERSION,
|
|
@@ -101,7 +101,6 @@ import {
|
|
|
101
101
|
isSandboxSpan,
|
|
102
102
|
isToolSpan
|
|
103
103
|
} from "./chunk-MA6HLL3S.js";
|
|
104
|
-
import "./chunk-GC4ATIKK.js";
|
|
105
104
|
import "./chunk-VSMTAMNK.js";
|
|
106
105
|
import "./chunk-ONWEPEDO.js";
|
|
107
106
|
import {
|
package/dist/wire/index.d.ts
CHANGED
|
@@ -37,29 +37,40 @@ interface CostReceipt extends CostCallBase, CostUsage {
|
|
|
37
37
|
costUsd: number;
|
|
38
38
|
costUnknown: boolean;
|
|
39
39
|
usageUnknown?: boolean;
|
|
40
|
+
/** Rates used to estimate cost locally. Absent when cost is provider-reported or unknown. */
|
|
40
41
|
pricing?: {
|
|
41
42
|
inputUsdPerThousand: number;
|
|
43
|
+
cachedInputUsdPerThousand?: number;
|
|
44
|
+
cacheWriteUsdPerThousand?: number;
|
|
42
45
|
outputUsdPerThousand: number;
|
|
43
46
|
};
|
|
47
|
+
/** Cost reported by the provider, not a local token-price calculation. */
|
|
44
48
|
actualCostUsd?: number;
|
|
45
49
|
error?: string;
|
|
46
50
|
}
|
|
47
51
|
interface CostReceiptInput extends CostUsage {
|
|
48
52
|
model: string;
|
|
53
|
+
/** Caller-supplied rates for a local estimate when the provider does not report billed cost. */
|
|
54
|
+
customTokenPricing?: CustomTokenPricing;
|
|
49
55
|
actualCostUsd?: number;
|
|
50
56
|
costUnknown?: boolean;
|
|
51
57
|
usageUnknown?: boolean;
|
|
52
58
|
}
|
|
53
59
|
/** Per-million token rates for a model or endpoint not covered by package pricing. */
|
|
54
60
|
interface CustomTokenPricing {
|
|
61
|
+
/** Non-cached input tokens. */
|
|
55
62
|
inputUsdPerMillion: number;
|
|
63
|
+
/** Cache-read tokens. Falls back to the normal input rate when omitted. */
|
|
64
|
+
cachedInputUsdPerMillion?: number;
|
|
65
|
+
/** Cache-creation or cache-write tokens. Falls back to the normal input rate when omitted. */
|
|
66
|
+
cacheWriteUsdPerMillion?: number;
|
|
56
67
|
outputUsdPerMillion: number;
|
|
57
68
|
}
|
|
58
69
|
type MaximumCharge = {
|
|
59
70
|
externallyEnforcedMaximumUsd: number;
|
|
60
71
|
} | ({
|
|
61
72
|
customTokenPricing: CustomTokenPricing;
|
|
62
|
-
} & Pick<CostUsage, 'inputTokens' | 'outputTokens'>) | ({
|
|
73
|
+
} & Pick<CostUsage, 'inputTokens' | 'outputTokens' | 'cachedTokens' | 'cacheWriteTokens'>) | ({
|
|
63
74
|
model: string;
|
|
64
75
|
} & CostUsage);
|
|
65
76
|
interface RunPaidCallInput<T> {
|
|
@@ -127,6 +138,8 @@ interface CostLedgerFilter {
|
|
|
127
138
|
interface CostLedgerWaitOptions {
|
|
128
139
|
/** Maximum time to wait for active provider calls. Default 5 seconds. */
|
|
129
140
|
timeoutMs?: number;
|
|
141
|
+
/** Wait only for calls matching this attribution filter. */
|
|
142
|
+
filter?: CostLedgerFilter;
|
|
130
143
|
}
|
|
131
144
|
/** Append-only storage. `append` must atomically reject stale revisions. */
|
|
132
145
|
interface CostLedgerPersistence {
|
package/dist/wire/index.js
CHANGED
|
@@ -34,9 +34,9 @@ import {
|
|
|
34
34
|
runRpcOnce,
|
|
35
35
|
startServer,
|
|
36
36
|
startServerAsync
|
|
37
|
-
} from "../chunk-
|
|
38
|
-
import "../chunk-
|
|
39
|
-
import "../chunk-
|
|
37
|
+
} from "../chunk-FO7HEH76.js";
|
|
38
|
+
import "../chunk-J5SQWP6Y.js";
|
|
39
|
+
import "../chunk-WS3NZZQQ.js";
|
|
40
40
|
import "../chunk-VI2UW6B6.js";
|
|
41
41
|
import "../chunk-PC4UYEBM.js";
|
|
42
42
|
import "../chunk-ONWEPEDO.js";
|