@tangle-network/agent-eval 0.128.2 → 0.129.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (137) hide show
  1. package/CHANGELOG.md +265 -0
  2. package/README.md +18 -0
  3. package/dist/analyst/index.d.ts +107 -165
  4. package/dist/analyst/index.js +5 -9
  5. package/dist/analyst/index.js.map +1 -1
  6. package/dist/belief-state/index.d.ts +2 -19
  7. package/dist/belief-state/index.js +30 -31
  8. package/dist/belief-state/index.js.map +1 -1
  9. package/dist/benchmarks/index.d.ts +5 -8
  10. package/dist/benchmarks/index.js +12 -11
  11. package/dist/builder-eval/index.js +1 -1
  12. package/dist/campaign/index.d.ts +30 -39
  13. package/dist/campaign/index.js +11 -10
  14. package/dist/{chunk-NKAGIDE2.js → chunk-2QU3YOPR.js} +15 -274
  15. package/dist/chunk-2QU3YOPR.js.map +1 -0
  16. package/dist/{chunk-EJGRPCO3.js → chunk-3OCR4R5I.js} +245 -134
  17. package/dist/chunk-3OCR4R5I.js.map +1 -0
  18. package/dist/{chunk-2JX3CFMB.js → chunk-56TAVBOK.js} +5 -2
  19. package/dist/chunk-56TAVBOK.js.map +1 -0
  20. package/dist/{chunk-DPUHNQLN.js → chunk-7FO3TNPI.js} +2 -2
  21. package/dist/{chunk-DJKY2TSY.js → chunk-BSO5JDQH.js} +27 -120
  22. package/dist/chunk-BSO5JDQH.js.map +1 -0
  23. package/dist/{chunk-EZJEIH2R.js → chunk-C6LXANRU.js} +11 -20
  24. package/dist/chunk-C6LXANRU.js.map +1 -0
  25. package/dist/{chunk-ZUUWPZCV.js → chunk-DODXQREJ.js} +4 -4
  26. package/dist/{chunk-2MKQIFS4.js → chunk-E7QXT7SX.js} +2 -2
  27. package/dist/{chunk-NYLOYM6N.js → chunk-EG66UGL4.js} +37 -28
  28. package/dist/chunk-EG66UGL4.js.map +1 -0
  29. package/dist/{chunk-P5W7RQKK.js → chunk-FXTVJPYD.js} +2 -2
  30. package/dist/{chunk-VZSRQ272.js → chunk-G7MGMCZD.js} +6 -2
  31. package/dist/chunk-G7MGMCZD.js.map +1 -0
  32. package/dist/{chunk-IHQDPH7D.js → chunk-H23X7XKK.js} +85 -75
  33. package/dist/chunk-H23X7XKK.js.map +1 -0
  34. package/dist/{chunk-VBQ3CRKH.js → chunk-HPWUNB47.js} +4 -6
  35. package/dist/chunk-HPWUNB47.js.map +1 -0
  36. package/dist/{chunk-XPRT64IE.js → chunk-IYCLP2N2.js} +3 -3
  37. package/dist/chunk-IYCLP2N2.js.map +1 -0
  38. package/dist/{chunk-XDWDC2MP.js → chunk-JQSF5DQT.js} +11 -5
  39. package/dist/chunk-JQSF5DQT.js.map +1 -0
  40. package/dist/{chunk-NACAGYSY.js → chunk-M4YBQKIJ.js} +11 -11
  41. package/dist/chunk-M4YBQKIJ.js.map +1 -0
  42. package/dist/{chunk-YJBNWCAA.js → chunk-NY44NC4A.js} +3 -3
  43. package/dist/chunk-OIUOT4QD.js +44 -0
  44. package/dist/chunk-OIUOT4QD.js.map +1 -0
  45. package/dist/chunk-OWN5NPMC.js +152 -0
  46. package/dist/chunk-OWN5NPMC.js.map +1 -0
  47. package/dist/chunk-PC5DOSM7.js +579 -0
  48. package/dist/chunk-PC5DOSM7.js.map +1 -0
  49. package/dist/{chunk-UB2LOJ6Q.js → chunk-QB6BDBP2.js} +23 -20
  50. package/dist/chunk-QB6BDBP2.js.map +1 -0
  51. package/dist/chunk-RXHCETDZ.js +536 -0
  52. package/dist/chunk-RXHCETDZ.js.map +1 -0
  53. package/dist/{chunk-PBE2LOSS.js → chunk-SFLLL76A.js} +7 -7
  54. package/dist/chunk-SFLLL76A.js.map +1 -0
  55. package/dist/{chunk-VGRCHJON.js → chunk-T6RLYGAD.js} +3 -8
  56. package/dist/chunk-T6RLYGAD.js.map +1 -0
  57. package/dist/{chunk-VLOATJQ2.js → chunk-TJVT4QFF.js} +21 -18
  58. package/dist/chunk-TJVT4QFF.js.map +1 -0
  59. package/dist/{chunk-S5YLIBFX.js → chunk-TQ7LNKZ3.js} +2 -2
  60. package/dist/{chunk-EOSZT7PL.js → chunk-U4L7JRPZ.js} +2 -297
  61. package/dist/chunk-U4L7JRPZ.js.map +1 -0
  62. package/dist/chunk-U4PHLT2N.js +419 -0
  63. package/dist/chunk-U4PHLT2N.js.map +1 -0
  64. package/dist/{chunk-WS3NZZQQ.js → chunk-VCZ5FQYW.js} +3 -4
  65. package/dist/chunk-VCZ5FQYW.js.map +1 -0
  66. package/dist/{chunk-BYT7ELPS.js → chunk-WVATSFCP.js} +2 -2
  67. package/dist/{chunk-TSN7JT6D.js → chunk-X4YIBDER.js} +21 -5
  68. package/dist/{chunk-TSN7JT6D.js.map → chunk-X4YIBDER.js.map} +1 -1
  69. package/dist/{chunk-TBL77AUT.js → chunk-YQN4ICPP.js} +5 -5
  70. package/dist/{chunk-MHELPNRP.js → chunk-ZHTZ4EYI.js} +1 -1
  71. package/dist/chunk-ZHTZ4EYI.js.map +1 -0
  72. package/dist/cli.js +6 -5
  73. package/dist/cli.js.map +1 -1
  74. package/dist/contract/index.d.ts +47 -87
  75. package/dist/contract/index.js +14 -13
  76. package/dist/contract/index.js.map +1 -1
  77. package/dist/control.js +3 -2
  78. package/dist/fuzz.js +3 -2
  79. package/dist/fuzz.js.map +1 -1
  80. package/dist/index.d.ts +659 -203
  81. package/dist/index.js +145 -117
  82. package/dist/index.js.map +1 -1
  83. package/dist/meta-eval/index.js +2 -2
  84. package/dist/multishot/index.d.ts +3 -4
  85. package/dist/multishot/index.js.map +1 -1
  86. package/dist/openapi.json +1 -1
  87. package/dist/pipelines/index.js +5 -5
  88. package/dist/reporting.d.ts +14 -0
  89. package/dist/reporting.js +7 -6
  90. package/dist/rl.d.ts +652 -82
  91. package/dist/rl.js +415 -171
  92. package/dist/rl.js.map +1 -1
  93. package/dist/rollout/index.d.ts +1071 -32
  94. package/dist/rollout/index.js +68 -10
  95. package/dist/run-campaign-OJJ7CZF4.js +18 -0
  96. package/dist/supervisor-run/index.d.ts +114 -4
  97. package/dist/supervisor-run/index.js +4 -3
  98. package/dist/traces.d.ts +1 -1
  99. package/dist/traces.js +6 -5
  100. package/dist/wire/index.d.ts +10 -11
  101. package/dist/wire/index.js +3 -3
  102. package/docs/feature-guide.md +1 -1
  103. package/docs/rollout.md +116 -2
  104. package/package.json +4 -4
  105. package/dist/chunk-2JX3CFMB.js.map +0 -1
  106. package/dist/chunk-DJKY2TSY.js.map +0 -1
  107. package/dist/chunk-EJGRPCO3.js.map +0 -1
  108. package/dist/chunk-EOSZT7PL.js.map +0 -1
  109. package/dist/chunk-EZJEIH2R.js.map +0 -1
  110. package/dist/chunk-IHQDPH7D.js.map +0 -1
  111. package/dist/chunk-MHELPNRP.js.map +0 -1
  112. package/dist/chunk-NACAGYSY.js.map +0 -1
  113. package/dist/chunk-NKAGIDE2.js.map +0 -1
  114. package/dist/chunk-NYLOYM6N.js.map +0 -1
  115. package/dist/chunk-PBE2LOSS.js.map +0 -1
  116. package/dist/chunk-TT4KNT67.js +0 -124
  117. package/dist/chunk-TT4KNT67.js.map +0 -1
  118. package/dist/chunk-UB2LOJ6Q.js.map +0 -1
  119. package/dist/chunk-UWZZKKU7.js +0 -237
  120. package/dist/chunk-UWZZKKU7.js.map +0 -1
  121. package/dist/chunk-VBQ3CRKH.js.map +0 -1
  122. package/dist/chunk-VGRCHJON.js.map +0 -1
  123. package/dist/chunk-VLOATJQ2.js.map +0 -1
  124. package/dist/chunk-VZSRQ272.js.map +0 -1
  125. package/dist/chunk-WS3NZZQQ.js.map +0 -1
  126. package/dist/chunk-XDWDC2MP.js.map +0 -1
  127. package/dist/chunk-XPRT64IE.js.map +0 -1
  128. package/dist/run-campaign-ISHFZ7FJ.js +0 -17
  129. /package/dist/{chunk-DPUHNQLN.js.map → chunk-7FO3TNPI.js.map} +0 -0
  130. /package/dist/{chunk-ZUUWPZCV.js.map → chunk-DODXQREJ.js.map} +0 -0
  131. /package/dist/{chunk-2MKQIFS4.js.map → chunk-E7QXT7SX.js.map} +0 -0
  132. /package/dist/{chunk-P5W7RQKK.js.map → chunk-FXTVJPYD.js.map} +0 -0
  133. /package/dist/{chunk-YJBNWCAA.js.map → chunk-NY44NC4A.js.map} +0 -0
  134. /package/dist/{chunk-S5YLIBFX.js.map → chunk-TQ7LNKZ3.js.map} +0 -0
  135. /package/dist/{chunk-BYT7ELPS.js.map → chunk-WVATSFCP.js.map} +0 -0
  136. /package/dist/{chunk-TBL77AUT.js.map → chunk-YQN4ICPP.js.map} +0 -0
  137. /package/dist/{run-campaign-ISHFZ7FJ.js.map → run-campaign-OJJ7CZF4.js.map} +0 -0
@@ -1,7 +1,11 @@
1
1
  import {
2
- mintRolloutRows,
3
- rolloutReward
4
- } from "../chunk-IHQDPH7D.js";
2
+ ATIF_SCHEMA_VERSION,
3
+ HARBOR_IMPORT_GAP,
4
+ fromHarborTrajectory,
5
+ relabelImportedSplit,
6
+ toHarborTrajectories,
7
+ toHarborTrajectory
8
+ } from "../chunk-RXHCETDZ.js";
5
9
  import {
6
10
  DEFAULT_CLAUDE_PROJECTS_DIR,
7
11
  DEFAULT_OPENCODE_DB,
@@ -12,57 +16,91 @@ import {
12
16
  openOpencodeDb,
13
17
  readClaudeTranscript,
14
18
  readOpencodeSessionMessages
15
- } from "../chunk-VBQ3CRKH.js";
19
+ } from "../chunk-HPWUNB47.js";
16
20
  import {
17
21
  FORMAT_FILES,
22
+ FORMAT_GATE_DISPOSITION,
18
23
  RELEASE_FORMATS,
19
24
  ROLLOUT_RELEASE_USAGE,
20
25
  SCRUB_RULES,
21
26
  addScrubCounts,
22
27
  appendRolloutLines,
28
+ assertGateReport,
23
29
  buildDatasetCard,
24
30
  buildHfDataset,
25
31
  defaultRolloutScrubber,
26
32
  emptyScrubCounts,
33
+ gatedRolloutIds,
34
+ measureFormatGate,
27
35
  parseRolloutReleaseArgs,
28
36
  planPushCommand,
29
37
  pushDataset,
38
+ readRolloutJournal,
30
39
  readRolloutLedger,
40
+ releaseRowRefs,
31
41
  runRolloutReleaseCli,
32
42
  scrubLines,
33
43
  scrubRolloutLine,
34
44
  scrubText,
45
+ writeRolloutLedger
46
+ } from "../chunk-3OCR4R5I.js";
47
+ import {
48
+ mintRolloutRows
49
+ } from "../chunk-H23X7XKK.js";
50
+ import {
51
+ realnessLabels,
35
52
  toJsonl,
36
53
  toRewardRows,
37
54
  toRftItem,
38
55
  toRftItems,
39
56
  toSftRows,
40
57
  toVerifiersRolloutOutput,
41
- toVerifiersRolloutOutputs,
42
- writeRolloutLedger
43
- } from "../chunk-EJGRPCO3.js";
58
+ toVerifiersRolloutOutputs
59
+ } from "../chunk-OWN5NPMC.js";
44
60
  import {
45
61
  CHAT_ROLES,
62
+ GATE_CHECKS,
63
+ GATE_CHECK_IDS,
64
+ GATE_POLICIES,
46
65
  ROLLOUT_CAPTURES,
47
66
  ROLLOUT_ROLES,
48
67
  ROLLOUT_SCHEMA,
49
68
  ROLLOUT_SPLITS,
50
69
  TRAINABLE_SPLITS,
70
+ assertMinted,
71
+ assertMintedLines,
51
72
  assertRolloutLine,
73
+ gateErrors,
74
+ gateGamedOutcome,
75
+ gatedEvidenceOf,
52
76
  isRolloutLine,
53
77
  isTrainableSplit,
54
78
  validateRolloutLine
55
- } from "../chunk-UWZZKKU7.js";
79
+ } from "../chunk-PC5DOSM7.js";
56
80
  import "../chunk-RZTMDUO7.js";
57
- import "../chunk-2JX3CFMB.js";
81
+ import "../chunk-56TAVBOK.js";
58
82
  import "../chunk-MA6HLL3S.js";
83
+ import {
84
+ isRealnessGated,
85
+ observedScore,
86
+ observedSplitScore,
87
+ scoreOrigin,
88
+ trainingReward,
89
+ trainingScore
90
+ } from "../chunk-OIUOT4QD.js";
59
91
  import "../chunk-ONWEPEDO.js";
60
92
  import "../chunk-PZ5AY32C.js";
61
93
  export {
94
+ ATIF_SCHEMA_VERSION,
62
95
  CHAT_ROLES,
63
96
  DEFAULT_CLAUDE_PROJECTS_DIR,
64
97
  DEFAULT_OPENCODE_DB,
65
98
  FORMAT_FILES,
99
+ FORMAT_GATE_DISPOSITION,
100
+ GATE_CHECKS,
101
+ GATE_CHECK_IDS,
102
+ GATE_POLICIES,
103
+ HARBOR_IMPORT_GAP,
66
104
  RELEASE_FORMATS,
67
105
  ROLLOUT_CAPTURES,
68
106
  ROLLOUT_RELEASE_USAGE,
@@ -73,6 +111,9 @@ export {
73
111
  TRAINABLE_SPLITS,
74
112
  addScrubCounts,
75
113
  appendRolloutLines,
114
+ assertGateReport,
115
+ assertMinted,
116
+ assertMintedLines,
76
117
  assertRolloutLine,
77
118
  buildDatasetCard,
78
119
  buildHfDataset,
@@ -82,21 +123,36 @@ export {
82
123
  findClaudeTranscripts,
83
124
  findOpencodeSessionById,
84
125
  findOpencodeSessionsByDirectory,
126
+ fromHarborTrajectory,
127
+ gateErrors,
128
+ gateGamedOutcome,
129
+ gatedEvidenceOf,
130
+ gatedRolloutIds,
131
+ isRealnessGated,
85
132
  isRolloutLine,
86
133
  isTrainableSplit,
134
+ measureFormatGate,
87
135
  mintRolloutRows,
136
+ observedScore,
137
+ observedSplitScore,
88
138
  openOpencodeDb,
89
139
  parseRolloutReleaseArgs,
90
140
  planPushCommand,
91
141
  pushDataset,
92
142
  readClaudeTranscript,
93
143
  readOpencodeSessionMessages,
144
+ readRolloutJournal,
94
145
  readRolloutLedger,
95
- rolloutReward,
146
+ realnessLabels,
147
+ relabelImportedSplit,
148
+ releaseRowRefs,
96
149
  runRolloutReleaseCli,
150
+ scoreOrigin,
97
151
  scrubLines,
98
152
  scrubRolloutLine,
99
153
  scrubText,
154
+ toHarborTrajectories,
155
+ toHarborTrajectory,
100
156
  toJsonl,
101
157
  toRewardRows,
102
158
  toRftItem,
@@ -104,6 +160,8 @@ export {
104
160
  toSftRows,
105
161
  toVerifiersRolloutOutput,
106
162
  toVerifiersRolloutOutputs,
163
+ trainingReward,
164
+ trainingScore,
107
165
  validateRolloutLine,
108
166
  writeRolloutLedger
109
167
  };
@@ -0,0 +1,18 @@
1
+ import {
2
+ planCampaignRun,
3
+ runCampaign
4
+ } from "./chunk-C6LXANRU.js";
5
+ import "./chunk-E7QXT7SX.js";
6
+ import "./chunk-ZHTZ4EYI.js";
7
+ import "./chunk-VCZ5FQYW.js";
8
+ import "./chunk-VI2UW6B6.js";
9
+ import "./chunk-56TAVBOK.js";
10
+ import "./chunk-MA6HLL3S.js";
11
+ import "./chunk-OIUOT4QD.js";
12
+ import "./chunk-ONWEPEDO.js";
13
+ import "./chunk-PZ5AY32C.js";
14
+ export {
15
+ planCampaignRun,
16
+ runCampaign
17
+ };
18
+ //# sourceMappingURL=run-campaign-OJJ7CZF4.js.map
@@ -23,6 +23,28 @@
23
23
  * `outcome.reward` is THE single scalar (null = no verdict exists — a
24
24
  * labeled gap, never 0). `outcome.realness_gated` is the anti-Goodhart
25
25
  * flag: a gated line must never export as a positive training example.
26
+ *
27
+ * That last sentence is enforced here, by `validateRolloutLine`, not merely
28
+ * documented. Validating `reward` and `realness_gated` independently — each a
29
+ * well-typed field, their COMBINATION unchecked — is what let a line claiming
30
+ * `{reward: 0.95, realness_gated: true}` validate clean and walk into every
31
+ * training export. The relationship between the two IS the invariant, so it is
32
+ * checked where every other structural claim about a line is checked.
33
+ *
34
+ * The invariant is about the OUTCOME, not about one field of it. Zeroing
35
+ * `reward` while `outcome.metrics` still carried the per-layer scores that
36
+ * reward was computed from exported the gamed signal anyway, in the dict the
37
+ * verifiers format reads as its per-rubric scores. So `gateGamedOutcome`
38
+ * transforms the whole outcome once, at `assertMinted` — the funnel every
39
+ * minted line passes — and the reward-bearing components are relocated to
40
+ * `provenance.gated_evidence`, which no exporter projects.
41
+ *
42
+ * WHICH checks each door applies is not decided in this file. `./gate-checks`
43
+ * owns the canonical list and the total per-entry-point policy; the three doors
44
+ * below (`validateRolloutLine`, `assertRewardGate`, `assertMinted`) each call
45
+ * `gateErrors` with their declared policy, so a check added to that list applies
46
+ * here without anyone editing this file, and a check deliberately skipped has to
47
+ * name itself there.
26
48
  */
27
49
  declare const ROLLOUT_SCHEMA = "tangle.rollout.v1";
28
50
  /** `agent` = a solo evaluation run (no multi-agent topology). */
@@ -50,6 +72,14 @@ interface ChatMessage {
50
72
  /** Required on role:"tool" — the ChatToolCall this result answers. */
51
73
  tool_call_id?: string;
52
74
  name?: string;
75
+ /**
76
+ * Harbor ATIF `is_copied_context` (RFC 0001 rule 7): this turn was COPIED IN
77
+ * from another trajectory's context, not produced by the agent on this line.
78
+ * The RFC makes excluding it from SFT a MUST, and `toSftRows` does — training
79
+ * on it teaches the model to author text it never authored, and credits this
80
+ * run for another one's work. Absent = false (authored here).
81
+ */
82
+ is_copied_context?: boolean;
53
83
  }
54
84
  interface ToolDef {
55
85
  type: 'function';
@@ -73,6 +103,21 @@ interface RolloutStep {
73
103
  output?: string;
74
104
  status?: 'ok' | 'error';
75
105
  durationMs?: number;
106
+ /**
107
+ * LLM inferences this span represents. 0 = deterministic dispatch with no
108
+ * model call — distinct from absent, which means the producer did not track it.
109
+ */
110
+ llm_call_count?: number;
111
+ /** Exact prompt tokenization. Removes the ambiguity of re-tokenizing text at train time. */
112
+ prompt_token_ids?: number[];
113
+ /** Exact completion tokenization; aligns index-wise with `logprobs`. */
114
+ completion_token_ids?: number[];
115
+ /**
116
+ * Per-completion-token log probabilities under the sampling policy. Required
117
+ * for off-policy correction (importance weighting) when the rollout was
118
+ * generated by a policy other than the one being trained.
119
+ */
120
+ logprobs?: number[];
76
121
  }
77
122
  interface RolloutTask {
78
123
  /** Benchmark/suite id (e.g. "swe-bench-verified") or the experiment id. */
@@ -117,11 +162,39 @@ interface RolloutOutcome {
117
162
  is_truncated: boolean;
118
163
  error: string | null;
119
164
  /**
120
- * Anti-Goodhart flag from `RunRecord.outcome.realness.gated`: the run
121
- * faked its success signal. Reward is forced to 0 at mint time and the
122
- * line never qualifies for SFT.
165
+ * Anti-Goodhart flag from `RunRecord.outcome.realness.gated`: the run faked
166
+ * its success signal. `true` requires `reward` to be 0 or null the
167
+ * validator rejects the line otherwise — and the line never qualifies for
168
+ * SFT. Required on the wire: a line that does not state the flag does not
169
+ * validate, so no producer can dodge the gate by omitting it.
170
+ *
171
+ * `true` ALSO requires `metrics` to be empty and `verdict` to be null: the
172
+ * numbers the reward was computed from are relocated to
173
+ * `provenance.gated_evidence` by `gateGamedOutcome`. See that function for
174
+ * why zeroing the scalar alone was not enough.
123
175
  */
124
176
  realness_gated: boolean;
177
+ /**
178
+ * Whether an authenticity SCREEN ever RAN on this reward — a different claim
179
+ * from `realness_gated`, which is the screen's VERDICT.
180
+ *
181
+ * `realness_gated: false` reads as "we looked and nothing fired". A producer
182
+ * with no screen at all was emitting exactly that, so a never-screened reward
183
+ * was indistinguishable on the wire from a screened-clean one, and the whole
184
+ * anti-Goodhart apparatus silently treated the first as the second. The two
185
+ * claims are now separable:
186
+ *
187
+ * - `true` — a screen ran; `realness_gated` is its verdict.
188
+ * - `false` — the producer declares it HAS no screen (`unscreenedRewardFields`).
189
+ * `assertMinted` REFUSES such a line when its reward is above
190
+ * zero: an unscreened positive reward is precisely the signal
191
+ * the gate exists to qualify, and nothing has qualified it.
192
+ * - absent — not stated. Pre-unification ledgers land here, as does a
193
+ * `RunRecord` carrying no `outcome.realness` at all. Absent is
194
+ * read as "unknown", never as `false` (which would refuse most
195
+ * of the existing corpus) and never as `true`.
196
+ */
197
+ realness_screened?: boolean;
125
198
  }
126
199
  interface RolloutCostBlock {
127
200
  usd: number | null;
@@ -131,6 +204,11 @@ interface RolloutCostBlock {
131
204
  cache_read: number | null;
132
205
  cache_write: number | null;
133
206
  wall_s: number | null;
207
+ /**
208
+ * Total LLM inferences across the invocation (ATIF `llm_call_count`,
209
+ * aggregated). Optional and additive: absent = not tracked, never 0.
210
+ */
211
+ llm_call_count?: number | null;
134
212
  }
135
213
  interface RolloutArtifacts {
136
214
  patch_path: string | null;
@@ -138,11 +216,43 @@ interface RolloutArtifacts {
138
216
  /** Source-of-truth transcript pointer (session id / jsonl path) for audit. */
139
217
  transcript_ref: string | null;
140
218
  }
219
+ /**
220
+ * The reward-bearing half of a GATED line's outcome, moved off `outcome` and
221
+ * parked here verbatim. Diagnostics, never training input — see
222
+ * `gateGamedOutcome`.
223
+ */
224
+ interface GatedEvidence {
225
+ /** `outcome.metrics` exactly as the producer measured it. */
226
+ metrics?: Record<string, unknown>;
227
+ /** `outcome.verdict` verbatim — the judge record that claimed the success. */
228
+ verdict?: unknown;
229
+ /**
230
+ * The per-step fields `tangle.rollout.v1` does not declare, parked here when
231
+ * the gate projected `steps[]` down to the schema's own key set.
232
+ *
233
+ * A per-step reward is training signal exactly like the scalar, and `steps`
234
+ * rides through `toRewardRows` verbatim — so a gated line was shipping its
235
+ * step-level credit assignment at full value beside a `reward` of 0.
236
+ */
237
+ steps?: unknown;
238
+ }
141
239
  interface RolloutProvenance {
142
240
  captured_at: string;
143
241
  capture: RolloutCapture;
144
- /** Present on gap lines: why `messages` could not be recovered. */
242
+ /**
243
+ * Why this line is incomplete. Required when `messages` is empty (the
244
+ * transcript could not be recovered); also set by interchange importers to
245
+ * name a MISSING LABEL — an imported trajectory carries no verdict, so
246
+ * `outcome.reward` is null and this says why.
247
+ */
145
248
  gap?: string;
249
+ /**
250
+ * Present only on a realness-gated line: the outcome fields the gate
251
+ * relocated, kept so an auditor can still see WHY the run was gated and what
252
+ * it claimed. Deliberately OUTSIDE `outcome`, because every training exporter
253
+ * reads `outcome` and none reads `provenance`.
254
+ */
255
+ gated_evidence?: GatedEvidence;
146
256
  }
147
257
  interface RolloutLine {
148
258
  schema: typeof ROLLOUT_SCHEMA;
@@ -27,9 +27,10 @@ import {
27
27
  unavailable,
28
28
  writeSupervisorRunReport,
29
29
  writeSupervisorRunReportSafe
30
- } from "../chunk-TSN7JT6D.js";
31
- import "../chunk-VBQ3CRKH.js";
32
- import "../chunk-UWZZKKU7.js";
30
+ } from "../chunk-X4YIBDER.js";
31
+ import "../chunk-HPWUNB47.js";
32
+ import "../chunk-PC5DOSM7.js";
33
+ import "../chunk-OIUOT4QD.js";
33
34
  import "../chunk-PZ5AY32C.js";
34
35
  export {
35
36
  DEFAULT_CANCEL_TOOLS,
package/dist/traces.d.ts CHANGED
@@ -1972,7 +1972,7 @@ interface OtlpFlatLine {
1972
1972
  }>;
1973
1973
  }
1974
1974
  interface FlattenOtlpOptions {
1975
- /** `'openinference'` (default) mirrors legacy per-span attributes into the
1975
+ /** `'openinference'` (default) maps source per-span attributes into the
1976
1976
  * canonical OpenInference vocabulary the analyst readers consume. `'none'`
1977
1977
  * passes attributes through untouched. */
1978
1978
  attributeVocabulary?: 'openinference' | 'none';
package/dist/traces.js CHANGED
@@ -1,6 +1,4 @@
1
1
  import {
2
- FileSystemTraceStore,
3
- InMemoryTraceStore,
4
2
  OTEL_AGENT_EVAL_SCOPE,
5
3
  ReplayCache,
6
4
  ReplayCacheMissError,
@@ -27,7 +25,7 @@ import {
27
25
  scoreTraceInsightReadiness,
28
26
  tokenizeDomainWords,
29
27
  traceAnalystOnRunComplete
30
- } from "./chunk-EOSZT7PL.js";
28
+ } from "./chunk-U4L7JRPZ.js";
31
29
  import "./chunk-7ZZMD7UK.js";
32
30
  import {
33
31
  extractUsage,
@@ -78,10 +76,12 @@ import {
78
76
  traceSpanKindToOpenInferenceKind
79
77
  } from "./chunk-P6FYH6K4.js";
80
78
  import {
79
+ FileSystemTraceStore,
80
+ InMemoryTraceStore,
81
81
  RunIntegrityError,
82
82
  assertRunCaptured,
83
83
  throwIfRunIncomplete
84
- } from "./chunk-TT4KNT67.js";
84
+ } from "./chunk-U4PHLT2N.js";
85
85
  import {
86
86
  FileSystemRawProviderSink,
87
87
  InMemoryRawProviderSink,
@@ -93,7 +93,7 @@ import {
93
93
  TraceEmitter,
94
94
  llmSpanFromProvider
95
95
  } from "./chunk-VQMK5FMP.js";
96
- import "./chunk-2JX3CFMB.js";
96
+ import "./chunk-56TAVBOK.js";
97
97
  import {
98
98
  FAILURE_CLASSES,
99
99
  TRACE_SCHEMA_VERSION,
@@ -103,6 +103,7 @@ import {
103
103
  isSandboxSpan,
104
104
  isToolSpan
105
105
  } from "./chunk-MA6HLL3S.js";
106
+ import "./chunk-OIUOT4QD.js";
106
107
  import "./chunk-ONWEPEDO.js";
107
108
  import {
108
109
  INPUT_VALUE,
@@ -652,9 +652,8 @@ type ProviderRedactor = (event: RawProviderEvent) => RawProviderEvent;
652
652
  * { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
653
653
  * )
654
654
  *
655
- * This is THE llm-calling seam for agent-eval primitives that need structured
656
- * output (semantic concept judge, reviewer directives, critic scores). Primitives
657
- * that need free-form text use `callLlm` and parse output themselves.
655
+ * `createChatClient` wraps this implementation for provider-neutral package
656
+ * entry points. Direct callers can use `callLlm` or `callLlmJson`.
658
657
  */
659
658
 
660
659
  type LlmThinkingMode = 'enabled' | 'disabled';
@@ -688,8 +687,8 @@ interface LlmClientOptions {
688
687
  * total attempts × `timeoutMs`.
689
688
  */
690
689
  deadlineMs?: number;
691
- /** Total provider attempts. Legacy option name; default 3 (1 initial + 2 retries). */
692
- maxRetries?: number;
690
+ /** Total provider attempts. Default 3. */
691
+ maximumAttempts?: number;
693
692
  /** Token rates used when the provider omits cost or package pricing does not cover the model. */
694
693
  customTokenPricing?: CustomTokenPricing;
695
694
  /**
@@ -909,8 +908,8 @@ declare const FeedbackLabelSchema: z.ZodObject<{
909
908
  id: z.ZodOptional<z.ZodString>;
910
909
  source: z.ZodEnum<{
911
910
  judge: "judge";
912
- user: "user";
913
911
  system: "system";
912
+ user: "user";
914
913
  policy: "policy";
915
914
  environment: "environment";
916
915
  metric: "metric";
@@ -957,9 +956,9 @@ declare const FeedbackAttemptSchema: z.ZodObject<{
957
956
  proposedAction: z.ZodOptional<z.ZodObject<{
958
957
  type: z.ZodString;
959
958
  risk: z.ZodOptional<z.ZodEnum<{
960
- medium: "medium";
961
959
  low: "low";
962
960
  high: "high";
961
+ medium: "medium";
963
962
  }>>;
964
963
  costUsd: z.ZodOptional<z.ZodNumber>;
965
964
  externalSideEffect: z.ZodOptional<z.ZodBoolean>;
@@ -970,8 +969,8 @@ declare const FeedbackAttemptSchema: z.ZodObject<{
970
969
  id: z.ZodOptional<z.ZodString>;
971
970
  source: z.ZodEnum<{
972
971
  judge: "judge";
973
- user: "user";
974
972
  system: "system";
973
+ user: "user";
975
974
  policy: "policy";
976
975
  environment: "environment";
977
976
  metric: "metric";
@@ -1029,9 +1028,9 @@ declare const FeedbackTrajectorySchema: z.ZodObject<{
1029
1028
  proposedAction: z.ZodOptional<z.ZodObject<{
1030
1029
  type: z.ZodString;
1031
1030
  risk: z.ZodOptional<z.ZodEnum<{
1032
- medium: "medium";
1033
1031
  low: "low";
1034
1032
  high: "high";
1033
+ medium: "medium";
1035
1034
  }>>;
1036
1035
  costUsd: z.ZodOptional<z.ZodNumber>;
1037
1036
  externalSideEffect: z.ZodOptional<z.ZodBoolean>;
@@ -1042,8 +1041,8 @@ declare const FeedbackTrajectorySchema: z.ZodObject<{
1042
1041
  id: z.ZodOptional<z.ZodString>;
1043
1042
  source: z.ZodEnum<{
1044
1043
  judge: "judge";
1045
- user: "user";
1046
1044
  system: "system";
1045
+ user: "user";
1047
1046
  policy: "policy";
1048
1047
  environment: "environment";
1049
1048
  metric: "metric";
@@ -1078,8 +1077,8 @@ declare const FeedbackTrajectorySchema: z.ZodObject<{
1078
1077
  id: z.ZodOptional<z.ZodString>;
1079
1078
  source: z.ZodEnum<{
1080
1079
  judge: "judge";
1081
- user: "user";
1082
1080
  system: "system";
1081
+ user: "user";
1083
1082
  policy: "policy";
1084
1083
  environment: "environment";
1085
1084
  metric: "metric";
@@ -34,9 +34,9 @@ import {
34
34
  runRpcOnce,
35
35
  startServer,
36
36
  startServerAsync
37
- } from "../chunk-YJBNWCAA.js";
38
- import "../chunk-PBE2LOSS.js";
39
- import "../chunk-WS3NZZQQ.js";
37
+ } from "../chunk-NY44NC4A.js";
38
+ import "../chunk-SFLLL76A.js";
39
+ import "../chunk-VCZ5FQYW.js";
40
40
  import "../chunk-VI2UW6B6.js";
41
41
  import "../chunk-PC4UYEBM.js";
42
42
  import "../chunk-ONWEPEDO.js";
@@ -151,7 +151,7 @@ Store as `FeedbackTrajectory`, then derive:
151
151
 
152
152
  | Area | Key exports | Best for | Notes |
153
153
  | --- | --- | --- | --- |
154
- | Judging | `createCustomJudge`, semantic judges, anti-slop, wire rubrics | Content, voice, semantic quality | Pair with objective checks when possible. |
154
+ | Judging | `llmJudge`, semantic judges, anti-slop, wire rubrics | Content, voice, semantic quality | Pair with objective checks when possible. |
155
155
  | Verification | `MultiLayerVerifier`, `JudgeRunner`, sandbox harness | Code and multi-step gates | Do not let semantic judges override failed builds. |
156
156
  | Control | `runAgentControlLoop`, `objectiveEval`, `subjectiveEval` | Long-running agent tasks | Supports budgets, cost, stop policies, trace spans. |
157
157
  | Propose/review | `runProposeReview`, `runProposeReviewAsControlLoop` | Iterative artifact repair | Good for code, docs, plans, briefs. |