@tangle-network/agent-eval 0.128.2 → 0.129.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +265 -0
- package/README.md +18 -0
- package/dist/analyst/index.d.ts +107 -165
- package/dist/analyst/index.js +5 -9
- package/dist/analyst/index.js.map +1 -1
- package/dist/belief-state/index.d.ts +2 -19
- package/dist/belief-state/index.js +30 -31
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +5 -8
- package/dist/benchmarks/index.js +12 -11
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +30 -39
- package/dist/campaign/index.js +11 -10
- package/dist/{chunk-NKAGIDE2.js → chunk-2QU3YOPR.js} +15 -274
- package/dist/chunk-2QU3YOPR.js.map +1 -0
- package/dist/{chunk-EJGRPCO3.js → chunk-3OCR4R5I.js} +245 -134
- package/dist/chunk-3OCR4R5I.js.map +1 -0
- package/dist/{chunk-2JX3CFMB.js → chunk-56TAVBOK.js} +5 -2
- package/dist/chunk-56TAVBOK.js.map +1 -0
- package/dist/{chunk-DPUHNQLN.js → chunk-7FO3TNPI.js} +2 -2
- package/dist/{chunk-DJKY2TSY.js → chunk-BSO5JDQH.js} +27 -120
- package/dist/chunk-BSO5JDQH.js.map +1 -0
- package/dist/{chunk-EZJEIH2R.js → chunk-C6LXANRU.js} +11 -20
- package/dist/chunk-C6LXANRU.js.map +1 -0
- package/dist/{chunk-ZUUWPZCV.js → chunk-DODXQREJ.js} +4 -4
- package/dist/{chunk-2MKQIFS4.js → chunk-E7QXT7SX.js} +2 -2
- package/dist/{chunk-NYLOYM6N.js → chunk-EG66UGL4.js} +37 -28
- package/dist/chunk-EG66UGL4.js.map +1 -0
- package/dist/{chunk-P5W7RQKK.js → chunk-FXTVJPYD.js} +2 -2
- package/dist/{chunk-VZSRQ272.js → chunk-G7MGMCZD.js} +6 -2
- package/dist/chunk-G7MGMCZD.js.map +1 -0
- package/dist/{chunk-IHQDPH7D.js → chunk-H23X7XKK.js} +85 -75
- package/dist/chunk-H23X7XKK.js.map +1 -0
- package/dist/{chunk-VBQ3CRKH.js → chunk-HPWUNB47.js} +4 -6
- package/dist/chunk-HPWUNB47.js.map +1 -0
- package/dist/{chunk-XPRT64IE.js → chunk-IYCLP2N2.js} +3 -3
- package/dist/chunk-IYCLP2N2.js.map +1 -0
- package/dist/{chunk-XDWDC2MP.js → chunk-JQSF5DQT.js} +11 -5
- package/dist/chunk-JQSF5DQT.js.map +1 -0
- package/dist/{chunk-NACAGYSY.js → chunk-M4YBQKIJ.js} +11 -11
- package/dist/chunk-M4YBQKIJ.js.map +1 -0
- package/dist/{chunk-YJBNWCAA.js → chunk-NY44NC4A.js} +3 -3
- package/dist/chunk-OIUOT4QD.js +44 -0
- package/dist/chunk-OIUOT4QD.js.map +1 -0
- package/dist/chunk-OWN5NPMC.js +152 -0
- package/dist/chunk-OWN5NPMC.js.map +1 -0
- package/dist/chunk-PC5DOSM7.js +579 -0
- package/dist/chunk-PC5DOSM7.js.map +1 -0
- package/dist/{chunk-UB2LOJ6Q.js → chunk-QB6BDBP2.js} +23 -20
- package/dist/chunk-QB6BDBP2.js.map +1 -0
- package/dist/chunk-RXHCETDZ.js +536 -0
- package/dist/chunk-RXHCETDZ.js.map +1 -0
- package/dist/{chunk-PBE2LOSS.js → chunk-SFLLL76A.js} +7 -7
- package/dist/chunk-SFLLL76A.js.map +1 -0
- package/dist/{chunk-VGRCHJON.js → chunk-T6RLYGAD.js} +3 -8
- package/dist/chunk-T6RLYGAD.js.map +1 -0
- package/dist/{chunk-VLOATJQ2.js → chunk-TJVT4QFF.js} +21 -18
- package/dist/chunk-TJVT4QFF.js.map +1 -0
- package/dist/{chunk-S5YLIBFX.js → chunk-TQ7LNKZ3.js} +2 -2
- package/dist/{chunk-EOSZT7PL.js → chunk-U4L7JRPZ.js} +2 -297
- package/dist/chunk-U4L7JRPZ.js.map +1 -0
- package/dist/chunk-U4PHLT2N.js +419 -0
- package/dist/chunk-U4PHLT2N.js.map +1 -0
- package/dist/{chunk-WS3NZZQQ.js → chunk-VCZ5FQYW.js} +3 -4
- package/dist/chunk-VCZ5FQYW.js.map +1 -0
- package/dist/{chunk-BYT7ELPS.js → chunk-WVATSFCP.js} +2 -2
- package/dist/{chunk-TSN7JT6D.js → chunk-X4YIBDER.js} +21 -5
- package/dist/{chunk-TSN7JT6D.js.map → chunk-X4YIBDER.js.map} +1 -1
- package/dist/{chunk-TBL77AUT.js → chunk-YQN4ICPP.js} +5 -5
- package/dist/{chunk-MHELPNRP.js → chunk-ZHTZ4EYI.js} +1 -1
- package/dist/chunk-ZHTZ4EYI.js.map +1 -0
- package/dist/cli.js +6 -5
- package/dist/cli.js.map +1 -1
- package/dist/contract/index.d.ts +47 -87
- package/dist/contract/index.js +14 -13
- package/dist/contract/index.js.map +1 -1
- package/dist/control.js +3 -2
- package/dist/fuzz.js +3 -2
- package/dist/fuzz.js.map +1 -1
- package/dist/index.d.ts +659 -203
- package/dist/index.js +145 -117
- package/dist/index.js.map +1 -1
- package/dist/meta-eval/index.js +2 -2
- package/dist/multishot/index.d.ts +3 -4
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.js +5 -5
- package/dist/reporting.d.ts +14 -0
- package/dist/reporting.js +7 -6
- package/dist/rl.d.ts +652 -82
- package/dist/rl.js +415 -171
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1071 -32
- package/dist/rollout/index.js +68 -10
- package/dist/run-campaign-OJJ7CZF4.js +18 -0
- package/dist/supervisor-run/index.d.ts +114 -4
- package/dist/supervisor-run/index.js +4 -3
- package/dist/traces.d.ts +1 -1
- package/dist/traces.js +6 -5
- package/dist/wire/index.d.ts +10 -11
- package/dist/wire/index.js +3 -3
- package/docs/feature-guide.md +1 -1
- package/docs/rollout.md +116 -2
- package/package.json +4 -4
- package/dist/chunk-2JX3CFMB.js.map +0 -1
- package/dist/chunk-DJKY2TSY.js.map +0 -1
- package/dist/chunk-EJGRPCO3.js.map +0 -1
- package/dist/chunk-EOSZT7PL.js.map +0 -1
- package/dist/chunk-EZJEIH2R.js.map +0 -1
- package/dist/chunk-IHQDPH7D.js.map +0 -1
- package/dist/chunk-MHELPNRP.js.map +0 -1
- package/dist/chunk-NACAGYSY.js.map +0 -1
- package/dist/chunk-NKAGIDE2.js.map +0 -1
- package/dist/chunk-NYLOYM6N.js.map +0 -1
- package/dist/chunk-PBE2LOSS.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-UB2LOJ6Q.js.map +0 -1
- package/dist/chunk-UWZZKKU7.js +0 -237
- package/dist/chunk-UWZZKKU7.js.map +0 -1
- package/dist/chunk-VBQ3CRKH.js.map +0 -1
- package/dist/chunk-VGRCHJON.js.map +0 -1
- package/dist/chunk-VLOATJQ2.js.map +0 -1
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WS3NZZQQ.js.map +0 -1
- package/dist/chunk-XDWDC2MP.js.map +0 -1
- package/dist/chunk-XPRT64IE.js.map +0 -1
- package/dist/run-campaign-ISHFZ7FJ.js +0 -17
- /package/dist/{chunk-DPUHNQLN.js.map → chunk-7FO3TNPI.js.map} +0 -0
- /package/dist/{chunk-ZUUWPZCV.js.map → chunk-DODXQREJ.js.map} +0 -0
- /package/dist/{chunk-2MKQIFS4.js.map → chunk-E7QXT7SX.js.map} +0 -0
- /package/dist/{chunk-P5W7RQKK.js.map → chunk-FXTVJPYD.js.map} +0 -0
- /package/dist/{chunk-YJBNWCAA.js.map → chunk-NY44NC4A.js.map} +0 -0
- /package/dist/{chunk-S5YLIBFX.js.map → chunk-TQ7LNKZ3.js.map} +0 -0
- /package/dist/{chunk-BYT7ELPS.js.map → chunk-WVATSFCP.js.map} +0 -0
- /package/dist/{chunk-TBL77AUT.js.map → chunk-YQN4ICPP.js.map} +0 -0
- /package/dist/{run-campaign-ISHFZ7FJ.js.map → run-campaign-OJJ7CZF4.js.map} +0 -0
package/dist/rollout/index.js
CHANGED
|
@@ -1,7 +1,11 @@
|
|
|
1
1
|
import {
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
2
|
+
ATIF_SCHEMA_VERSION,
|
|
3
|
+
HARBOR_IMPORT_GAP,
|
|
4
|
+
fromHarborTrajectory,
|
|
5
|
+
relabelImportedSplit,
|
|
6
|
+
toHarborTrajectories,
|
|
7
|
+
toHarborTrajectory
|
|
8
|
+
} from "../chunk-RXHCETDZ.js";
|
|
5
9
|
import {
|
|
6
10
|
DEFAULT_CLAUDE_PROJECTS_DIR,
|
|
7
11
|
DEFAULT_OPENCODE_DB,
|
|
@@ -12,57 +16,91 @@ import {
|
|
|
12
16
|
openOpencodeDb,
|
|
13
17
|
readClaudeTranscript,
|
|
14
18
|
readOpencodeSessionMessages
|
|
15
|
-
} from "../chunk-
|
|
19
|
+
} from "../chunk-HPWUNB47.js";
|
|
16
20
|
import {
|
|
17
21
|
FORMAT_FILES,
|
|
22
|
+
FORMAT_GATE_DISPOSITION,
|
|
18
23
|
RELEASE_FORMATS,
|
|
19
24
|
ROLLOUT_RELEASE_USAGE,
|
|
20
25
|
SCRUB_RULES,
|
|
21
26
|
addScrubCounts,
|
|
22
27
|
appendRolloutLines,
|
|
28
|
+
assertGateReport,
|
|
23
29
|
buildDatasetCard,
|
|
24
30
|
buildHfDataset,
|
|
25
31
|
defaultRolloutScrubber,
|
|
26
32
|
emptyScrubCounts,
|
|
33
|
+
gatedRolloutIds,
|
|
34
|
+
measureFormatGate,
|
|
27
35
|
parseRolloutReleaseArgs,
|
|
28
36
|
planPushCommand,
|
|
29
37
|
pushDataset,
|
|
38
|
+
readRolloutJournal,
|
|
30
39
|
readRolloutLedger,
|
|
40
|
+
releaseRowRefs,
|
|
31
41
|
runRolloutReleaseCli,
|
|
32
42
|
scrubLines,
|
|
33
43
|
scrubRolloutLine,
|
|
34
44
|
scrubText,
|
|
45
|
+
writeRolloutLedger
|
|
46
|
+
} from "../chunk-3OCR4R5I.js";
|
|
47
|
+
import {
|
|
48
|
+
mintRolloutRows
|
|
49
|
+
} from "../chunk-H23X7XKK.js";
|
|
50
|
+
import {
|
|
51
|
+
realnessLabels,
|
|
35
52
|
toJsonl,
|
|
36
53
|
toRewardRows,
|
|
37
54
|
toRftItem,
|
|
38
55
|
toRftItems,
|
|
39
56
|
toSftRows,
|
|
40
57
|
toVerifiersRolloutOutput,
|
|
41
|
-
toVerifiersRolloutOutputs
|
|
42
|
-
|
|
43
|
-
} from "../chunk-EJGRPCO3.js";
|
|
58
|
+
toVerifiersRolloutOutputs
|
|
59
|
+
} from "../chunk-OWN5NPMC.js";
|
|
44
60
|
import {
|
|
45
61
|
CHAT_ROLES,
|
|
62
|
+
GATE_CHECKS,
|
|
63
|
+
GATE_CHECK_IDS,
|
|
64
|
+
GATE_POLICIES,
|
|
46
65
|
ROLLOUT_CAPTURES,
|
|
47
66
|
ROLLOUT_ROLES,
|
|
48
67
|
ROLLOUT_SCHEMA,
|
|
49
68
|
ROLLOUT_SPLITS,
|
|
50
69
|
TRAINABLE_SPLITS,
|
|
70
|
+
assertMinted,
|
|
71
|
+
assertMintedLines,
|
|
51
72
|
assertRolloutLine,
|
|
73
|
+
gateErrors,
|
|
74
|
+
gateGamedOutcome,
|
|
75
|
+
gatedEvidenceOf,
|
|
52
76
|
isRolloutLine,
|
|
53
77
|
isTrainableSplit,
|
|
54
78
|
validateRolloutLine
|
|
55
|
-
} from "../chunk-
|
|
79
|
+
} from "../chunk-PC5DOSM7.js";
|
|
56
80
|
import "../chunk-RZTMDUO7.js";
|
|
57
|
-
import "../chunk-
|
|
81
|
+
import "../chunk-56TAVBOK.js";
|
|
58
82
|
import "../chunk-MA6HLL3S.js";
|
|
83
|
+
import {
|
|
84
|
+
isRealnessGated,
|
|
85
|
+
observedScore,
|
|
86
|
+
observedSplitScore,
|
|
87
|
+
scoreOrigin,
|
|
88
|
+
trainingReward,
|
|
89
|
+
trainingScore
|
|
90
|
+
} from "../chunk-OIUOT4QD.js";
|
|
59
91
|
import "../chunk-ONWEPEDO.js";
|
|
60
92
|
import "../chunk-PZ5AY32C.js";
|
|
61
93
|
export {
|
|
94
|
+
ATIF_SCHEMA_VERSION,
|
|
62
95
|
CHAT_ROLES,
|
|
63
96
|
DEFAULT_CLAUDE_PROJECTS_DIR,
|
|
64
97
|
DEFAULT_OPENCODE_DB,
|
|
65
98
|
FORMAT_FILES,
|
|
99
|
+
FORMAT_GATE_DISPOSITION,
|
|
100
|
+
GATE_CHECKS,
|
|
101
|
+
GATE_CHECK_IDS,
|
|
102
|
+
GATE_POLICIES,
|
|
103
|
+
HARBOR_IMPORT_GAP,
|
|
66
104
|
RELEASE_FORMATS,
|
|
67
105
|
ROLLOUT_CAPTURES,
|
|
68
106
|
ROLLOUT_RELEASE_USAGE,
|
|
@@ -73,6 +111,9 @@ export {
|
|
|
73
111
|
TRAINABLE_SPLITS,
|
|
74
112
|
addScrubCounts,
|
|
75
113
|
appendRolloutLines,
|
|
114
|
+
assertGateReport,
|
|
115
|
+
assertMinted,
|
|
116
|
+
assertMintedLines,
|
|
76
117
|
assertRolloutLine,
|
|
77
118
|
buildDatasetCard,
|
|
78
119
|
buildHfDataset,
|
|
@@ -82,21 +123,36 @@ export {
|
|
|
82
123
|
findClaudeTranscripts,
|
|
83
124
|
findOpencodeSessionById,
|
|
84
125
|
findOpencodeSessionsByDirectory,
|
|
126
|
+
fromHarborTrajectory,
|
|
127
|
+
gateErrors,
|
|
128
|
+
gateGamedOutcome,
|
|
129
|
+
gatedEvidenceOf,
|
|
130
|
+
gatedRolloutIds,
|
|
131
|
+
isRealnessGated,
|
|
85
132
|
isRolloutLine,
|
|
86
133
|
isTrainableSplit,
|
|
134
|
+
measureFormatGate,
|
|
87
135
|
mintRolloutRows,
|
|
136
|
+
observedScore,
|
|
137
|
+
observedSplitScore,
|
|
88
138
|
openOpencodeDb,
|
|
89
139
|
parseRolloutReleaseArgs,
|
|
90
140
|
planPushCommand,
|
|
91
141
|
pushDataset,
|
|
92
142
|
readClaudeTranscript,
|
|
93
143
|
readOpencodeSessionMessages,
|
|
144
|
+
readRolloutJournal,
|
|
94
145
|
readRolloutLedger,
|
|
95
|
-
|
|
146
|
+
realnessLabels,
|
|
147
|
+
relabelImportedSplit,
|
|
148
|
+
releaseRowRefs,
|
|
96
149
|
runRolloutReleaseCli,
|
|
150
|
+
scoreOrigin,
|
|
97
151
|
scrubLines,
|
|
98
152
|
scrubRolloutLine,
|
|
99
153
|
scrubText,
|
|
154
|
+
toHarborTrajectories,
|
|
155
|
+
toHarborTrajectory,
|
|
100
156
|
toJsonl,
|
|
101
157
|
toRewardRows,
|
|
102
158
|
toRftItem,
|
|
@@ -104,6 +160,8 @@ export {
|
|
|
104
160
|
toSftRows,
|
|
105
161
|
toVerifiersRolloutOutput,
|
|
106
162
|
toVerifiersRolloutOutputs,
|
|
163
|
+
trainingReward,
|
|
164
|
+
trainingScore,
|
|
107
165
|
validateRolloutLine,
|
|
108
166
|
writeRolloutLedger
|
|
109
167
|
};
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
import {
|
|
2
|
+
planCampaignRun,
|
|
3
|
+
runCampaign
|
|
4
|
+
} from "./chunk-C6LXANRU.js";
|
|
5
|
+
import "./chunk-E7QXT7SX.js";
|
|
6
|
+
import "./chunk-ZHTZ4EYI.js";
|
|
7
|
+
import "./chunk-VCZ5FQYW.js";
|
|
8
|
+
import "./chunk-VI2UW6B6.js";
|
|
9
|
+
import "./chunk-56TAVBOK.js";
|
|
10
|
+
import "./chunk-MA6HLL3S.js";
|
|
11
|
+
import "./chunk-OIUOT4QD.js";
|
|
12
|
+
import "./chunk-ONWEPEDO.js";
|
|
13
|
+
import "./chunk-PZ5AY32C.js";
|
|
14
|
+
export {
|
|
15
|
+
planCampaignRun,
|
|
16
|
+
runCampaign
|
|
17
|
+
};
|
|
18
|
+
//# sourceMappingURL=run-campaign-OJJ7CZF4.js.map
|
|
@@ -23,6 +23,28 @@
|
|
|
23
23
|
* `outcome.reward` is THE single scalar (null = no verdict exists — a
|
|
24
24
|
* labeled gap, never 0). `outcome.realness_gated` is the anti-Goodhart
|
|
25
25
|
* flag: a gated line must never export as a positive training example.
|
|
26
|
+
*
|
|
27
|
+
* That last sentence is enforced here, by `validateRolloutLine`, not merely
|
|
28
|
+
* documented. Validating `reward` and `realness_gated` independently — each a
|
|
29
|
+
* well-typed field, their COMBINATION unchecked — is what let a line claiming
|
|
30
|
+
* `{reward: 0.95, realness_gated: true}` validate clean and walk into every
|
|
31
|
+
* training export. The relationship between the two IS the invariant, so it is
|
|
32
|
+
* checked where every other structural claim about a line is checked.
|
|
33
|
+
*
|
|
34
|
+
* The invariant is about the OUTCOME, not about one field of it. Zeroing
|
|
35
|
+
* `reward` while `outcome.metrics` still carried the per-layer scores that
|
|
36
|
+
* reward was computed from exported the gamed signal anyway, in the dict the
|
|
37
|
+
* verifiers format reads as its per-rubric scores. So `gateGamedOutcome`
|
|
38
|
+
* transforms the whole outcome once, at `assertMinted` — the funnel every
|
|
39
|
+
* minted line passes — and the reward-bearing components are relocated to
|
|
40
|
+
* `provenance.gated_evidence`, which no exporter projects.
|
|
41
|
+
*
|
|
42
|
+
* WHICH checks each door applies is not decided in this file. `./gate-checks`
|
|
43
|
+
* owns the canonical list and the total per-entry-point policy; the three doors
|
|
44
|
+
* below (`validateRolloutLine`, `assertRewardGate`, `assertMinted`) each call
|
|
45
|
+
* `gateErrors` with their declared policy, so a check added to that list applies
|
|
46
|
+
* here without anyone editing this file, and a check deliberately skipped has to
|
|
47
|
+
* name itself there.
|
|
26
48
|
*/
|
|
27
49
|
declare const ROLLOUT_SCHEMA = "tangle.rollout.v1";
|
|
28
50
|
/** `agent` = a solo evaluation run (no multi-agent topology). */
|
|
@@ -50,6 +72,14 @@ interface ChatMessage {
|
|
|
50
72
|
/** Required on role:"tool" — the ChatToolCall this result answers. */
|
|
51
73
|
tool_call_id?: string;
|
|
52
74
|
name?: string;
|
|
75
|
+
/**
|
|
76
|
+
* Harbor ATIF `is_copied_context` (RFC 0001 rule 7): this turn was COPIED IN
|
|
77
|
+
* from another trajectory's context, not produced by the agent on this line.
|
|
78
|
+
* The RFC makes excluding it from SFT a MUST, and `toSftRows` does — training
|
|
79
|
+
* on it teaches the model to author text it never authored, and credits this
|
|
80
|
+
* run for another one's work. Absent = false (authored here).
|
|
81
|
+
*/
|
|
82
|
+
is_copied_context?: boolean;
|
|
53
83
|
}
|
|
54
84
|
interface ToolDef {
|
|
55
85
|
type: 'function';
|
|
@@ -73,6 +103,21 @@ interface RolloutStep {
|
|
|
73
103
|
output?: string;
|
|
74
104
|
status?: 'ok' | 'error';
|
|
75
105
|
durationMs?: number;
|
|
106
|
+
/**
|
|
107
|
+
* LLM inferences this span represents. 0 = deterministic dispatch with no
|
|
108
|
+
* model call — distinct from absent, which means the producer did not track it.
|
|
109
|
+
*/
|
|
110
|
+
llm_call_count?: number;
|
|
111
|
+
/** Exact prompt tokenization. Removes the ambiguity of re-tokenizing text at train time. */
|
|
112
|
+
prompt_token_ids?: number[];
|
|
113
|
+
/** Exact completion tokenization; aligns index-wise with `logprobs`. */
|
|
114
|
+
completion_token_ids?: number[];
|
|
115
|
+
/**
|
|
116
|
+
* Per-completion-token log probabilities under the sampling policy. Required
|
|
117
|
+
* for off-policy correction (importance weighting) when the rollout was
|
|
118
|
+
* generated by a policy other than the one being trained.
|
|
119
|
+
*/
|
|
120
|
+
logprobs?: number[];
|
|
76
121
|
}
|
|
77
122
|
interface RolloutTask {
|
|
78
123
|
/** Benchmark/suite id (e.g. "swe-bench-verified") or the experiment id. */
|
|
@@ -117,11 +162,39 @@ interface RolloutOutcome {
|
|
|
117
162
|
is_truncated: boolean;
|
|
118
163
|
error: string | null;
|
|
119
164
|
/**
|
|
120
|
-
* Anti-Goodhart flag from `RunRecord.outcome.realness.gated`: the run
|
|
121
|
-
*
|
|
122
|
-
* line never qualifies for
|
|
165
|
+
* Anti-Goodhart flag from `RunRecord.outcome.realness.gated`: the run faked
|
|
166
|
+
* its success signal. `true` requires `reward` to be 0 or null — the
|
|
167
|
+
* validator rejects the line otherwise — and the line never qualifies for
|
|
168
|
+
* SFT. Required on the wire: a line that does not state the flag does not
|
|
169
|
+
* validate, so no producer can dodge the gate by omitting it.
|
|
170
|
+
*
|
|
171
|
+
* `true` ALSO requires `metrics` to be empty and `verdict` to be null: the
|
|
172
|
+
* numbers the reward was computed from are relocated to
|
|
173
|
+
* `provenance.gated_evidence` by `gateGamedOutcome`. See that function for
|
|
174
|
+
* why zeroing the scalar alone was not enough.
|
|
123
175
|
*/
|
|
124
176
|
realness_gated: boolean;
|
|
177
|
+
/**
|
|
178
|
+
* Whether an authenticity SCREEN ever RAN on this reward — a different claim
|
|
179
|
+
* from `realness_gated`, which is the screen's VERDICT.
|
|
180
|
+
*
|
|
181
|
+
* `realness_gated: false` reads as "we looked and nothing fired". A producer
|
|
182
|
+
* with no screen at all was emitting exactly that, so a never-screened reward
|
|
183
|
+
* was indistinguishable on the wire from a screened-clean one, and the whole
|
|
184
|
+
* anti-Goodhart apparatus silently treated the first as the second. The two
|
|
185
|
+
* claims are now separable:
|
|
186
|
+
*
|
|
187
|
+
* - `true` — a screen ran; `realness_gated` is its verdict.
|
|
188
|
+
* - `false` — the producer declares it HAS no screen (`unscreenedRewardFields`).
|
|
189
|
+
* `assertMinted` REFUSES such a line when its reward is above
|
|
190
|
+
* zero: an unscreened positive reward is precisely the signal
|
|
191
|
+
* the gate exists to qualify, and nothing has qualified it.
|
|
192
|
+
* - absent — not stated. Pre-unification ledgers land here, as does a
|
|
193
|
+
* `RunRecord` carrying no `outcome.realness` at all. Absent is
|
|
194
|
+
* read as "unknown", never as `false` (which would refuse most
|
|
195
|
+
* of the existing corpus) and never as `true`.
|
|
196
|
+
*/
|
|
197
|
+
realness_screened?: boolean;
|
|
125
198
|
}
|
|
126
199
|
interface RolloutCostBlock {
|
|
127
200
|
usd: number | null;
|
|
@@ -131,6 +204,11 @@ interface RolloutCostBlock {
|
|
|
131
204
|
cache_read: number | null;
|
|
132
205
|
cache_write: number | null;
|
|
133
206
|
wall_s: number | null;
|
|
207
|
+
/**
|
|
208
|
+
* Total LLM inferences across the invocation (ATIF `llm_call_count`,
|
|
209
|
+
* aggregated). Optional and additive: absent = not tracked, never 0.
|
|
210
|
+
*/
|
|
211
|
+
llm_call_count?: number | null;
|
|
134
212
|
}
|
|
135
213
|
interface RolloutArtifacts {
|
|
136
214
|
patch_path: string | null;
|
|
@@ -138,11 +216,43 @@ interface RolloutArtifacts {
|
|
|
138
216
|
/** Source-of-truth transcript pointer (session id / jsonl path) for audit. */
|
|
139
217
|
transcript_ref: string | null;
|
|
140
218
|
}
|
|
219
|
+
/**
|
|
220
|
+
* The reward-bearing half of a GATED line's outcome, moved off `outcome` and
|
|
221
|
+
* parked here verbatim. Diagnostics, never training input — see
|
|
222
|
+
* `gateGamedOutcome`.
|
|
223
|
+
*/
|
|
224
|
+
interface GatedEvidence {
|
|
225
|
+
/** `outcome.metrics` exactly as the producer measured it. */
|
|
226
|
+
metrics?: Record<string, unknown>;
|
|
227
|
+
/** `outcome.verdict` verbatim — the judge record that claimed the success. */
|
|
228
|
+
verdict?: unknown;
|
|
229
|
+
/**
|
|
230
|
+
* The per-step fields `tangle.rollout.v1` does not declare, parked here when
|
|
231
|
+
* the gate projected `steps[]` down to the schema's own key set.
|
|
232
|
+
*
|
|
233
|
+
* A per-step reward is training signal exactly like the scalar, and `steps`
|
|
234
|
+
* rides through `toRewardRows` verbatim — so a gated line was shipping its
|
|
235
|
+
* step-level credit assignment at full value beside a `reward` of 0.
|
|
236
|
+
*/
|
|
237
|
+
steps?: unknown;
|
|
238
|
+
}
|
|
141
239
|
interface RolloutProvenance {
|
|
142
240
|
captured_at: string;
|
|
143
241
|
capture: RolloutCapture;
|
|
144
|
-
/**
|
|
242
|
+
/**
|
|
243
|
+
* Why this line is incomplete. Required when `messages` is empty (the
|
|
244
|
+
* transcript could not be recovered); also set by interchange importers to
|
|
245
|
+
* name a MISSING LABEL — an imported trajectory carries no verdict, so
|
|
246
|
+
* `outcome.reward` is null and this says why.
|
|
247
|
+
*/
|
|
145
248
|
gap?: string;
|
|
249
|
+
/**
|
|
250
|
+
* Present only on a realness-gated line: the outcome fields the gate
|
|
251
|
+
* relocated, kept so an auditor can still see WHY the run was gated and what
|
|
252
|
+
* it claimed. Deliberately OUTSIDE `outcome`, because every training exporter
|
|
253
|
+
* reads `outcome` and none reads `provenance`.
|
|
254
|
+
*/
|
|
255
|
+
gated_evidence?: GatedEvidence;
|
|
146
256
|
}
|
|
147
257
|
interface RolloutLine {
|
|
148
258
|
schema: typeof ROLLOUT_SCHEMA;
|
|
@@ -27,9 +27,10 @@ import {
|
|
|
27
27
|
unavailable,
|
|
28
28
|
writeSupervisorRunReport,
|
|
29
29
|
writeSupervisorRunReportSafe
|
|
30
|
-
} from "../chunk-
|
|
31
|
-
import "../chunk-
|
|
32
|
-
import "../chunk-
|
|
30
|
+
} from "../chunk-X4YIBDER.js";
|
|
31
|
+
import "../chunk-HPWUNB47.js";
|
|
32
|
+
import "../chunk-PC5DOSM7.js";
|
|
33
|
+
import "../chunk-OIUOT4QD.js";
|
|
33
34
|
import "../chunk-PZ5AY32C.js";
|
|
34
35
|
export {
|
|
35
36
|
DEFAULT_CANCEL_TOOLS,
|
package/dist/traces.d.ts
CHANGED
|
@@ -1972,7 +1972,7 @@ interface OtlpFlatLine {
|
|
|
1972
1972
|
}>;
|
|
1973
1973
|
}
|
|
1974
1974
|
interface FlattenOtlpOptions {
|
|
1975
|
-
/** `'openinference'` (default)
|
|
1975
|
+
/** `'openinference'` (default) maps source per-span attributes into the
|
|
1976
1976
|
* canonical OpenInference vocabulary the analyst readers consume. `'none'`
|
|
1977
1977
|
* passes attributes through untouched. */
|
|
1978
1978
|
attributeVocabulary?: 'openinference' | 'none';
|
package/dist/traces.js
CHANGED
|
@@ -1,6 +1,4 @@
|
|
|
1
1
|
import {
|
|
2
|
-
FileSystemTraceStore,
|
|
3
|
-
InMemoryTraceStore,
|
|
4
2
|
OTEL_AGENT_EVAL_SCOPE,
|
|
5
3
|
ReplayCache,
|
|
6
4
|
ReplayCacheMissError,
|
|
@@ -27,7 +25,7 @@ import {
|
|
|
27
25
|
scoreTraceInsightReadiness,
|
|
28
26
|
tokenizeDomainWords,
|
|
29
27
|
traceAnalystOnRunComplete
|
|
30
|
-
} from "./chunk-
|
|
28
|
+
} from "./chunk-U4L7JRPZ.js";
|
|
31
29
|
import "./chunk-7ZZMD7UK.js";
|
|
32
30
|
import {
|
|
33
31
|
extractUsage,
|
|
@@ -78,10 +76,12 @@ import {
|
|
|
78
76
|
traceSpanKindToOpenInferenceKind
|
|
79
77
|
} from "./chunk-P6FYH6K4.js";
|
|
80
78
|
import {
|
|
79
|
+
FileSystemTraceStore,
|
|
80
|
+
InMemoryTraceStore,
|
|
81
81
|
RunIntegrityError,
|
|
82
82
|
assertRunCaptured,
|
|
83
83
|
throwIfRunIncomplete
|
|
84
|
-
} from "./chunk-
|
|
84
|
+
} from "./chunk-U4PHLT2N.js";
|
|
85
85
|
import {
|
|
86
86
|
FileSystemRawProviderSink,
|
|
87
87
|
InMemoryRawProviderSink,
|
|
@@ -93,7 +93,7 @@ import {
|
|
|
93
93
|
TraceEmitter,
|
|
94
94
|
llmSpanFromProvider
|
|
95
95
|
} from "./chunk-VQMK5FMP.js";
|
|
96
|
-
import "./chunk-
|
|
96
|
+
import "./chunk-56TAVBOK.js";
|
|
97
97
|
import {
|
|
98
98
|
FAILURE_CLASSES,
|
|
99
99
|
TRACE_SCHEMA_VERSION,
|
|
@@ -103,6 +103,7 @@ import {
|
|
|
103
103
|
isSandboxSpan,
|
|
104
104
|
isToolSpan
|
|
105
105
|
} from "./chunk-MA6HLL3S.js";
|
|
106
|
+
import "./chunk-OIUOT4QD.js";
|
|
106
107
|
import "./chunk-ONWEPEDO.js";
|
|
107
108
|
import {
|
|
108
109
|
INPUT_VALUE,
|
package/dist/wire/index.d.ts
CHANGED
|
@@ -652,9 +652,8 @@ type ProviderRedactor = (event: RawProviderEvent) => RawProviderEvent;
|
|
|
652
652
|
* { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
|
|
653
653
|
* )
|
|
654
654
|
*
|
|
655
|
-
*
|
|
656
|
-
*
|
|
657
|
-
* that need free-form text use `callLlm` and parse output themselves.
|
|
655
|
+
* `createChatClient` wraps this implementation for provider-neutral package
|
|
656
|
+
* entry points. Direct callers can use `callLlm` or `callLlmJson`.
|
|
658
657
|
*/
|
|
659
658
|
|
|
660
659
|
type LlmThinkingMode = 'enabled' | 'disabled';
|
|
@@ -688,8 +687,8 @@ interface LlmClientOptions {
|
|
|
688
687
|
* total attempts × `timeoutMs`.
|
|
689
688
|
*/
|
|
690
689
|
deadlineMs?: number;
|
|
691
|
-
/** Total provider attempts.
|
|
692
|
-
|
|
690
|
+
/** Total provider attempts. Default 3. */
|
|
691
|
+
maximumAttempts?: number;
|
|
693
692
|
/** Token rates used when the provider omits cost or package pricing does not cover the model. */
|
|
694
693
|
customTokenPricing?: CustomTokenPricing;
|
|
695
694
|
/**
|
|
@@ -909,8 +908,8 @@ declare const FeedbackLabelSchema: z.ZodObject<{
|
|
|
909
908
|
id: z.ZodOptional<z.ZodString>;
|
|
910
909
|
source: z.ZodEnum<{
|
|
911
910
|
judge: "judge";
|
|
912
|
-
user: "user";
|
|
913
911
|
system: "system";
|
|
912
|
+
user: "user";
|
|
914
913
|
policy: "policy";
|
|
915
914
|
environment: "environment";
|
|
916
915
|
metric: "metric";
|
|
@@ -957,9 +956,9 @@ declare const FeedbackAttemptSchema: z.ZodObject<{
|
|
|
957
956
|
proposedAction: z.ZodOptional<z.ZodObject<{
|
|
958
957
|
type: z.ZodString;
|
|
959
958
|
risk: z.ZodOptional<z.ZodEnum<{
|
|
960
|
-
medium: "medium";
|
|
961
959
|
low: "low";
|
|
962
960
|
high: "high";
|
|
961
|
+
medium: "medium";
|
|
963
962
|
}>>;
|
|
964
963
|
costUsd: z.ZodOptional<z.ZodNumber>;
|
|
965
964
|
externalSideEffect: z.ZodOptional<z.ZodBoolean>;
|
|
@@ -970,8 +969,8 @@ declare const FeedbackAttemptSchema: z.ZodObject<{
|
|
|
970
969
|
id: z.ZodOptional<z.ZodString>;
|
|
971
970
|
source: z.ZodEnum<{
|
|
972
971
|
judge: "judge";
|
|
973
|
-
user: "user";
|
|
974
972
|
system: "system";
|
|
973
|
+
user: "user";
|
|
975
974
|
policy: "policy";
|
|
976
975
|
environment: "environment";
|
|
977
976
|
metric: "metric";
|
|
@@ -1029,9 +1028,9 @@ declare const FeedbackTrajectorySchema: z.ZodObject<{
|
|
|
1029
1028
|
proposedAction: z.ZodOptional<z.ZodObject<{
|
|
1030
1029
|
type: z.ZodString;
|
|
1031
1030
|
risk: z.ZodOptional<z.ZodEnum<{
|
|
1032
|
-
medium: "medium";
|
|
1033
1031
|
low: "low";
|
|
1034
1032
|
high: "high";
|
|
1033
|
+
medium: "medium";
|
|
1035
1034
|
}>>;
|
|
1036
1035
|
costUsd: z.ZodOptional<z.ZodNumber>;
|
|
1037
1036
|
externalSideEffect: z.ZodOptional<z.ZodBoolean>;
|
|
@@ -1042,8 +1041,8 @@ declare const FeedbackTrajectorySchema: z.ZodObject<{
|
|
|
1042
1041
|
id: z.ZodOptional<z.ZodString>;
|
|
1043
1042
|
source: z.ZodEnum<{
|
|
1044
1043
|
judge: "judge";
|
|
1045
|
-
user: "user";
|
|
1046
1044
|
system: "system";
|
|
1045
|
+
user: "user";
|
|
1047
1046
|
policy: "policy";
|
|
1048
1047
|
environment: "environment";
|
|
1049
1048
|
metric: "metric";
|
|
@@ -1078,8 +1077,8 @@ declare const FeedbackTrajectorySchema: z.ZodObject<{
|
|
|
1078
1077
|
id: z.ZodOptional<z.ZodString>;
|
|
1079
1078
|
source: z.ZodEnum<{
|
|
1080
1079
|
judge: "judge";
|
|
1081
|
-
user: "user";
|
|
1082
1080
|
system: "system";
|
|
1081
|
+
user: "user";
|
|
1083
1082
|
policy: "policy";
|
|
1084
1083
|
environment: "environment";
|
|
1085
1084
|
metric: "metric";
|
package/dist/wire/index.js
CHANGED
|
@@ -34,9 +34,9 @@ import {
|
|
|
34
34
|
runRpcOnce,
|
|
35
35
|
startServer,
|
|
36
36
|
startServerAsync
|
|
37
|
-
} from "../chunk-
|
|
38
|
-
import "../chunk-
|
|
39
|
-
import "../chunk-
|
|
37
|
+
} from "../chunk-NY44NC4A.js";
|
|
38
|
+
import "../chunk-SFLLL76A.js";
|
|
39
|
+
import "../chunk-VCZ5FQYW.js";
|
|
40
40
|
import "../chunk-VI2UW6B6.js";
|
|
41
41
|
import "../chunk-PC4UYEBM.js";
|
|
42
42
|
import "../chunk-ONWEPEDO.js";
|
package/docs/feature-guide.md
CHANGED
|
@@ -151,7 +151,7 @@ Store as `FeedbackTrajectory`, then derive:
|
|
|
151
151
|
|
|
152
152
|
| Area | Key exports | Best for | Notes |
|
|
153
153
|
| --- | --- | --- | --- |
|
|
154
|
-
| Judging | `
|
|
154
|
+
| Judging | `llmJudge`, semantic judges, anti-slop, wire rubrics | Content, voice, semantic quality | Pair with objective checks when possible. |
|
|
155
155
|
| Verification | `MultiLayerVerifier`, `JudgeRunner`, sandbox harness | Code and multi-step gates | Do not let semantic judges override failed builds. |
|
|
156
156
|
| Control | `runAgentControlLoop`, `objectiveEval`, `subjectiveEval` | Long-running agent tasks | Supports budgets, cost, stop policies, trace spans. |
|
|
157
157
|
| Propose/review | `runProposeReview`, `runProposeReviewAsControlLoop` | Iterative artifact repair | Good for code, docs, plans, briefs. |
|