@hybridlabor-api/bdb-synapse 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +173 -0
- package/bin/synapse +0 -0
- package/cmd/rubriceval/main.go +308 -0
- package/cmd/synapse/main.go +280 -0
- package/cmd/synapse/main_test.go +16 -0
- package/go.mod +7 -0
- package/go.sum +6 -0
- package/internal/adapter/adapter.go +1117 -0
- package/internal/adapter/adapter_test.go +518 -0
- package/internal/adapter/agy/adapter.go +193 -0
- package/internal/adapter/claudecode/adapter.go +415 -0
- package/internal/adapter/claudecode/adapter_test.go +260 -0
- package/internal/adapter/claudecode/agents.go +387 -0
- package/internal/adapter/claudecode/agents_test.go +480 -0
- package/internal/adapter/claudecode/summary_inputs_test.go +19 -0
- package/internal/adapter/codex/adapter.go +919 -0
- package/internal/adapter/codex/adapter_test.go +920 -0
- package/internal/adapter/codex/agents.go +401 -0
- package/internal/adapter/codex/agents_test.go +610 -0
- package/internal/adapter/codex/summary_inputs_test.go +20 -0
- package/internal/adapter/pi/adapter.go +483 -0
- package/internal/adapter/pi/adapter_test.go +517 -0
- package/internal/citymap/builder.go +1124 -0
- package/internal/citymap/builder_test.go +818 -0
- package/internal/judge/cache.go +180 -0
- package/internal/judge/cli.go +240 -0
- package/internal/judge/cli_test.go +68 -0
- package/internal/judge/fresh_summary_test.go +37 -0
- package/internal/judge/input.go +233 -0
- package/internal/judge/judge.go +416 -0
- package/internal/judge/judge_test.go +288 -0
- package/internal/judge/prompt.go +129 -0
- package/internal/judge/rubric.go +275 -0
- package/internal/judge/rubric_test.go +642 -0
- package/internal/model/agent.go +63 -0
- package/internal/model/agent_schema_test.go +86 -0
- package/internal/model/agent_test.go +64 -0
- package/internal/model/model.go +175 -0
- package/internal/model/report.go +166 -0
- package/internal/model/stats.go +151 -0
- package/internal/model/stats_test.go +89 -0
- package/internal/model/trace_schema_test.go +67 -0
- package/internal/server/analyze.go +286 -0
- package/internal/server/analyze_test.go +297 -0
- package/internal/server/codex_index_test.go +40 -0
- package/internal/server/hardening_test.go +147 -0
- package/internal/server/reportindex.go +91 -0
- package/internal/server/reportindex_test.go +116 -0
- package/internal/server/server.go +1099 -0
- package/internal/server/server_test.go +1389 -0
- package/internal/server/static/assets/fraunces-latin-ext-standard-italic-CGbN9UgK.woff2 +0 -0
- package/internal/server/static/assets/fraunces-latin-ext-standard-normal-CJcjJNj7.woff2 +0 -0
- package/internal/server/static/assets/fraunces-latin-standard-italic-lSdLDfvT.woff2 +0 -0
- package/internal/server/static/assets/fraunces-latin-standard-normal-DihXLNYH.woff2 +0 -0
- package/internal/server/static/assets/fraunces-vietnamese-standard-italic-DxWqP7Ku.woff2 +0 -0
- package/internal/server/static/assets/fraunces-vietnamese-standard-normal-Czevyj-6.woff2 +0 -0
- package/internal/server/static/assets/index-BNoY_BiB.css +1 -0
- package/internal/server/static/assets/index-C_adLrJr.js +3 -0
- package/internal/server/static/assets/react-gcHzaSmV.js +10 -0
- package/internal/server/static/assets/schibsted-grotesk-latin-ext-wght-normal-hsMS0n0O.woff2 +0 -0
- package/internal/server/static/assets/schibsted-grotesk-latin-wght-normal-Bb8VGrTG.woff2 +0 -0
- package/internal/server/static/assets/three-DnGjZfD1.js +4012 -0
- package/internal/server/static/index.html +26 -0
- package/internal/server/tracestore.go +173 -0
- package/internal/server/tracestore_test.go +51 -0
- package/internal/textutil/truncate.go +30 -0
- package/internal/textutil/truncate_test.go +39 -0
- package/package.json +35 -0
- package/web/e2e/agent-lens.spec.ts +688 -0
- package/web/index.html +23 -0
- package/web/package-lock.json +1933 -0
- package/web/package.json +33 -0
- package/web/playwright.config.ts +24 -0
- package/web/src/App.tsx +876 -0
- package/web/src/api/client.ts +74 -0
- package/web/src/main.tsx +12 -0
- package/web/src/playback/recorder.ts +160 -0
- package/web/src/playback/reducer.ts +91 -0
- package/web/src/scene/CityScene.tsx +638 -0
- package/web/src/scene/TreeScene.tsx +656 -0
- package/web/src/scene/dirLabels.ts +145 -0
- package/web/src/scene/sceneUtils.ts +144 -0
- package/web/src/scene/textures.ts +60 -0
- package/web/src/scene/trail.ts +79 -0
- package/web/src/scene/treeLayout.ts +169 -0
- package/web/src/state/filters.ts +40 -0
- package/web/src/state/store.ts +83 -0
- package/web/src/styles.css +2565 -0
- package/web/src/types.ts +315 -0
- package/web/src/ui/AgentsPanel.tsx +376 -0
- package/web/src/ui/Dock.tsx +104 -0
- package/web/src/ui/Hud.tsx +335 -0
- package/web/src/ui/Inspector.tsx +107 -0
- package/web/src/ui/LogoMark.tsx +38 -0
- package/web/src/ui/ReportPanel.tsx +491 -0
- package/web/src/ui/SessionRail.tsx +316 -0
- package/web/src/ui/Timeline.tsx +458 -0
- package/web/src/ui/ViewPanel.tsx +45 -0
- package/web/src/ui/shortcuts.ts +4 -0
- package/web/tsconfig.json +21 -0
- package/web/vite.config.ts +28 -0
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
package judge
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"crypto/sha256"
|
|
5
|
+
"encoding/hex"
|
|
6
|
+
"encoding/json"
|
|
7
|
+
"fmt"
|
|
8
|
+
"sort"
|
|
9
|
+
"strconv"
|
|
10
|
+
"strings"
|
|
11
|
+
|
|
12
|
+
"github.com/hybridlabor-api/bdb-synapse/internal/adapter"
|
|
13
|
+
"github.com/hybridlabor-api/bdb-synapse/internal/model"
|
|
14
|
+
"github.com/hybridlabor-api/bdb-synapse/internal/textutil"
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
const (
|
|
18
|
+
maxUserMessages = 12
|
|
19
|
+
maxUserMessageLen = 600
|
|
20
|
+
maxSummaryLen = 160
|
|
21
|
+
maxNarrativeEvents = 2000
|
|
22
|
+
// maxTaskMessages bounds the rubric phase's task-evidence section — much
|
|
23
|
+
// wider than the scoring budget because the rubric must see every task
|
|
24
|
+
// the user raised. Anchors, the task digest, and the weak-text gate all
|
|
25
|
+
// read exactly this set; nothing rubric-related may consult a different
|
|
26
|
+
// message list.
|
|
27
|
+
maxTaskMessages = 48
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
// BuildInput renders one trace as the scoring judge's evidence document:
|
|
31
|
+
// session meta, budgeted user task wording, precomputed stats, and a
|
|
32
|
+
// one-line-per-event narrative. The judge reads only this — never the raw
|
|
33
|
+
// session log.
|
|
34
|
+
func BuildInput(trace *model.Trace) string {
|
|
35
|
+
return buildDocument(trace, renderedUserMessages(trace.Marks))
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
// BuildRubricInput renders the rubric generator's evidence document: same
|
|
39
|
+
// stats and narrative, but the user-message section is the full task
|
|
40
|
+
// evidence (taskMessages) rather than the scoring budget — a task raised in
|
|
41
|
+
// the middle of a long session must be visible to the generator that is
|
|
42
|
+
// asked to enumerate tasks.
|
|
43
|
+
func BuildRubricInput(trace *model.Trace) string {
|
|
44
|
+
return buildDocument(trace, taskMessages(trace.Marks))
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
func buildDocument(trace *model.Trace, keep []userMessage) string {
|
|
48
|
+
var b strings.Builder
|
|
49
|
+
sess := trace.Session
|
|
50
|
+
b.WriteString("# Session under evaluation\n\n")
|
|
51
|
+
fmt.Fprintf(&b, "- harness: %s model: %s\n", sess.Harness, orUnknown(sess.Model))
|
|
52
|
+
fmt.Fprintf(&b, "- cwd: %s events: %d\n", sess.Cwd, sess.EventCount)
|
|
53
|
+
fmt.Fprintf(&b, "- started: %s ended: %s\n\n", sess.StartedAt, sess.EndedAt)
|
|
54
|
+
|
|
55
|
+
writeMessages(&b, keep)
|
|
56
|
+
writeStats(&b, trace.Stats)
|
|
57
|
+
writeNarrative(&b, trace)
|
|
58
|
+
return b.String()
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
// InputDigest fingerprints the evidence document BuildInput renders for the
|
|
62
|
+
// trace. Unlike a bare event count it moves when user messages, tool results,
|
|
63
|
+
// or stats change, so freshness checks see every input the judge saw.
|
|
64
|
+
func InputDigest(trace *model.Trace) string {
|
|
65
|
+
sum := sha256.Sum256([]byte(BuildInput(trace)))
|
|
66
|
+
return hex.EncodeToString(sum[:])
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
// userMessage is one user-message mark after injected-wrapper filtering.
|
|
70
|
+
// Ordinal is 1-based over the filtered list and is what the evidence document
|
|
71
|
+
// renders as [user #N]; seq is the mark seq it resolves to.
|
|
72
|
+
type userMessage struct {
|
|
73
|
+
ordinal int
|
|
74
|
+
seq int
|
|
75
|
+
text string
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
// filteredUserMessages returns every user message that may appear in the
|
|
79
|
+
// evidence document, before the rendering budget is applied.
|
|
80
|
+
func filteredUserMessages(marks []model.Mark) []userMessage {
|
|
81
|
+
var messages []userMessage
|
|
82
|
+
for _, mark := range marks {
|
|
83
|
+
if mark.Type != "user-message" {
|
|
84
|
+
continue
|
|
85
|
+
}
|
|
86
|
+
text := strings.TrimSpace(mark.Note)
|
|
87
|
+
// Adapters already drop injected wrappers before marks exist; the
|
|
88
|
+
// re-check here keeps judge input clean even for traces built by
|
|
89
|
+
// older adapters.
|
|
90
|
+
if text == "" || adapter.InjectedUserMessage(text) {
|
|
91
|
+
continue
|
|
92
|
+
}
|
|
93
|
+
messages = append(messages, userMessage{ordinal: len(messages) + 1, seq: mark.Seq, text: text})
|
|
94
|
+
}
|
|
95
|
+
return messages
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
// budgetMessages keeps the first message (it states the task) plus the
|
|
99
|
+
// newest budget-1 (they carry corrections); mid-session chatter gives way.
|
|
100
|
+
func budgetMessages(messages []userMessage, budget int) []userMessage {
|
|
101
|
+
if len(messages) > budget {
|
|
102
|
+
messages = append([]userMessage{messages[0]}, messages[len(messages)-(budget-1):]...)
|
|
103
|
+
}
|
|
104
|
+
return messages
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
// renderedUserMessages is the scoring document's message budget.
|
|
108
|
+
func renderedUserMessages(marks []model.Mark) []userMessage {
|
|
109
|
+
return budgetMessages(filteredUserMessages(marks), maxUserMessages)
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
// taskMessages is the single task-evidence set of the rubric phase: what the
|
|
113
|
+
// generator reads, what anchors may reference, what the digest fingerprints,
|
|
114
|
+
// and what the weak-text gate measures.
|
|
115
|
+
func taskMessages(marks []model.Mark) []userMessage {
|
|
116
|
+
return budgetMessages(filteredUserMessages(marks), maxTaskMessages)
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
func writeMessages(b *strings.Builder, keep []userMessage) {
|
|
120
|
+
b.WriteString("## User messages (the task; later ones are follow-ups/corrections)\n\n")
|
|
121
|
+
if len(keep) == 0 {
|
|
122
|
+
b.WriteString("(no user message text available)\n\n")
|
|
123
|
+
return
|
|
124
|
+
}
|
|
125
|
+
previous := 0
|
|
126
|
+
for _, message := range keep {
|
|
127
|
+
if message.ordinal != previous+1 {
|
|
128
|
+
fmt.Fprintf(b, "…%d intermediate user messages omitted.\n\n", message.ordinal-previous-1)
|
|
129
|
+
}
|
|
130
|
+
previous = message.ordinal
|
|
131
|
+
fmt.Fprintf(b, "[user #%d] %s\n\n", message.ordinal, truncateRunes(message.text, maxUserMessageLen))
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
// taskTextRunes measures the task signal available to a rubric — over the
|
|
136
|
+
// same set the generator will actually see.
|
|
137
|
+
func taskTextRunes(marks []model.Mark) int {
|
|
138
|
+
total := 0
|
|
139
|
+
for _, message := range taskMessages(marks) {
|
|
140
|
+
total += len([]rune(message.text))
|
|
141
|
+
}
|
|
142
|
+
return total
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
// taskSection renders the rubric phase's user-message section.
|
|
146
|
+
func taskSection(marks []model.Mark) string {
|
|
147
|
+
var b strings.Builder
|
|
148
|
+
writeMessages(&b, taskMessages(marks))
|
|
149
|
+
return b.String()
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
// TaskDigest fingerprints the exact task section the rubric generator reads.
|
|
153
|
+
// Unlike InputDigest it ignores events and stats: a session that only grew
|
|
154
|
+
// in activity keeps its rubric, while any change to the task wording, the
|
|
155
|
+
// generation source mode, or the rubric prompt forces regeneration.
|
|
156
|
+
func TaskDigest(trace *model.Trace, source string) string {
|
|
157
|
+
sum := sha256.Sum256([]byte(strings.Join([]string{
|
|
158
|
+
trace.Session.Harness,
|
|
159
|
+
taskSection(trace.Marks),
|
|
160
|
+
source,
|
|
161
|
+
strconv.Itoa(RubricPromptVersion),
|
|
162
|
+
}, "\n")))
|
|
163
|
+
return hex.EncodeToString(sum[:])
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
func writeStats(b *strings.Builder, stats model.Stats) {
|
|
167
|
+
b.WriteString("## Deterministic stats (precomputed, trust these numbers)\n\n")
|
|
168
|
+
encoded, err := json.MarshalIndent(stats, "", " ")
|
|
169
|
+
if err != nil {
|
|
170
|
+
encoded = []byte("{}")
|
|
171
|
+
}
|
|
172
|
+
b.WriteString("```json\n")
|
|
173
|
+
b.Write(encoded)
|
|
174
|
+
b.WriteString("\n```\n\n")
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
func writeNarrative(b *strings.Builder, trace *model.Trace) {
|
|
178
|
+
b.WriteString("## Event narrative (seq | action | targets | summary; ERR = tool errored)\n\n")
|
|
179
|
+
marksBySeq := map[int][]string{}
|
|
180
|
+
for _, mark := range trace.Marks {
|
|
181
|
+
marksBySeq[mark.Seq] = append(marksBySeq[mark.Seq], mark.Type)
|
|
182
|
+
}
|
|
183
|
+
seqs := make([]int, 0, len(marksBySeq))
|
|
184
|
+
for seq := range marksBySeq {
|
|
185
|
+
seqs = append(seqs, seq)
|
|
186
|
+
}
|
|
187
|
+
sort.Ints(seqs)
|
|
188
|
+
|
|
189
|
+
for i, event := range trace.Events {
|
|
190
|
+
if i >= maxNarrativeEvents {
|
|
191
|
+
fmt.Fprintf(b, "…%d later events omitted.\n", len(trace.Events)-maxNarrativeEvents)
|
|
192
|
+
break
|
|
193
|
+
}
|
|
194
|
+
for _, markType := range marksBySeq[event.Seq] {
|
|
195
|
+
fmt.Fprintf(b, "--- mark: %s ---\n", markType)
|
|
196
|
+
}
|
|
197
|
+
paths := make([]string, 0, 3)
|
|
198
|
+
for _, target := range event.Targets {
|
|
199
|
+
if len(paths) == 3 {
|
|
200
|
+
break
|
|
201
|
+
}
|
|
202
|
+
paths = append(paths, target.Path)
|
|
203
|
+
}
|
|
204
|
+
pathList := "-"
|
|
205
|
+
if len(paths) > 0 {
|
|
206
|
+
pathList = strings.Join(paths, ",")
|
|
207
|
+
}
|
|
208
|
+
errFlag := ""
|
|
209
|
+
if event.IsError {
|
|
210
|
+
errFlag = " ERR"
|
|
211
|
+
}
|
|
212
|
+
fmt.Fprintf(b, "%d | %s%s | %s | %s\n", event.Seq, event.Action, errFlag, pathList, truncateRunes(event.Summary, maxSummaryLen))
|
|
213
|
+
}
|
|
214
|
+
// Marks that point past the last event (e.g. a closing user message).
|
|
215
|
+
for _, seq := range seqs {
|
|
216
|
+
if seq >= len(trace.Events) {
|
|
217
|
+
for _, markType := range marksBySeq[seq] {
|
|
218
|
+
fmt.Fprintf(b, "--- mark: %s ---\n", markType)
|
|
219
|
+
}
|
|
220
|
+
}
|
|
221
|
+
}
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
func truncateRunes(s string, limit int) string {
|
|
225
|
+
return textutil.TruncateRunes(s, limit, " …[truncated]")
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
func orUnknown(s string) string {
|
|
229
|
+
if s == "" {
|
|
230
|
+
return "?"
|
|
231
|
+
}
|
|
232
|
+
return s
|
|
233
|
+
}
|
|
@@ -0,0 +1,416 @@
|
|
|
1
|
+
package judge
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"context"
|
|
5
|
+
"encoding/json"
|
|
6
|
+
"fmt"
|
|
7
|
+
"strings"
|
|
8
|
+
"time"
|
|
9
|
+
|
|
10
|
+
"github.com/hybridlabor-api/bdb-synapse/internal/model"
|
|
11
|
+
)
|
|
12
|
+
|
|
13
|
+
// DefaultTimeout bounds one whole evaluation. Two sealed calls (rubric, then
|
|
14
|
+
// scoring) at a measured ~30-40s each leave the same headroom the single-call
|
|
15
|
+
// pipeline had at five minutes.
|
|
16
|
+
const DefaultTimeout = 10 * time.Minute
|
|
17
|
+
|
|
18
|
+
type Options struct {
|
|
19
|
+
// Runner overrides the subprocess runner; nil selects CLIRunner{CLI, Model}.
|
|
20
|
+
Runner Runner
|
|
21
|
+
// CLI names the judge CLI ("claude" or "codex"); empty auto-detects.
|
|
22
|
+
CLI string
|
|
23
|
+
// Model overrides the CLI's default model; empty keeps the default.
|
|
24
|
+
Model string
|
|
25
|
+
// NoRubric skips the rubric layer entirely: one dimensions-only call,
|
|
26
|
+
// and the report carries no rubric block.
|
|
27
|
+
NoRubric bool
|
|
28
|
+
// CachedReport is the previous report for this session, if any; a scored
|
|
29
|
+
// rubric whose task digest still matches is reused instead of regenerated,
|
|
30
|
+
// so criteria stay stable across re-evaluations.
|
|
31
|
+
CachedReport *model.Report
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
// Analyze runs the judge over one trace and returns the evaluation report.
|
|
35
|
+
// The rubric layer resolves first (skip, reuse, or generate-with-degrade);
|
|
36
|
+
// scoring is a single unified call covering the four fixed dimensions plus
|
|
37
|
+
// any rubric criteria. The judge only contributes findings; verdicts are
|
|
38
|
+
// rolled up mechanically. Invalid judge output is retried once before failing.
|
|
39
|
+
func Analyze(ctx context.Context, trace *model.Trace, opts Options) (*model.Report, error) {
|
|
40
|
+
runner := opts.Runner
|
|
41
|
+
if runner == nil {
|
|
42
|
+
cli := opts.CLI
|
|
43
|
+
if cli == "" {
|
|
44
|
+
detected, err := DetectCLI()
|
|
45
|
+
if err != nil {
|
|
46
|
+
return nil, err
|
|
47
|
+
}
|
|
48
|
+
cli = detected
|
|
49
|
+
}
|
|
50
|
+
runner = CLIRunner{CLI: cli, Model: opts.Model}
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
input := BuildInput(trace)
|
|
54
|
+
var rubric *model.Rubric
|
|
55
|
+
if !opts.NoRubric {
|
|
56
|
+
acquired, err := acquireRubric(ctx, runner, trace, opts.CachedReport)
|
|
57
|
+
if err != nil {
|
|
58
|
+
return nil, err
|
|
59
|
+
}
|
|
60
|
+
rubric = acquired
|
|
61
|
+
}
|
|
62
|
+
sysPrompt, scoringInput := prompt, input
|
|
63
|
+
if rubric != nil && rubric.Status == model.RubricStatusScored {
|
|
64
|
+
sysPrompt = scoringPrompt
|
|
65
|
+
scoringInput = "# RUBRIC (data)\n\n" + scoringRubricJSON(rubric) + "\n\n# SESSION\n\n" + input
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
var lastErr error
|
|
69
|
+
for attempt := 0; attempt < 2; attempt++ {
|
|
70
|
+
result, err := runner.Run(ctx, sysPrompt, scoringInput)
|
|
71
|
+
if err != nil {
|
|
72
|
+
return nil, err
|
|
73
|
+
}
|
|
74
|
+
report, err := parseOutput(result.Text, trace, rubric)
|
|
75
|
+
if err != nil {
|
|
76
|
+
lastErr = err
|
|
77
|
+
continue
|
|
78
|
+
}
|
|
79
|
+
// Prefer the model the CLI says it used; fall back to what was asked
|
|
80
|
+
// for so the report never silently drops the information.
|
|
81
|
+
judgeModel := result.Model
|
|
82
|
+
if judgeModel == "" {
|
|
83
|
+
judgeModel = opts.Model
|
|
84
|
+
}
|
|
85
|
+
report.Judge = model.ReportJudge{
|
|
86
|
+
CLI: runner.Name(),
|
|
87
|
+
Model: judgeModel,
|
|
88
|
+
RequestedModel: opts.Model,
|
|
89
|
+
PromptVersion: PromptVersion,
|
|
90
|
+
GeneratedAt: time.Now().UTC().Format(time.RFC3339),
|
|
91
|
+
InputDigest: InputDigest(trace),
|
|
92
|
+
}
|
|
93
|
+
if report.Rubric != nil && report.Rubric.Status == model.RubricStatusScored {
|
|
94
|
+
report.Judge.RubricPromptVersion = RubricPromptVersion
|
|
95
|
+
}
|
|
96
|
+
return report, nil
|
|
97
|
+
}
|
|
98
|
+
return nil, fmt.Errorf("judge output invalid after retry: %w", lastErr)
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
// llmFinding and llmOutput mirror the JSON shapes the scoring prompts request.
|
|
102
|
+
type llmFinding struct {
|
|
103
|
+
Claim string `json:"claim"`
|
|
104
|
+
Severity string `json:"severity"`
|
|
105
|
+
EvidenceSeqs []int `json:"evidence_seqs"`
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
type llmOutput struct {
|
|
109
|
+
TaskSummary string `json:"task_summary"`
|
|
110
|
+
Dimensions []struct {
|
|
111
|
+
Name string `json:"name"`
|
|
112
|
+
Findings []llmFinding `json:"findings"`
|
|
113
|
+
} `json:"dimensions"`
|
|
114
|
+
Criteria []struct {
|
|
115
|
+
ID string `json:"id"`
|
|
116
|
+
Coverage string `json:"coverage"`
|
|
117
|
+
Findings []llmFinding `json:"findings"`
|
|
118
|
+
} `json:"criteria"`
|
|
119
|
+
RubricNote string `json:"rubric_note"`
|
|
120
|
+
NotableMoments []struct {
|
|
121
|
+
Seq int `json:"seq"`
|
|
122
|
+
Note string `json:"note"`
|
|
123
|
+
} `json:"notable_moments"`
|
|
124
|
+
Narrative string `json:"narrative"`
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
func parseOutput(raw string, trace *model.Trace, rubric *model.Rubric) (*model.Report, error) {
|
|
128
|
+
payload, err := extractJSON(raw)
|
|
129
|
+
if err != nil {
|
|
130
|
+
return nil, err
|
|
131
|
+
}
|
|
132
|
+
var out llmOutput
|
|
133
|
+
if err := json.Unmarshal([]byte(payload), &out); err != nil {
|
|
134
|
+
return nil, fmt.Errorf("judge JSON: %w", err)
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
validSeqs := make(map[int]bool, len(trace.Events))
|
|
138
|
+
for _, event := range trace.Events {
|
|
139
|
+
validSeqs[event.Seq] = true
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
byName := map[string]*model.ReportDimension{}
|
|
143
|
+
for _, dim := range out.Dimensions {
|
|
144
|
+
if !knownDimension(dim.Name) {
|
|
145
|
+
continue
|
|
146
|
+
}
|
|
147
|
+
target, ok := byName[dim.Name]
|
|
148
|
+
if !ok {
|
|
149
|
+
target = &model.ReportDimension{Name: dim.Name, Findings: []model.ReportFinding{}}
|
|
150
|
+
byName[dim.Name] = target
|
|
151
|
+
}
|
|
152
|
+
findings, err := filterFindings(dim.Findings, validSeqs)
|
|
153
|
+
if err != nil {
|
|
154
|
+
return nil, err
|
|
155
|
+
}
|
|
156
|
+
target.Findings = append(target.Findings, findings...)
|
|
157
|
+
}
|
|
158
|
+
if len(byName) != len(model.DimensionNames) {
|
|
159
|
+
return nil, fmt.Errorf("judge output covers %d of %d dimensions", len(byName), len(model.DimensionNames))
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
report := &model.Report{
|
|
163
|
+
Version: 1,
|
|
164
|
+
Session: model.ReportSession{
|
|
165
|
+
ID: trace.Session.ID,
|
|
166
|
+
Harness: trace.Session.Harness,
|
|
167
|
+
Model: trace.Session.Model,
|
|
168
|
+
EventCount: trace.Session.EventCount,
|
|
169
|
+
UserTurns: trace.Stats.UserTurns,
|
|
170
|
+
},
|
|
171
|
+
TaskSummary: out.TaskSummary,
|
|
172
|
+
Narrative: out.Narrative,
|
|
173
|
+
}
|
|
174
|
+
for _, name := range model.DimensionNames {
|
|
175
|
+
dim := byName[name]
|
|
176
|
+
if len(trace.Events) == 0 {
|
|
177
|
+
// A conversation-only trace has nothing citable: every finding was
|
|
178
|
+
// just dropped, and a good verdict here would be praise on zero
|
|
179
|
+
// evidence. No events, no signal — for all four dimensions.
|
|
180
|
+
dim.Verdict = model.VerdictInsufficientData
|
|
181
|
+
} else {
|
|
182
|
+
dim.Verdict = rollupVerdict(name, dim.Findings, trace.Stats.Observability)
|
|
183
|
+
}
|
|
184
|
+
report.Dimensions = append(report.Dimensions, *dim)
|
|
185
|
+
}
|
|
186
|
+
if rubric != nil {
|
|
187
|
+
if rubric.Status == model.RubricStatusScored {
|
|
188
|
+
scored, err := scoreRubric(rubric, &out, validSeqs)
|
|
189
|
+
if err != nil {
|
|
190
|
+
return nil, err
|
|
191
|
+
}
|
|
192
|
+
report.Rubric = scored
|
|
193
|
+
} else {
|
|
194
|
+
report.Rubric = rubric
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
for _, moment := range out.NotableMoments {
|
|
198
|
+
if validSeqs[moment.Seq] && moment.Note != "" {
|
|
199
|
+
report.NotableMoments = append(report.NotableMoments, model.ReportMoment{Seq: moment.Seq, Note: moment.Note})
|
|
200
|
+
}
|
|
201
|
+
}
|
|
202
|
+
return report, nil
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
// filterFindings applies the evidence discipline shared by dimensions and
|
|
206
|
+
// rubric criteria: hallucinated seqs are stripped, a finding with no valid
|
|
207
|
+
// citation left may not enter the report, and an unrecognized severity
|
|
208
|
+
// invalidates the whole output — silently downgrading a misspelled "problem"
|
|
209
|
+
// to info would launder a red flag into a good verdict.
|
|
210
|
+
func filterFindings(raw []llmFinding, validSeqs map[int]bool) ([]model.ReportFinding, error) {
|
|
211
|
+
findings := make([]model.ReportFinding, 0, len(raw))
|
|
212
|
+
for _, finding := range raw {
|
|
213
|
+
if finding.Claim == "" {
|
|
214
|
+
continue
|
|
215
|
+
}
|
|
216
|
+
seqs := make([]int, 0, len(finding.EvidenceSeqs))
|
|
217
|
+
for _, seq := range finding.EvidenceSeqs {
|
|
218
|
+
if validSeqs[seq] {
|
|
219
|
+
seqs = append(seqs, seq)
|
|
220
|
+
}
|
|
221
|
+
}
|
|
222
|
+
if len(seqs) == 0 {
|
|
223
|
+
continue
|
|
224
|
+
}
|
|
225
|
+
severity, err := normalizeSeverity(finding.Severity)
|
|
226
|
+
if err != nil {
|
|
227
|
+
return nil, err
|
|
228
|
+
}
|
|
229
|
+
findings = append(findings, model.ReportFinding{
|
|
230
|
+
Claim: finding.Claim,
|
|
231
|
+
Severity: severity,
|
|
232
|
+
EvidenceSeqs: seqs,
|
|
233
|
+
})
|
|
234
|
+
}
|
|
235
|
+
return findings, nil
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
// scoreRubric merges the scoring output into a copy of the rubric. The judge
|
|
239
|
+
// echoes a flat criteria list; grouping comes from the rubric itself, so a
|
|
240
|
+
// criterion can never land in the wrong task. Every rubric criterion must be
|
|
241
|
+
// scored — a missing one invalidates the output; unknown and duplicate ids
|
|
242
|
+
// are dropped.
|
|
243
|
+
func scoreRubric(rubric *model.Rubric, out *llmOutput, validSeqs map[int]bool) (*model.Rubric, error) {
|
|
244
|
+
type score struct {
|
|
245
|
+
coverage string
|
|
246
|
+
findings []model.ReportFinding
|
|
247
|
+
}
|
|
248
|
+
expected := map[string]bool{}
|
|
249
|
+
for _, task := range rubric.Tasks {
|
|
250
|
+
for _, criterion := range task.Criteria {
|
|
251
|
+
expected[criterion.ID] = true
|
|
252
|
+
}
|
|
253
|
+
}
|
|
254
|
+
scores := map[string]score{}
|
|
255
|
+
for _, criterion := range out.Criteria {
|
|
256
|
+
// Unknown ids are dropped before any validation: an invented entry is
|
|
257
|
+
// noise the contract discards, and its malformed coverage or findings
|
|
258
|
+
// must not be able to fail the whole scoring pass.
|
|
259
|
+
if !expected[criterion.ID] {
|
|
260
|
+
continue
|
|
261
|
+
}
|
|
262
|
+
if _, dup := scores[criterion.ID]; dup {
|
|
263
|
+
continue
|
|
264
|
+
}
|
|
265
|
+
coverage, err := normalizeCoverage(criterion.Coverage)
|
|
266
|
+
if err != nil {
|
|
267
|
+
return nil, err
|
|
268
|
+
}
|
|
269
|
+
findings, err := filterFindings(criterion.Findings, validSeqs)
|
|
270
|
+
if err != nil {
|
|
271
|
+
return nil, err
|
|
272
|
+
}
|
|
273
|
+
scores[criterion.ID] = score{coverage: coverage, findings: findings}
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
scored := &model.Rubric{
|
|
277
|
+
Status: rubric.Status,
|
|
278
|
+
Source: rubric.Source,
|
|
279
|
+
TaskDigest: rubric.TaskDigest,
|
|
280
|
+
Note: truncateRunes(strings.TrimSpace(out.RubricNote), maxRubricTextRunes),
|
|
281
|
+
Tasks: make([]model.RubricTask, len(rubric.Tasks)),
|
|
282
|
+
}
|
|
283
|
+
for i, task := range rubric.Tasks {
|
|
284
|
+
copied := task
|
|
285
|
+
copied.Criteria = make([]model.RubricCriterion, len(task.Criteria))
|
|
286
|
+
for j, criterion := range task.Criteria {
|
|
287
|
+
result, ok := scores[criterion.ID]
|
|
288
|
+
if !ok {
|
|
289
|
+
return nil, fmt.Errorf("judge output misses rubric criterion %q", criterion.ID)
|
|
290
|
+
}
|
|
291
|
+
criterion.Coverage = result.coverage
|
|
292
|
+
criterion.Findings = result.findings
|
|
293
|
+
criterion.Verdict = rollupCriterion(result.coverage, result.findings)
|
|
294
|
+
copied.Criteria[j] = criterion
|
|
295
|
+
}
|
|
296
|
+
scored.Tasks[i] = copied
|
|
297
|
+
}
|
|
298
|
+
return scored, nil
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
// rollupVerdict derives the dimension verdict from finding severities; the
|
|
302
|
+
// judge never decides verdicts. Blind spots recorded by the deterministic
|
|
303
|
+
// layer force insufficient-data regardless of what the judge observed.
|
|
304
|
+
func rollupVerdict(name string, findings []model.ReportFinding, obs model.Observability) string {
|
|
305
|
+
if obs.Reads == model.ObservabilityUnavailable && (name == model.DimensionExploration || name == model.DimensionWandering) {
|
|
306
|
+
return model.VerdictInsufficientData
|
|
307
|
+
}
|
|
308
|
+
if obs.Errors == model.ObservabilityUnavailable && name == model.DimensionVerification {
|
|
309
|
+
return model.VerdictInsufficientData
|
|
310
|
+
}
|
|
311
|
+
return rollupSeverities(findings)
|
|
312
|
+
}
|
|
313
|
+
|
|
314
|
+
// rollupCriterion is the rubric-layer analogue: the coverage grade plays the
|
|
315
|
+
// role observability plays for dimensions — a criterion the log cannot
|
|
316
|
+
// evidence reads as insufficient data, never as a flaw.
|
|
317
|
+
func rollupCriterion(coverage string, findings []model.ReportFinding) string {
|
|
318
|
+
if coverage == model.CoverageNone {
|
|
319
|
+
return model.VerdictInsufficientData
|
|
320
|
+
}
|
|
321
|
+
return rollupSeverities(findings)
|
|
322
|
+
}
|
|
323
|
+
|
|
324
|
+
func rollupSeverities(findings []model.ReportFinding) string {
|
|
325
|
+
verdict := model.VerdictGood
|
|
326
|
+
for _, finding := range findings {
|
|
327
|
+
switch finding.Severity {
|
|
328
|
+
case model.SeverityProblem:
|
|
329
|
+
return model.VerdictProblem
|
|
330
|
+
case model.SeverityWarning:
|
|
331
|
+
verdict = model.VerdictWarning
|
|
332
|
+
}
|
|
333
|
+
}
|
|
334
|
+
return verdict
|
|
335
|
+
}
|
|
336
|
+
|
|
337
|
+
func knownDimension(name string) bool {
|
|
338
|
+
for _, known := range model.DimensionNames {
|
|
339
|
+
if name == known {
|
|
340
|
+
return true
|
|
341
|
+
}
|
|
342
|
+
}
|
|
343
|
+
return false
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
// normalizeSeverity forgives casing and whitespace but nothing else: an
|
|
347
|
+
// unrecognized severity is judge output we cannot trust to aggregate.
|
|
348
|
+
func normalizeSeverity(severity string) (string, error) {
|
|
349
|
+
switch strings.ToLower(strings.TrimSpace(severity)) {
|
|
350
|
+
case model.SeverityInfo:
|
|
351
|
+
return model.SeverityInfo, nil
|
|
352
|
+
case model.SeverityWarning:
|
|
353
|
+
return model.SeverityWarning, nil
|
|
354
|
+
case model.SeverityProblem:
|
|
355
|
+
return model.SeverityProblem, nil
|
|
356
|
+
default:
|
|
357
|
+
return "", fmt.Errorf("judge output: unknown severity %q", severity)
|
|
358
|
+
}
|
|
359
|
+
}
|
|
360
|
+
|
|
361
|
+
// normalizeCoverage applies the same strictness to coverage: an unknown grade
|
|
362
|
+
// could silently flip a criterion between scored and insufficient-data.
|
|
363
|
+
func normalizeCoverage(coverage string) (string, error) {
|
|
364
|
+
switch strings.ToLower(strings.TrimSpace(coverage)) {
|
|
365
|
+
case model.CoverageSufficient:
|
|
366
|
+
return model.CoverageSufficient, nil
|
|
367
|
+
case model.CoveragePartial:
|
|
368
|
+
return model.CoveragePartial, nil
|
|
369
|
+
case model.CoverageNone:
|
|
370
|
+
return model.CoverageNone, nil
|
|
371
|
+
default:
|
|
372
|
+
return "", fmt.Errorf("judge output: unknown coverage %q", coverage)
|
|
373
|
+
}
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
// extractJSON returns the first balanced top-level JSON object in text,
|
|
377
|
+
// tolerating judge CLIs that wrap output in logs or markdown fences.
|
|
378
|
+
func extractJSON(text string) (string, error) {
|
|
379
|
+
start := -1
|
|
380
|
+
depth := 0
|
|
381
|
+
inString := false
|
|
382
|
+
escaped := false
|
|
383
|
+
for i, r := range text {
|
|
384
|
+
if start == -1 {
|
|
385
|
+
if r == '{' {
|
|
386
|
+
start = i
|
|
387
|
+
depth = 1
|
|
388
|
+
}
|
|
389
|
+
continue
|
|
390
|
+
}
|
|
391
|
+
if escaped {
|
|
392
|
+
escaped = false
|
|
393
|
+
continue
|
|
394
|
+
}
|
|
395
|
+
switch r {
|
|
396
|
+
case '\\':
|
|
397
|
+
if inString {
|
|
398
|
+
escaped = true
|
|
399
|
+
}
|
|
400
|
+
case '"':
|
|
401
|
+
inString = !inString
|
|
402
|
+
case '{':
|
|
403
|
+
if !inString {
|
|
404
|
+
depth++
|
|
405
|
+
}
|
|
406
|
+
case '}':
|
|
407
|
+
if !inString {
|
|
408
|
+
depth--
|
|
409
|
+
if depth == 0 {
|
|
410
|
+
return text[start : i+1], nil
|
|
411
|
+
}
|
|
412
|
+
}
|
|
413
|
+
}
|
|
414
|
+
}
|
|
415
|
+
return "", fmt.Errorf("no JSON object in judge output")
|
|
416
|
+
}
|