@hybridlabor-api/bdb-synapse 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +173 -0
- package/bin/synapse +0 -0
- package/cmd/rubriceval/main.go +308 -0
- package/cmd/synapse/main.go +280 -0
- package/cmd/synapse/main_test.go +16 -0
- package/go.mod +7 -0
- package/go.sum +6 -0
- package/internal/adapter/adapter.go +1117 -0
- package/internal/adapter/adapter_test.go +518 -0
- package/internal/adapter/agy/adapter.go +193 -0
- package/internal/adapter/claudecode/adapter.go +415 -0
- package/internal/adapter/claudecode/adapter_test.go +260 -0
- package/internal/adapter/claudecode/agents.go +387 -0
- package/internal/adapter/claudecode/agents_test.go +480 -0
- package/internal/adapter/claudecode/summary_inputs_test.go +19 -0
- package/internal/adapter/codex/adapter.go +919 -0
- package/internal/adapter/codex/adapter_test.go +920 -0
- package/internal/adapter/codex/agents.go +401 -0
- package/internal/adapter/codex/agents_test.go +610 -0
- package/internal/adapter/codex/summary_inputs_test.go +20 -0
- package/internal/adapter/pi/adapter.go +483 -0
- package/internal/adapter/pi/adapter_test.go +517 -0
- package/internal/citymap/builder.go +1124 -0
- package/internal/citymap/builder_test.go +818 -0
- package/internal/judge/cache.go +180 -0
- package/internal/judge/cli.go +240 -0
- package/internal/judge/cli_test.go +68 -0
- package/internal/judge/fresh_summary_test.go +37 -0
- package/internal/judge/input.go +233 -0
- package/internal/judge/judge.go +416 -0
- package/internal/judge/judge_test.go +288 -0
- package/internal/judge/prompt.go +129 -0
- package/internal/judge/rubric.go +275 -0
- package/internal/judge/rubric_test.go +642 -0
- package/internal/model/agent.go +63 -0
- package/internal/model/agent_schema_test.go +86 -0
- package/internal/model/agent_test.go +64 -0
- package/internal/model/model.go +175 -0
- package/internal/model/report.go +166 -0
- package/internal/model/stats.go +151 -0
- package/internal/model/stats_test.go +89 -0
- package/internal/model/trace_schema_test.go +67 -0
- package/internal/server/analyze.go +286 -0
- package/internal/server/analyze_test.go +297 -0
- package/internal/server/codex_index_test.go +40 -0
- package/internal/server/hardening_test.go +147 -0
- package/internal/server/reportindex.go +91 -0
- package/internal/server/reportindex_test.go +116 -0
- package/internal/server/server.go +1099 -0
- package/internal/server/server_test.go +1389 -0
- package/internal/server/static/assets/fraunces-latin-ext-standard-italic-CGbN9UgK.woff2 +0 -0
- package/internal/server/static/assets/fraunces-latin-ext-standard-normal-CJcjJNj7.woff2 +0 -0
- package/internal/server/static/assets/fraunces-latin-standard-italic-lSdLDfvT.woff2 +0 -0
- package/internal/server/static/assets/fraunces-latin-standard-normal-DihXLNYH.woff2 +0 -0
- package/internal/server/static/assets/fraunces-vietnamese-standard-italic-DxWqP7Ku.woff2 +0 -0
- package/internal/server/static/assets/fraunces-vietnamese-standard-normal-Czevyj-6.woff2 +0 -0
- package/internal/server/static/assets/index-BNoY_BiB.css +1 -0
- package/internal/server/static/assets/index-C_adLrJr.js +3 -0
- package/internal/server/static/assets/react-gcHzaSmV.js +10 -0
- package/internal/server/static/assets/schibsted-grotesk-latin-ext-wght-normal-hsMS0n0O.woff2 +0 -0
- package/internal/server/static/assets/schibsted-grotesk-latin-wght-normal-Bb8VGrTG.woff2 +0 -0
- package/internal/server/static/assets/three-DnGjZfD1.js +4012 -0
- package/internal/server/static/index.html +26 -0
- package/internal/server/tracestore.go +173 -0
- package/internal/server/tracestore_test.go +51 -0
- package/internal/textutil/truncate.go +30 -0
- package/internal/textutil/truncate_test.go +39 -0
- package/package.json +35 -0
- package/web/e2e/agent-lens.spec.ts +688 -0
- package/web/index.html +23 -0
- package/web/package-lock.json +1933 -0
- package/web/package.json +33 -0
- package/web/playwright.config.ts +24 -0
- package/web/src/App.tsx +876 -0
- package/web/src/api/client.ts +74 -0
- package/web/src/main.tsx +12 -0
- package/web/src/playback/recorder.ts +160 -0
- package/web/src/playback/reducer.ts +91 -0
- package/web/src/scene/CityScene.tsx +638 -0
- package/web/src/scene/TreeScene.tsx +656 -0
- package/web/src/scene/dirLabels.ts +145 -0
- package/web/src/scene/sceneUtils.ts +144 -0
- package/web/src/scene/textures.ts +60 -0
- package/web/src/scene/trail.ts +79 -0
- package/web/src/scene/treeLayout.ts +169 -0
- package/web/src/state/filters.ts +40 -0
- package/web/src/state/store.ts +83 -0
- package/web/src/styles.css +2565 -0
- package/web/src/types.ts +315 -0
- package/web/src/ui/AgentsPanel.tsx +376 -0
- package/web/src/ui/Dock.tsx +104 -0
- package/web/src/ui/Hud.tsx +335 -0
- package/web/src/ui/Inspector.tsx +107 -0
- package/web/src/ui/LogoMark.tsx +38 -0
- package/web/src/ui/ReportPanel.tsx +491 -0
- package/web/src/ui/SessionRail.tsx +316 -0
- package/web/src/ui/Timeline.tsx +458 -0
- package/web/src/ui/ViewPanel.tsx +45 -0
- package/web/src/ui/shortcuts.ts +4 -0
- package/web/tsconfig.json +21 -0
- package/web/vite.config.ts +28 -0
|
@@ -0,0 +1,275 @@
|
|
|
1
|
+
package judge
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"context"
|
|
5
|
+
"encoding/json"
|
|
6
|
+
"fmt"
|
|
7
|
+
"regexp"
|
|
8
|
+
"sort"
|
|
9
|
+
"strings"
|
|
10
|
+
|
|
11
|
+
"github.com/hybridlabor-api/bdb-synapse/internal/model"
|
|
12
|
+
)
|
|
13
|
+
|
|
14
|
+
// Rubric shape bounds. The prompt asks for less (4-6 single-task, ≤10 total);
|
|
15
|
+
// validation accepts a margin, and anything beyond it invalidates the output.
|
|
16
|
+
// The byte cap bounds both the injection surface a hostile trace gets to
|
|
17
|
+
// shape and what the panel is asked to render.
|
|
18
|
+
const (
|
|
19
|
+
maxRubricTasks = 6
|
|
20
|
+
maxCriteriaPerTask = 6
|
|
21
|
+
minRubricCriteria = 3
|
|
22
|
+
maxRubricCriteria = 12
|
|
23
|
+
maxCriterionIDLen = 48
|
|
24
|
+
maxRubricTitleRunes = 80
|
|
25
|
+
maxRubricTextRunes = 500
|
|
26
|
+
maxRubricJSONBytes = 12 * 1024
|
|
27
|
+
// weakTaskTextRunes is the floor under which user messages carry too
|
|
28
|
+
// little task signal to derive criteria from ("continue", "ok" sessions).
|
|
29
|
+
// Initial guess — calibrated against historical sessions in M1.5.
|
|
30
|
+
weakTaskTextRunes = 30
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
var criterionIDPattern = regexp.MustCompile(`^[a-z0-9]+(-[a-z0-9]+)*$`)
|
|
34
|
+
|
|
35
|
+
// llmRubric mirrors the JSON shape rubricPrompt requests.
|
|
36
|
+
type llmRubric struct {
|
|
37
|
+
Tasks []struct {
|
|
38
|
+
Title string `json:"title"`
|
|
39
|
+
Type string `json:"type"`
|
|
40
|
+
AnchorUserMessages []int `json:"anchor_user_messages"`
|
|
41
|
+
Criteria []struct {
|
|
42
|
+
ID string `json:"id"`
|
|
43
|
+
Title string `json:"title"`
|
|
44
|
+
Why string `json:"why"`
|
|
45
|
+
Good string `json:"good"`
|
|
46
|
+
Bad string `json:"bad"`
|
|
47
|
+
} `json:"criteria"`
|
|
48
|
+
} `json:"tasks"`
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
// acquireRubric resolves the rubric layer for one run: a deterministic skip,
|
|
52
|
+
// a reuse of the cached report's rubric, or up to two generation attempts.
|
|
53
|
+
// Generation failure degrades (status unavailable) rather than erroring —
|
|
54
|
+
// the fixed dimensions must never be blocked by the rubric layer. Only a
|
|
55
|
+
// subprocess failure is a hard error, since scoring would hit it too.
|
|
56
|
+
func acquireRubric(ctx context.Context, runner Runner, trace *model.Trace, cached *model.Report) (*model.Rubric, error) {
|
|
57
|
+
// Conversation-only sessions (no tool events) leave nothing to cite:
|
|
58
|
+
// scoring would drop every finding and hand out good verdicts on zero
|
|
59
|
+
// evidence — the M1.5 bench caught exactly that.
|
|
60
|
+
if len(trace.Events) == 0 {
|
|
61
|
+
return &model.Rubric{Status: model.RubricStatusUnavailable, Reason: model.RubricReasonNoEvents}, nil
|
|
62
|
+
}
|
|
63
|
+
// One task-evidence contract: the generator's input, the anchor
|
|
64
|
+
// validation set, the digest, and the weak-text gate all read
|
|
65
|
+
// taskMessages — never a differently budgeted list.
|
|
66
|
+
messages := taskMessages(trace.Marks)
|
|
67
|
+
if len(messages) == 0 {
|
|
68
|
+
return &model.Rubric{Status: model.RubricStatusUnavailable, Reason: model.RubricReasonNoTaskText}, nil
|
|
69
|
+
}
|
|
70
|
+
if taskTextRunes(trace.Marks) < weakTaskTextRunes {
|
|
71
|
+
return &model.Rubric{Status: model.RubricStatusUnavailable, Reason: model.RubricReasonWeakTaskText}, nil
|
|
72
|
+
}
|
|
73
|
+
digest := TaskDigest(trace, model.RubricSourceFull)
|
|
74
|
+
if reused := reusableRubric(cached, digest); reused != nil {
|
|
75
|
+
return reused, nil
|
|
76
|
+
}
|
|
77
|
+
input := BuildRubricInput(trace)
|
|
78
|
+
for attempt := 0; attempt < 2; attempt++ {
|
|
79
|
+
result, err := runner.Run(ctx, rubricPrompt, input)
|
|
80
|
+
if err != nil {
|
|
81
|
+
return nil, err
|
|
82
|
+
}
|
|
83
|
+
tasks, err := parseRubric(result.Text, messages)
|
|
84
|
+
if err != nil {
|
|
85
|
+
continue
|
|
86
|
+
}
|
|
87
|
+
return &model.Rubric{
|
|
88
|
+
Status: model.RubricStatusScored,
|
|
89
|
+
Source: model.RubricSourceFull,
|
|
90
|
+
TaskDigest: digest,
|
|
91
|
+
Tasks: tasks,
|
|
92
|
+
}, nil
|
|
93
|
+
}
|
|
94
|
+
return &model.Rubric{Status: model.RubricStatusUnavailable, Reason: model.RubricReasonGenerationFailed}, nil
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
// reusableRubric lifts the cached report's rubric when the task wording and
|
|
98
|
+
// rubric prompt are unchanged: criteria stay stable across re-evaluations,
|
|
99
|
+
// and the run saves the generation call. Scores are stripped — the new
|
|
100
|
+
// scoring pass owns them.
|
|
101
|
+
func reusableRubric(cached *model.Report, digest string) *model.Rubric {
|
|
102
|
+
if cached == nil || cached.Rubric == nil ||
|
|
103
|
+
cached.Rubric.Status != model.RubricStatusScored ||
|
|
104
|
+
cached.Rubric.TaskDigest != digest ||
|
|
105
|
+
cached.Judge.RubricPromptVersion != RubricPromptVersion {
|
|
106
|
+
return nil
|
|
107
|
+
}
|
|
108
|
+
tasks := make([]model.RubricTask, len(cached.Rubric.Tasks))
|
|
109
|
+
for i, task := range cached.Rubric.Tasks {
|
|
110
|
+
copied := task
|
|
111
|
+
copied.AnchorUserMessages = append([]int(nil), task.AnchorUserMessages...)
|
|
112
|
+
copied.AnchorSeqs = append([]int(nil), task.AnchorSeqs...)
|
|
113
|
+
copied.Criteria = make([]model.RubricCriterion, len(task.Criteria))
|
|
114
|
+
for j, criterion := range task.Criteria {
|
|
115
|
+
copied.Criteria[j] = model.RubricCriterion{
|
|
116
|
+
ID: criterion.ID,
|
|
117
|
+
Title: criterion.Title,
|
|
118
|
+
Why: criterion.Why,
|
|
119
|
+
Good: criterion.Good,
|
|
120
|
+
Bad: criterion.Bad,
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
tasks[i] = copied
|
|
124
|
+
}
|
|
125
|
+
return &model.Rubric{
|
|
126
|
+
Status: model.RubricStatusScored,
|
|
127
|
+
Source: cached.Rubric.Source,
|
|
128
|
+
TaskDigest: digest,
|
|
129
|
+
Tasks: tasks,
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
// parseRubric validates one generation attempt against the shape bounds and
|
|
134
|
+
// the session's real user messages, and derives anchor seqs. Any violation
|
|
135
|
+
// invalidates the whole output — a rubric is trusted downstream (it goes
|
|
136
|
+
// into the scoring prompt and the report), so nothing malformed may pass.
|
|
137
|
+
func parseRubric(raw string, messages []userMessage) ([]model.RubricTask, error) {
|
|
138
|
+
payload, err := extractJSON(raw)
|
|
139
|
+
if err != nil {
|
|
140
|
+
return nil, err
|
|
141
|
+
}
|
|
142
|
+
if len(payload) > maxRubricJSONBytes {
|
|
143
|
+
return nil, fmt.Errorf("rubric: %d bytes exceeds the %d cap", len(payload), maxRubricJSONBytes)
|
|
144
|
+
}
|
|
145
|
+
var out llmRubric
|
|
146
|
+
if err := json.Unmarshal([]byte(payload), &out); err != nil {
|
|
147
|
+
return nil, fmt.Errorf("rubric JSON: %w", err)
|
|
148
|
+
}
|
|
149
|
+
if len(out.Tasks) == 0 || len(out.Tasks) > maxRubricTasks {
|
|
150
|
+
return nil, fmt.Errorf("rubric: %d tasks, want 1-%d", len(out.Tasks), maxRubricTasks)
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
seqByOrdinal := make(map[int]int, len(messages))
|
|
154
|
+
for _, message := range messages {
|
|
155
|
+
seqByOrdinal[message.ordinal] = message.seq
|
|
156
|
+
}
|
|
157
|
+
seenOrdinals := map[int]bool{}
|
|
158
|
+
seenIDs := map[string]bool{}
|
|
159
|
+
totalCriteria := 0
|
|
160
|
+
tasks := make([]model.RubricTask, 0, len(out.Tasks))
|
|
161
|
+
for _, task := range out.Tasks {
|
|
162
|
+
title := strings.TrimSpace(task.Title)
|
|
163
|
+
if title == "" || len([]rune(title)) > maxRubricTitleRunes {
|
|
164
|
+
return nil, fmt.Errorf("rubric: bad task title %q", task.Title)
|
|
165
|
+
}
|
|
166
|
+
if len(task.AnchorUserMessages) == 0 {
|
|
167
|
+
return nil, fmt.Errorf("rubric: task %q has no anchor user messages", title)
|
|
168
|
+
}
|
|
169
|
+
anchors := append([]int(nil), task.AnchorUserMessages...)
|
|
170
|
+
sort.Ints(anchors)
|
|
171
|
+
seqs := make([]int, 0, len(anchors))
|
|
172
|
+
for _, ordinal := range anchors {
|
|
173
|
+
seq, ok := seqByOrdinal[ordinal]
|
|
174
|
+
if !ok {
|
|
175
|
+
return nil, fmt.Errorf("rubric: task %q anchors unknown user message #%d", title, ordinal)
|
|
176
|
+
}
|
|
177
|
+
if seenOrdinals[ordinal] {
|
|
178
|
+
return nil, fmt.Errorf("rubric: user message #%d anchored by more than one task", ordinal)
|
|
179
|
+
}
|
|
180
|
+
seenOrdinals[ordinal] = true
|
|
181
|
+
if len(seqs) == 0 || seqs[len(seqs)-1] != seq {
|
|
182
|
+
seqs = append(seqs, seq)
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
if len(task.Criteria) == 0 || len(task.Criteria) > maxCriteriaPerTask {
|
|
186
|
+
return nil, fmt.Errorf("rubric: task %q has %d criteria, want 1-%d", title, len(task.Criteria), maxCriteriaPerTask)
|
|
187
|
+
}
|
|
188
|
+
criteria := make([]model.RubricCriterion, 0, len(task.Criteria))
|
|
189
|
+
for _, criterion := range task.Criteria {
|
|
190
|
+
id := strings.TrimSpace(criterion.ID)
|
|
191
|
+
if len(id) > maxCriterionIDLen || !criterionIDPattern.MatchString(id) {
|
|
192
|
+
return nil, fmt.Errorf("rubric: bad criterion id %q", criterion.ID)
|
|
193
|
+
}
|
|
194
|
+
if seenIDs[id] {
|
|
195
|
+
return nil, fmt.Errorf("rubric: duplicate criterion id %q", id)
|
|
196
|
+
}
|
|
197
|
+
seenIDs[id] = true
|
|
198
|
+
ctitle := strings.TrimSpace(criterion.Title)
|
|
199
|
+
if ctitle == "" || len([]rune(ctitle)) > maxRubricTitleRunes {
|
|
200
|
+
return nil, fmt.Errorf("rubric: bad title for criterion %q", id)
|
|
201
|
+
}
|
|
202
|
+
for _, text := range []string{criterion.Why, criterion.Good, criterion.Bad} {
|
|
203
|
+
if len([]rune(text)) > maxRubricTextRunes {
|
|
204
|
+
return nil, fmt.Errorf("rubric: overlong text on criterion %q", id)
|
|
205
|
+
}
|
|
206
|
+
}
|
|
207
|
+
criteria = append(criteria, model.RubricCriterion{
|
|
208
|
+
ID: id,
|
|
209
|
+
Title: ctitle,
|
|
210
|
+
Why: strings.TrimSpace(criterion.Why),
|
|
211
|
+
Good: strings.TrimSpace(criterion.Good),
|
|
212
|
+
Bad: strings.TrimSpace(criterion.Bad),
|
|
213
|
+
})
|
|
214
|
+
}
|
|
215
|
+
totalCriteria += len(criteria)
|
|
216
|
+
tasks = append(tasks, model.RubricTask{
|
|
217
|
+
Title: title,
|
|
218
|
+
Type: strings.TrimSpace(strings.ToLower(task.Type)),
|
|
219
|
+
AnchorUserMessages: anchors,
|
|
220
|
+
AnchorSeqs: seqs,
|
|
221
|
+
Criteria: criteria,
|
|
222
|
+
})
|
|
223
|
+
}
|
|
224
|
+
if totalCriteria < minRubricCriteria || totalCriteria > maxRubricCriteria {
|
|
225
|
+
return nil, fmt.Errorf("rubric: %d criteria total, want %d-%d", totalCriteria, minRubricCriteria, maxRubricCriteria)
|
|
226
|
+
}
|
|
227
|
+
return tasks, nil
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
// scoringRubricJSON renders the rubric as the data section of the scoring
|
|
231
|
+
// input: criteria definitions only — anchors and digests are bookkeeping the
|
|
232
|
+
// scorer has no use for.
|
|
233
|
+
func scoringRubricJSON(rubric *model.Rubric) string {
|
|
234
|
+
type criterion struct {
|
|
235
|
+
ID string `json:"id"`
|
|
236
|
+
Title string `json:"title"`
|
|
237
|
+
Why string `json:"why,omitempty"`
|
|
238
|
+
Good string `json:"good,omitempty"`
|
|
239
|
+
Bad string `json:"bad,omitempty"`
|
|
240
|
+
}
|
|
241
|
+
type task struct {
|
|
242
|
+
Title string `json:"title"`
|
|
243
|
+
Type string `json:"type,omitempty"`
|
|
244
|
+
Criteria []criterion `json:"criteria"`
|
|
245
|
+
}
|
|
246
|
+
tasks := make([]task, 0, len(rubric.Tasks))
|
|
247
|
+
for _, t := range rubric.Tasks {
|
|
248
|
+
out := task{Title: t.Title, Type: t.Type}
|
|
249
|
+
for _, c := range t.Criteria {
|
|
250
|
+
out.Criteria = append(out.Criteria, criterion{ID: c.ID, Title: c.Title, Why: c.Why, Good: c.Good, Bad: c.Bad})
|
|
251
|
+
}
|
|
252
|
+
tasks = append(tasks, out)
|
|
253
|
+
}
|
|
254
|
+
encoded, err := json.Marshal(map[string]any{"tasks": tasks})
|
|
255
|
+
if err != nil {
|
|
256
|
+
return `{"tasks":[]}`
|
|
257
|
+
}
|
|
258
|
+
return string(encoded)
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
// RubricSatisfied reports whether a cached report already answers a
|
|
262
|
+
// rubric-enabled request: it carries a scored rubric, or records a
|
|
263
|
+
// deterministic skip a re-run would only repeat. generation-failed is
|
|
264
|
+
// transient and worth a fresh run.
|
|
265
|
+
func RubricSatisfied(report *model.Report) bool {
|
|
266
|
+
if report == nil || report.Rubric == nil {
|
|
267
|
+
return false
|
|
268
|
+
}
|
|
269
|
+
if report.Rubric.Status == model.RubricStatusScored {
|
|
270
|
+
return true
|
|
271
|
+
}
|
|
272
|
+
return report.Rubric.Reason == model.RubricReasonNoTaskText ||
|
|
273
|
+
report.Rubric.Reason == model.RubricReasonWeakTaskText ||
|
|
274
|
+
report.Rubric.Reason == model.RubricReasonNoEvents
|
|
275
|
+
}
|