@hybridlabor-api/bdb-synapse 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (102) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +173 -0
  3. package/bin/synapse +0 -0
  4. package/cmd/rubriceval/main.go +308 -0
  5. package/cmd/synapse/main.go +280 -0
  6. package/cmd/synapse/main_test.go +16 -0
  7. package/go.mod +7 -0
  8. package/go.sum +6 -0
  9. package/internal/adapter/adapter.go +1117 -0
  10. package/internal/adapter/adapter_test.go +518 -0
  11. package/internal/adapter/agy/adapter.go +193 -0
  12. package/internal/adapter/claudecode/adapter.go +415 -0
  13. package/internal/adapter/claudecode/adapter_test.go +260 -0
  14. package/internal/adapter/claudecode/agents.go +387 -0
  15. package/internal/adapter/claudecode/agents_test.go +480 -0
  16. package/internal/adapter/claudecode/summary_inputs_test.go +19 -0
  17. package/internal/adapter/codex/adapter.go +919 -0
  18. package/internal/adapter/codex/adapter_test.go +920 -0
  19. package/internal/adapter/codex/agents.go +401 -0
  20. package/internal/adapter/codex/agents_test.go +610 -0
  21. package/internal/adapter/codex/summary_inputs_test.go +20 -0
  22. package/internal/adapter/pi/adapter.go +483 -0
  23. package/internal/adapter/pi/adapter_test.go +517 -0
  24. package/internal/citymap/builder.go +1124 -0
  25. package/internal/citymap/builder_test.go +818 -0
  26. package/internal/judge/cache.go +180 -0
  27. package/internal/judge/cli.go +240 -0
  28. package/internal/judge/cli_test.go +68 -0
  29. package/internal/judge/fresh_summary_test.go +37 -0
  30. package/internal/judge/input.go +233 -0
  31. package/internal/judge/judge.go +416 -0
  32. package/internal/judge/judge_test.go +288 -0
  33. package/internal/judge/prompt.go +129 -0
  34. package/internal/judge/rubric.go +275 -0
  35. package/internal/judge/rubric_test.go +642 -0
  36. package/internal/model/agent.go +63 -0
  37. package/internal/model/agent_schema_test.go +86 -0
  38. package/internal/model/agent_test.go +64 -0
  39. package/internal/model/model.go +175 -0
  40. package/internal/model/report.go +166 -0
  41. package/internal/model/stats.go +151 -0
  42. package/internal/model/stats_test.go +89 -0
  43. package/internal/model/trace_schema_test.go +67 -0
  44. package/internal/server/analyze.go +286 -0
  45. package/internal/server/analyze_test.go +297 -0
  46. package/internal/server/codex_index_test.go +40 -0
  47. package/internal/server/hardening_test.go +147 -0
  48. package/internal/server/reportindex.go +91 -0
  49. package/internal/server/reportindex_test.go +116 -0
  50. package/internal/server/server.go +1099 -0
  51. package/internal/server/server_test.go +1389 -0
  52. package/internal/server/static/assets/fraunces-latin-ext-standard-italic-CGbN9UgK.woff2 +0 -0
  53. package/internal/server/static/assets/fraunces-latin-ext-standard-normal-CJcjJNj7.woff2 +0 -0
  54. package/internal/server/static/assets/fraunces-latin-standard-italic-lSdLDfvT.woff2 +0 -0
  55. package/internal/server/static/assets/fraunces-latin-standard-normal-DihXLNYH.woff2 +0 -0
  56. package/internal/server/static/assets/fraunces-vietnamese-standard-italic-DxWqP7Ku.woff2 +0 -0
  57. package/internal/server/static/assets/fraunces-vietnamese-standard-normal-Czevyj-6.woff2 +0 -0
  58. package/internal/server/static/assets/index-BNoY_BiB.css +1 -0
  59. package/internal/server/static/assets/index-C_adLrJr.js +3 -0
  60. package/internal/server/static/assets/react-gcHzaSmV.js +10 -0
  61. package/internal/server/static/assets/schibsted-grotesk-latin-ext-wght-normal-hsMS0n0O.woff2 +0 -0
  62. package/internal/server/static/assets/schibsted-grotesk-latin-wght-normal-Bb8VGrTG.woff2 +0 -0
  63. package/internal/server/static/assets/three-DnGjZfD1.js +4012 -0
  64. package/internal/server/static/index.html +26 -0
  65. package/internal/server/tracestore.go +173 -0
  66. package/internal/server/tracestore_test.go +51 -0
  67. package/internal/textutil/truncate.go +30 -0
  68. package/internal/textutil/truncate_test.go +39 -0
  69. package/package.json +35 -0
  70. package/web/e2e/agent-lens.spec.ts +688 -0
  71. package/web/index.html +23 -0
  72. package/web/package-lock.json +1933 -0
  73. package/web/package.json +33 -0
  74. package/web/playwright.config.ts +24 -0
  75. package/web/src/App.tsx +876 -0
  76. package/web/src/api/client.ts +74 -0
  77. package/web/src/main.tsx +12 -0
  78. package/web/src/playback/recorder.ts +160 -0
  79. package/web/src/playback/reducer.ts +91 -0
  80. package/web/src/scene/CityScene.tsx +638 -0
  81. package/web/src/scene/TreeScene.tsx +656 -0
  82. package/web/src/scene/dirLabels.ts +145 -0
  83. package/web/src/scene/sceneUtils.ts +144 -0
  84. package/web/src/scene/textures.ts +60 -0
  85. package/web/src/scene/trail.ts +79 -0
  86. package/web/src/scene/treeLayout.ts +169 -0
  87. package/web/src/state/filters.ts +40 -0
  88. package/web/src/state/store.ts +83 -0
  89. package/web/src/styles.css +2565 -0
  90. package/web/src/types.ts +315 -0
  91. package/web/src/ui/AgentsPanel.tsx +376 -0
  92. package/web/src/ui/Dock.tsx +104 -0
  93. package/web/src/ui/Hud.tsx +335 -0
  94. package/web/src/ui/Inspector.tsx +107 -0
  95. package/web/src/ui/LogoMark.tsx +38 -0
  96. package/web/src/ui/ReportPanel.tsx +491 -0
  97. package/web/src/ui/SessionRail.tsx +316 -0
  98. package/web/src/ui/Timeline.tsx +458 -0
  99. package/web/src/ui/ViewPanel.tsx +45 -0
  100. package/web/src/ui/shortcuts.ts +4 -0
  101. package/web/tsconfig.json +21 -0
  102. package/web/vite.config.ts +28 -0
@@ -0,0 +1,275 @@
1
+ package judge
2
+
3
+ import (
4
+ "context"
5
+ "encoding/json"
6
+ "fmt"
7
+ "regexp"
8
+ "sort"
9
+ "strings"
10
+
11
+ "github.com/hybridlabor-api/bdb-synapse/internal/model"
12
+ )
13
+
14
+ // Rubric shape bounds. The prompt asks for less (4-6 single-task, ≤10 total);
15
+ // validation accepts a margin, and anything beyond it invalidates the output.
16
+ // The byte cap bounds both the injection surface a hostile trace gets to
17
+ // shape and what the panel is asked to render.
18
+ const (
19
+ maxRubricTasks = 6
20
+ maxCriteriaPerTask = 6
21
+ minRubricCriteria = 3
22
+ maxRubricCriteria = 12
23
+ maxCriterionIDLen = 48
24
+ maxRubricTitleRunes = 80
25
+ maxRubricTextRunes = 500
26
+ maxRubricJSONBytes = 12 * 1024
27
+ // weakTaskTextRunes is the floor under which user messages carry too
28
+ // little task signal to derive criteria from ("continue", "ok" sessions).
29
+ // Initial guess — calibrated against historical sessions in M1.5.
30
+ weakTaskTextRunes = 30
31
+ )
32
+
33
+ var criterionIDPattern = regexp.MustCompile(`^[a-z0-9]+(-[a-z0-9]+)*$`)
34
+
35
+ // llmRubric mirrors the JSON shape rubricPrompt requests.
36
+ type llmRubric struct {
37
+ Tasks []struct {
38
+ Title string `json:"title"`
39
+ Type string `json:"type"`
40
+ AnchorUserMessages []int `json:"anchor_user_messages"`
41
+ Criteria []struct {
42
+ ID string `json:"id"`
43
+ Title string `json:"title"`
44
+ Why string `json:"why"`
45
+ Good string `json:"good"`
46
+ Bad string `json:"bad"`
47
+ } `json:"criteria"`
48
+ } `json:"tasks"`
49
+ }
50
+
51
+ // acquireRubric resolves the rubric layer for one run: a deterministic skip,
52
+ // a reuse of the cached report's rubric, or up to two generation attempts.
53
+ // Generation failure degrades (status unavailable) rather than erroring —
54
+ // the fixed dimensions must never be blocked by the rubric layer. Only a
55
+ // subprocess failure is a hard error, since scoring would hit it too.
56
+ func acquireRubric(ctx context.Context, runner Runner, trace *model.Trace, cached *model.Report) (*model.Rubric, error) {
57
+ // Conversation-only sessions (no tool events) leave nothing to cite:
58
+ // scoring would drop every finding and hand out good verdicts on zero
59
+ // evidence — the M1.5 bench caught exactly that.
60
+ if len(trace.Events) == 0 {
61
+ return &model.Rubric{Status: model.RubricStatusUnavailable, Reason: model.RubricReasonNoEvents}, nil
62
+ }
63
+ // One task-evidence contract: the generator's input, the anchor
64
+ // validation set, the digest, and the weak-text gate all read
65
+ // taskMessages — never a differently budgeted list.
66
+ messages := taskMessages(trace.Marks)
67
+ if len(messages) == 0 {
68
+ return &model.Rubric{Status: model.RubricStatusUnavailable, Reason: model.RubricReasonNoTaskText}, nil
69
+ }
70
+ if taskTextRunes(trace.Marks) < weakTaskTextRunes {
71
+ return &model.Rubric{Status: model.RubricStatusUnavailable, Reason: model.RubricReasonWeakTaskText}, nil
72
+ }
73
+ digest := TaskDigest(trace, model.RubricSourceFull)
74
+ if reused := reusableRubric(cached, digest); reused != nil {
75
+ return reused, nil
76
+ }
77
+ input := BuildRubricInput(trace)
78
+ for attempt := 0; attempt < 2; attempt++ {
79
+ result, err := runner.Run(ctx, rubricPrompt, input)
80
+ if err != nil {
81
+ return nil, err
82
+ }
83
+ tasks, err := parseRubric(result.Text, messages)
84
+ if err != nil {
85
+ continue
86
+ }
87
+ return &model.Rubric{
88
+ Status: model.RubricStatusScored,
89
+ Source: model.RubricSourceFull,
90
+ TaskDigest: digest,
91
+ Tasks: tasks,
92
+ }, nil
93
+ }
94
+ return &model.Rubric{Status: model.RubricStatusUnavailable, Reason: model.RubricReasonGenerationFailed}, nil
95
+ }
96
+
97
+ // reusableRubric lifts the cached report's rubric when the task wording and
98
+ // rubric prompt are unchanged: criteria stay stable across re-evaluations,
99
+ // and the run saves the generation call. Scores are stripped — the new
100
+ // scoring pass owns them.
101
+ func reusableRubric(cached *model.Report, digest string) *model.Rubric {
102
+ if cached == nil || cached.Rubric == nil ||
103
+ cached.Rubric.Status != model.RubricStatusScored ||
104
+ cached.Rubric.TaskDigest != digest ||
105
+ cached.Judge.RubricPromptVersion != RubricPromptVersion {
106
+ return nil
107
+ }
108
+ tasks := make([]model.RubricTask, len(cached.Rubric.Tasks))
109
+ for i, task := range cached.Rubric.Tasks {
110
+ copied := task
111
+ copied.AnchorUserMessages = append([]int(nil), task.AnchorUserMessages...)
112
+ copied.AnchorSeqs = append([]int(nil), task.AnchorSeqs...)
113
+ copied.Criteria = make([]model.RubricCriterion, len(task.Criteria))
114
+ for j, criterion := range task.Criteria {
115
+ copied.Criteria[j] = model.RubricCriterion{
116
+ ID: criterion.ID,
117
+ Title: criterion.Title,
118
+ Why: criterion.Why,
119
+ Good: criterion.Good,
120
+ Bad: criterion.Bad,
121
+ }
122
+ }
123
+ tasks[i] = copied
124
+ }
125
+ return &model.Rubric{
126
+ Status: model.RubricStatusScored,
127
+ Source: cached.Rubric.Source,
128
+ TaskDigest: digest,
129
+ Tasks: tasks,
130
+ }
131
+ }
132
+
133
+ // parseRubric validates one generation attempt against the shape bounds and
134
+ // the session's real user messages, and derives anchor seqs. Any violation
135
+ // invalidates the whole output — a rubric is trusted downstream (it goes
136
+ // into the scoring prompt and the report), so nothing malformed may pass.
137
+ func parseRubric(raw string, messages []userMessage) ([]model.RubricTask, error) {
138
+ payload, err := extractJSON(raw)
139
+ if err != nil {
140
+ return nil, err
141
+ }
142
+ if len(payload) > maxRubricJSONBytes {
143
+ return nil, fmt.Errorf("rubric: %d bytes exceeds the %d cap", len(payload), maxRubricJSONBytes)
144
+ }
145
+ var out llmRubric
146
+ if err := json.Unmarshal([]byte(payload), &out); err != nil {
147
+ return nil, fmt.Errorf("rubric JSON: %w", err)
148
+ }
149
+ if len(out.Tasks) == 0 || len(out.Tasks) > maxRubricTasks {
150
+ return nil, fmt.Errorf("rubric: %d tasks, want 1-%d", len(out.Tasks), maxRubricTasks)
151
+ }
152
+
153
+ seqByOrdinal := make(map[int]int, len(messages))
154
+ for _, message := range messages {
155
+ seqByOrdinal[message.ordinal] = message.seq
156
+ }
157
+ seenOrdinals := map[int]bool{}
158
+ seenIDs := map[string]bool{}
159
+ totalCriteria := 0
160
+ tasks := make([]model.RubricTask, 0, len(out.Tasks))
161
+ for _, task := range out.Tasks {
162
+ title := strings.TrimSpace(task.Title)
163
+ if title == "" || len([]rune(title)) > maxRubricTitleRunes {
164
+ return nil, fmt.Errorf("rubric: bad task title %q", task.Title)
165
+ }
166
+ if len(task.AnchorUserMessages) == 0 {
167
+ return nil, fmt.Errorf("rubric: task %q has no anchor user messages", title)
168
+ }
169
+ anchors := append([]int(nil), task.AnchorUserMessages...)
170
+ sort.Ints(anchors)
171
+ seqs := make([]int, 0, len(anchors))
172
+ for _, ordinal := range anchors {
173
+ seq, ok := seqByOrdinal[ordinal]
174
+ if !ok {
175
+ return nil, fmt.Errorf("rubric: task %q anchors unknown user message #%d", title, ordinal)
176
+ }
177
+ if seenOrdinals[ordinal] {
178
+ return nil, fmt.Errorf("rubric: user message #%d anchored by more than one task", ordinal)
179
+ }
180
+ seenOrdinals[ordinal] = true
181
+ if len(seqs) == 0 || seqs[len(seqs)-1] != seq {
182
+ seqs = append(seqs, seq)
183
+ }
184
+ }
185
+ if len(task.Criteria) == 0 || len(task.Criteria) > maxCriteriaPerTask {
186
+ return nil, fmt.Errorf("rubric: task %q has %d criteria, want 1-%d", title, len(task.Criteria), maxCriteriaPerTask)
187
+ }
188
+ criteria := make([]model.RubricCriterion, 0, len(task.Criteria))
189
+ for _, criterion := range task.Criteria {
190
+ id := strings.TrimSpace(criterion.ID)
191
+ if len(id) > maxCriterionIDLen || !criterionIDPattern.MatchString(id) {
192
+ return nil, fmt.Errorf("rubric: bad criterion id %q", criterion.ID)
193
+ }
194
+ if seenIDs[id] {
195
+ return nil, fmt.Errorf("rubric: duplicate criterion id %q", id)
196
+ }
197
+ seenIDs[id] = true
198
+ ctitle := strings.TrimSpace(criterion.Title)
199
+ if ctitle == "" || len([]rune(ctitle)) > maxRubricTitleRunes {
200
+ return nil, fmt.Errorf("rubric: bad title for criterion %q", id)
201
+ }
202
+ for _, text := range []string{criterion.Why, criterion.Good, criterion.Bad} {
203
+ if len([]rune(text)) > maxRubricTextRunes {
204
+ return nil, fmt.Errorf("rubric: overlong text on criterion %q", id)
205
+ }
206
+ }
207
+ criteria = append(criteria, model.RubricCriterion{
208
+ ID: id,
209
+ Title: ctitle,
210
+ Why: strings.TrimSpace(criterion.Why),
211
+ Good: strings.TrimSpace(criterion.Good),
212
+ Bad: strings.TrimSpace(criterion.Bad),
213
+ })
214
+ }
215
+ totalCriteria += len(criteria)
216
+ tasks = append(tasks, model.RubricTask{
217
+ Title: title,
218
+ Type: strings.TrimSpace(strings.ToLower(task.Type)),
219
+ AnchorUserMessages: anchors,
220
+ AnchorSeqs: seqs,
221
+ Criteria: criteria,
222
+ })
223
+ }
224
+ if totalCriteria < minRubricCriteria || totalCriteria > maxRubricCriteria {
225
+ return nil, fmt.Errorf("rubric: %d criteria total, want %d-%d", totalCriteria, minRubricCriteria, maxRubricCriteria)
226
+ }
227
+ return tasks, nil
228
+ }
229
+
230
+ // scoringRubricJSON renders the rubric as the data section of the scoring
231
+ // input: criteria definitions only — anchors and digests are bookkeeping the
232
+ // scorer has no use for.
233
+ func scoringRubricJSON(rubric *model.Rubric) string {
234
+ type criterion struct {
235
+ ID string `json:"id"`
236
+ Title string `json:"title"`
237
+ Why string `json:"why,omitempty"`
238
+ Good string `json:"good,omitempty"`
239
+ Bad string `json:"bad,omitempty"`
240
+ }
241
+ type task struct {
242
+ Title string `json:"title"`
243
+ Type string `json:"type,omitempty"`
244
+ Criteria []criterion `json:"criteria"`
245
+ }
246
+ tasks := make([]task, 0, len(rubric.Tasks))
247
+ for _, t := range rubric.Tasks {
248
+ out := task{Title: t.Title, Type: t.Type}
249
+ for _, c := range t.Criteria {
250
+ out.Criteria = append(out.Criteria, criterion{ID: c.ID, Title: c.Title, Why: c.Why, Good: c.Good, Bad: c.Bad})
251
+ }
252
+ tasks = append(tasks, out)
253
+ }
254
+ encoded, err := json.Marshal(map[string]any{"tasks": tasks})
255
+ if err != nil {
256
+ return `{"tasks":[]}`
257
+ }
258
+ return string(encoded)
259
+ }
260
+
261
+ // RubricSatisfied reports whether a cached report already answers a
262
+ // rubric-enabled request: it carries a scored rubric, or records a
263
+ // deterministic skip a re-run would only repeat. generation-failed is
264
+ // transient and worth a fresh run.
265
+ func RubricSatisfied(report *model.Report) bool {
266
+ if report == nil || report.Rubric == nil {
267
+ return false
268
+ }
269
+ if report.Rubric.Status == model.RubricStatusScored {
270
+ return true
271
+ }
272
+ return report.Rubric.Reason == model.RubricReasonNoTaskText ||
273
+ report.Rubric.Reason == model.RubricReasonWeakTaskText ||
274
+ report.Rubric.Reason == model.RubricReasonNoEvents
275
+ }