@hybridlabor-api/bdb-synapse 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (102) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +173 -0
  3. package/bin/synapse +0 -0
  4. package/cmd/rubriceval/main.go +308 -0
  5. package/cmd/synapse/main.go +280 -0
  6. package/cmd/synapse/main_test.go +16 -0
  7. package/go.mod +7 -0
  8. package/go.sum +6 -0
  9. package/internal/adapter/adapter.go +1117 -0
  10. package/internal/adapter/adapter_test.go +518 -0
  11. package/internal/adapter/agy/adapter.go +193 -0
  12. package/internal/adapter/claudecode/adapter.go +415 -0
  13. package/internal/adapter/claudecode/adapter_test.go +260 -0
  14. package/internal/adapter/claudecode/agents.go +387 -0
  15. package/internal/adapter/claudecode/agents_test.go +480 -0
  16. package/internal/adapter/claudecode/summary_inputs_test.go +19 -0
  17. package/internal/adapter/codex/adapter.go +919 -0
  18. package/internal/adapter/codex/adapter_test.go +920 -0
  19. package/internal/adapter/codex/agents.go +401 -0
  20. package/internal/adapter/codex/agents_test.go +610 -0
  21. package/internal/adapter/codex/summary_inputs_test.go +20 -0
  22. package/internal/adapter/pi/adapter.go +483 -0
  23. package/internal/adapter/pi/adapter_test.go +517 -0
  24. package/internal/citymap/builder.go +1124 -0
  25. package/internal/citymap/builder_test.go +818 -0
  26. package/internal/judge/cache.go +180 -0
  27. package/internal/judge/cli.go +240 -0
  28. package/internal/judge/cli_test.go +68 -0
  29. package/internal/judge/fresh_summary_test.go +37 -0
  30. package/internal/judge/input.go +233 -0
  31. package/internal/judge/judge.go +416 -0
  32. package/internal/judge/judge_test.go +288 -0
  33. package/internal/judge/prompt.go +129 -0
  34. package/internal/judge/rubric.go +275 -0
  35. package/internal/judge/rubric_test.go +642 -0
  36. package/internal/model/agent.go +63 -0
  37. package/internal/model/agent_schema_test.go +86 -0
  38. package/internal/model/agent_test.go +64 -0
  39. package/internal/model/model.go +175 -0
  40. package/internal/model/report.go +166 -0
  41. package/internal/model/stats.go +151 -0
  42. package/internal/model/stats_test.go +89 -0
  43. package/internal/model/trace_schema_test.go +67 -0
  44. package/internal/server/analyze.go +286 -0
  45. package/internal/server/analyze_test.go +297 -0
  46. package/internal/server/codex_index_test.go +40 -0
  47. package/internal/server/hardening_test.go +147 -0
  48. package/internal/server/reportindex.go +91 -0
  49. package/internal/server/reportindex_test.go +116 -0
  50. package/internal/server/server.go +1099 -0
  51. package/internal/server/server_test.go +1389 -0
  52. package/internal/server/static/assets/fraunces-latin-ext-standard-italic-CGbN9UgK.woff2 +0 -0
  53. package/internal/server/static/assets/fraunces-latin-ext-standard-normal-CJcjJNj7.woff2 +0 -0
  54. package/internal/server/static/assets/fraunces-latin-standard-italic-lSdLDfvT.woff2 +0 -0
  55. package/internal/server/static/assets/fraunces-latin-standard-normal-DihXLNYH.woff2 +0 -0
  56. package/internal/server/static/assets/fraunces-vietnamese-standard-italic-DxWqP7Ku.woff2 +0 -0
  57. package/internal/server/static/assets/fraunces-vietnamese-standard-normal-Czevyj-6.woff2 +0 -0
  58. package/internal/server/static/assets/index-BNoY_BiB.css +1 -0
  59. package/internal/server/static/assets/index-C_adLrJr.js +3 -0
  60. package/internal/server/static/assets/react-gcHzaSmV.js +10 -0
  61. package/internal/server/static/assets/schibsted-grotesk-latin-ext-wght-normal-hsMS0n0O.woff2 +0 -0
  62. package/internal/server/static/assets/schibsted-grotesk-latin-wght-normal-Bb8VGrTG.woff2 +0 -0
  63. package/internal/server/static/assets/three-DnGjZfD1.js +4012 -0
  64. package/internal/server/static/index.html +26 -0
  65. package/internal/server/tracestore.go +173 -0
  66. package/internal/server/tracestore_test.go +51 -0
  67. package/internal/textutil/truncate.go +30 -0
  68. package/internal/textutil/truncate_test.go +39 -0
  69. package/package.json +35 -0
  70. package/web/e2e/agent-lens.spec.ts +688 -0
  71. package/web/index.html +23 -0
  72. package/web/package-lock.json +1933 -0
  73. package/web/package.json +33 -0
  74. package/web/playwright.config.ts +24 -0
  75. package/web/src/App.tsx +876 -0
  76. package/web/src/api/client.ts +74 -0
  77. package/web/src/main.tsx +12 -0
  78. package/web/src/playback/recorder.ts +160 -0
  79. package/web/src/playback/reducer.ts +91 -0
  80. package/web/src/scene/CityScene.tsx +638 -0
  81. package/web/src/scene/TreeScene.tsx +656 -0
  82. package/web/src/scene/dirLabels.ts +145 -0
  83. package/web/src/scene/sceneUtils.ts +144 -0
  84. package/web/src/scene/textures.ts +60 -0
  85. package/web/src/scene/trail.ts +79 -0
  86. package/web/src/scene/treeLayout.ts +169 -0
  87. package/web/src/state/filters.ts +40 -0
  88. package/web/src/state/store.ts +83 -0
  89. package/web/src/styles.css +2565 -0
  90. package/web/src/types.ts +315 -0
  91. package/web/src/ui/AgentsPanel.tsx +376 -0
  92. package/web/src/ui/Dock.tsx +104 -0
  93. package/web/src/ui/Hud.tsx +335 -0
  94. package/web/src/ui/Inspector.tsx +107 -0
  95. package/web/src/ui/LogoMark.tsx +38 -0
  96. package/web/src/ui/ReportPanel.tsx +491 -0
  97. package/web/src/ui/SessionRail.tsx +316 -0
  98. package/web/src/ui/Timeline.tsx +458 -0
  99. package/web/src/ui/ViewPanel.tsx +45 -0
  100. package/web/src/ui/shortcuts.ts +4 -0
  101. package/web/tsconfig.json +21 -0
  102. package/web/vite.config.ts +28 -0
@@ -0,0 +1,180 @@
1
+ package judge
2
+
3
+ import (
4
+ "encoding/json"
5
+ "os"
6
+ "path/filepath"
7
+
8
+ "github.com/hybridlabor-api/bdb-synapse/internal/model"
9
+ )
10
+
11
+ // Cache persists reports under one file per session key so re-opening a
12
+ // session never re-runs the judge. Reports are expensive; traces are not.
13
+ type Cache struct {
14
+ Dir string
15
+ }
16
+
17
+ // DefaultCacheDir is ~/.synapse/reports —.synapse's own data directory,
18
+ // never inside ~/.claude, ~/.codex, or the inspected repository.
19
+ func DefaultCacheDir() string {
20
+ home, err := os.UserHomeDir()
21
+ if err != nil {
22
+ return ""
23
+ }
24
+ return filepath.Join(home, ".synapse", "reports")
25
+ }
26
+
27
+ func (c Cache) path(sessionKey string) string {
28
+ return filepath.Join(c.Dir, sessionKey+".json")
29
+ }
30
+
31
+ // Path returns the on-disk location of the session's report file, for
32
+ // callers that fingerprint reports without loading them.
33
+ func (c Cache) Path(sessionKey string) string {
34
+ return c.path(sessionKey)
35
+ }
36
+
37
+ // Load returns the cached report for the session key, or nil when absent or
38
+ // unreadable (a corrupt cache entry is treated as a miss, not an error).
39
+ func (c Cache) Load(sessionKey string) *model.Report {
40
+ if c.Dir == "" || sessionKey == "" {
41
+ return nil
42
+ }
43
+ data, err := os.ReadFile(c.path(sessionKey))
44
+ if err != nil {
45
+ return nil
46
+ }
47
+ var report model.Report
48
+ if json.Unmarshal(data, &report) != nil {
49
+ return nil
50
+ }
51
+ // Syntactically valid but hollow payloads ("null", "{}", hand-edited
52
+ // files) must read as a miss: the UI dereferences dimensions and judge
53
+ // unconditionally.
54
+ if report.Version < 1 || len(report.Dimensions) == 0 || report.Judge.CLI == "" {
55
+ return nil
56
+ }
57
+ // A dimension's nil findings serializes as JSON null, which the panel
58
+ // maps over unconditionally; normalize rather than reject. Rubric
59
+ // criteria get the same treatment.
60
+ for i := range report.Dimensions {
61
+ if report.Dimensions[i].Findings == nil {
62
+ report.Dimensions[i].Findings = []model.ReportFinding{}
63
+ }
64
+ }
65
+ if report.Rubric != nil {
66
+ for i := range report.Rubric.Tasks {
67
+ for j := range report.Rubric.Tasks[i].Criteria {
68
+ if report.Rubric.Tasks[i].Criteria[j].Findings == nil {
69
+ report.Rubric.Tasks[i].Criteria[j].Findings = []model.ReportFinding{}
70
+ }
71
+ }
72
+ }
73
+ }
74
+ return &report
75
+ }
76
+
77
+ // FreshAgainstTrace reports whether a cached report still matches the trace
78
+ // it would be regenerated from: same prompt version and the same judge input
79
+ // digest — event counts alone miss user messages (stored as marks) and
80
+ // content edits. The rubric layer is checked against the current task
81
+ // evidence separately, because its input window is wider than the scoring
82
+ // document's. The judge CLI is deliberately not part of freshness — a valid
83
+ // report stays valid. Reports from before the digest existed are stale by
84
+ // construction.
85
+ func FreshAgainstTrace(report *model.Report, trace *model.Trace) bool {
86
+ if report == nil ||
87
+ report.Judge.PromptVersion != PromptVersion ||
88
+ report.Judge.InputDigest != InputDigest(trace) {
89
+ return false
90
+ }
91
+ return rubricFresh(report, trace)
92
+ }
93
+
94
+ // FreshAgainstSummary is the cheap approximation of FreshAgainstTrace for
95
+ // callers holding only a session summary (the list view): it shares the
96
+ // prompt-version and digest-presence preconditions but compares the
97
+ // summary's event and user-turn counts instead of recomputing the input
98
+ // digest, so no trace parse is needed. It may call a report fresh that
99
+ // FreshAgainstTrace grades stale (a content edit that keeps both counts);
100
+ // the panel's full check corrects that. User turns catch message-only
101
+ // growth the event count is blind to.
102
+ func FreshAgainstSummary(report *model.Report, meta model.SessionMeta) bool {
103
+ return report != nil &&
104
+ report.Judge.PromptVersion == PromptVersion &&
105
+ report.Judge.InputDigest != "" &&
106
+ report.Session.EventCount == meta.EventCount &&
107
+ report.Session.UserTurns == meta.UserTurns
108
+ }
109
+
110
+ // rubricFresh verifies the rubric layer against the CURRENT task evidence.
111
+ // InputDigest reads the scoring document, whose message window is tighter
112
+ // than the rubric phase's: a mid-window message revision can leave the
113
+ // scoring digest untouched while the task evidence — and therefore the
114
+ // rubric that would be regenerated — has changed.
115
+ func rubricFresh(report *model.Report, trace *model.Trace) bool {
116
+ rubric := report.Rubric
117
+ if rubric == nil {
118
+ return true
119
+ }
120
+ switch rubric.Status {
121
+ case model.RubricStatusScored:
122
+ if report.Judge.RubricPromptVersion != RubricPromptVersion {
123
+ return false
124
+ }
125
+ // A scored rubric must name a valid source and still fingerprint the
126
+ // task evidence it would be regenerated from.
127
+ if rubric.Source != model.RubricSourceFull && rubric.Source != model.RubricSourceTask {
128
+ return false
129
+ }
130
+ return rubric.TaskDigest == TaskDigest(trace, rubric.Source)
131
+ case model.RubricStatusUnavailable:
132
+ // Deterministic skips stay fresh only while their condition still
133
+ // holds for the current task evidence. no-events needs no recheck
134
+ // (events appearing changes the narrative, which InputDigest covers),
135
+ // and generation-failed deliberately stays fresh — re-running it is
136
+ // the user's explicit call, and RubricSatisfied already refuses to
137
+ // treat it as settled.
138
+ switch rubric.Reason {
139
+ case model.RubricReasonNoTaskText:
140
+ return len(taskMessages(trace.Marks)) == 0
141
+ case model.RubricReasonWeakTaskText:
142
+ return len(taskMessages(trace.Marks)) > 0 && taskTextRunes(trace.Marks) < weakTaskTextRunes
143
+ }
144
+ }
145
+ return true
146
+ }
147
+
148
+ func (c Cache) Store(sessionKey string, report *model.Report) error {
149
+ if c.Dir == "" || sessionKey == "" {
150
+ return nil
151
+ }
152
+ if err := os.MkdirAll(c.Dir, 0o755); err != nil {
153
+ return err
154
+ }
155
+ data, err := json.MarshalIndent(report, "", " ")
156
+ if err != nil {
157
+ return err
158
+ }
159
+ // A unique temp file per writer: the CLI and the server may finish
160
+ // evaluating the same session concurrently, and a shared name would let
161
+ // them truncate each other mid-write. Last rename wins, atomically.
162
+ tmp, err := os.CreateTemp(c.Dir, sessionKey+"-*.tmp")
163
+ if err != nil {
164
+ return err
165
+ }
166
+ if _, err := tmp.Write(data); err != nil {
167
+ tmp.Close()
168
+ os.Remove(tmp.Name())
169
+ return err
170
+ }
171
+ if err := tmp.Close(); err != nil {
172
+ os.Remove(tmp.Name())
173
+ return err
174
+ }
175
+ if err := os.Rename(tmp.Name(), c.path(sessionKey)); err != nil {
176
+ os.Remove(tmp.Name())
177
+ return err
178
+ }
179
+ return nil
180
+ }
@@ -0,0 +1,240 @@
1
+ package judge
2
+
3
+ import (
4
+ "bytes"
5
+ "context"
6
+ "encoding/json"
7
+ "fmt"
8
+ "os"
9
+ "os/exec"
10
+ "path/filepath"
11
+ "regexp"
12
+ "strings"
13
+
14
+ "github.com/hybridlabor-api/bdb-synapse/internal/textutil"
15
+ )
16
+
17
+ // RunResult carries the judge's raw text plus which model produced it. The
18
+ // model is recorded on the report — verdicts from different judges are only
19
+ // comparable when the report says who judged.
20
+ type RunResult struct {
21
+ Text string
22
+ Model string
23
+ }
24
+
25
+ // Runner abstracts the judge CLI subprocess so tests can stub it.
26
+ type Runner interface {
27
+ Run(ctx context.Context, prompt, input string) (RunResult, error)
28
+ Name() string
29
+ }
30
+
31
+ // SupportedCLIs lists judge CLIs in detection preference order.
32
+ var SupportedCLIs = []string{"claude", "codex"}
33
+
34
+ // DetectCLI returns the first supported judge CLI found on PATH.
35
+ func DetectCLI() (string, error) {
36
+ if clis := DetectCLIs(); len(clis) > 0 {
37
+ return clis[0], nil
38
+ }
39
+ return "", fmt.Errorf("no judge CLI found on PATH (looked for %s)", strings.Join(SupportedCLIs, ", "))
40
+ }
41
+
42
+ // DetectCLIs returns every supported judge CLI found on PATH, in preference
43
+ // order; the UI offers the full list so the user can pick the judge.
44
+ func DetectCLIs() []string {
45
+ var clis []string
46
+ for _, cli := range SupportedCLIs {
47
+ if _, err := exec.LookPath(cli); err == nil {
48
+ clis = append(clis, cli)
49
+ }
50
+ }
51
+ return clis
52
+ }
53
+
54
+ // WorkDir returns ~/.synapse/judge, the neutral directory judge subprocesses
55
+ // run in. It holds no repository and no project instructions, and adapters use
56
+ // IsWorkDir to recognize sessions recorded there as.synapse's own judge runs
57
+ // (a fallback for codex CLIs that predate --ephemeral).
58
+ func WorkDir() string {
59
+ home, err := os.UserHomeDir()
60
+ if err != nil {
61
+ return ""
62
+ }
63
+ return filepath.Join(home, ".synapse", "judge")
64
+ }
65
+
66
+ // IsWorkDir reports whether path is the judge working directory.
67
+ func IsWorkDir(path string) bool {
68
+ dir := WorkDir()
69
+ return dir != "" && path != "" && filepath.Clean(path) == dir
70
+ }
71
+
72
+ func ensureWorkDir() (string, error) {
73
+ dir := WorkDir()
74
+ if dir == "" {
75
+ return "", fmt.Errorf("judge workdir: cannot resolve home directory")
76
+ }
77
+ if err := os.MkdirAll(dir, 0o755); err != nil {
78
+ return "", fmt.Errorf("judge workdir: %w", err)
79
+ }
80
+ return dir, nil
81
+ }
82
+
83
+ // CLIRunner shells out to a local agent CLI in non-interactive mode.
84
+ //
85
+ // The trace under evaluation is untrusted input (a prompt injection in the
86
+ // evaluated session must not reach tools), so the judge runs sealed: no
87
+ // tools, no MCP servers, no user or project settings, and nothing the judge
88
+ // produces may surface as a session for.synapse itself to scan.
89
+ type CLIRunner struct {
90
+ CLI string
91
+ // Model overrides the CLI's default model when set (an alias like
92
+ // "sonnet" or a full name like "gpt-5.6-sol").
93
+ Model string
94
+ }
95
+
96
+ func (r CLIRunner) Name() string { return r.CLI }
97
+
98
+ func (r CLIRunner) Run(ctx context.Context, prompt, input string) (RunResult, error) {
99
+ workdir, err := ensureWorkDir()
100
+ if err != nil {
101
+ return RunResult{}, err
102
+ }
103
+ var cmd *exec.Cmd
104
+ switch r.CLI {
105
+ case "claude":
106
+ // --output-format json wraps the reply in a result envelope whose
107
+ // modelUsage names the model that actually answered; see
108
+ // parseClaudeEnvelope.
109
+ args := []string{"-p",
110
+ "--no-session-persistence", // never a session file for.synapse to re-scan
111
+ "--tools", "",
112
+ "--strict-mcp-config", // with no --mcp-config: zero MCP servers
113
+ "--setting-sources", "", // no user/project settings, hooks, or allowlists
114
+ "--output-format", "json",
115
+ }
116
+ if r.Model != "" {
117
+ args = append(args, "--model", r.Model)
118
+ }
119
+ cmd = exec.CommandContext(ctx, "claude", append(args, prompt)...)
120
+ cmd.Stdin = strings.NewReader(input)
121
+ case "codex":
122
+ // "-" reads the whole prompt from stdin — as an argv it would hit
123
+ // per-argument size limits on big traces. codex interleaves its own
124
+ // logging on stdout; extractJSON in the parser copes with the noise,
125
+ // and the preamble's "model:" line names the model. The run is still
126
+ // pinned to the judge workdir, and the codex adapter marks sessions
127
+ // recorded there auxiliary — a belt for CLIs that predate
128
+ // --ephemeral and for sessions old versions already wrote.
129
+ //
130
+ // The judge must be a pure text function: the feature flags and
131
+ // config below strip every tool a prompt injection in the evaluated
132
+ // trace could reach for. Ground-truth verified (not model
133
+ // self-reports): with this set the judge cannot read local files,
134
+ // fetch URLs, apply patches, or spawn collaboration agents — each
135
+ // attempt fails at the tool router or sandbox.
136
+ args := codexExecArgs(workdir)
137
+ if r.Model != "" {
138
+ args = append(args, "-c", "model="+r.Model)
139
+ }
140
+ cmd = exec.CommandContext(ctx, "codex", append(args, "-")...)
141
+ cmd.Stdin = strings.NewReader(prompt + "\n\n" + input)
142
+ default:
143
+ return RunResult{}, fmt.Errorf("unsupported judge CLI %q", r.CLI)
144
+ }
145
+ cmd.Dir = workdir
146
+ var stdout, stderr bytes.Buffer
147
+ cmd.Stdout = &stdout
148
+ cmd.Stderr = &stderr
149
+ if err := cmd.Run(); err != nil {
150
+ detail := strings.TrimSpace(stderr.String())
151
+ if detail == "" {
152
+ detail = strings.TrimSpace(stdout.String())
153
+ }
154
+ detail = truncateFailureDetail(detail)
155
+ return RunResult{}, fmt.Errorf("%s failed: %w: %s", r.CLI, err, detail)
156
+ }
157
+ if r.CLI == "claude" {
158
+ return parseClaudeEnvelope(stdout.String()), nil
159
+ }
160
+ // codex prints its config preamble (with the "model:" line) on stderr;
161
+ // older versions used stdout, so check both.
162
+ model := codexModel(stderr.String())
163
+ if model == "" {
164
+ model = codexModel(stdout.String())
165
+ }
166
+ return RunResult{Text: stdout.String(), Model: model}, nil
167
+ }
168
+
169
+ func truncateFailureDetail(detail string) string {
170
+ return textutil.TruncateRunes(detail, 500, "")
171
+ }
172
+
173
+ func codexExecArgs(workdir string) []string {
174
+ return []string{"exec",
175
+ "--ephemeral", // no session file for.synapse to re-scan
176
+ "--ignore-user-config", // no user MCP servers or profiles; auth stays
177
+ "--ignore-rules", // no user/project execpolicy rules
178
+ "--disable", "shell_tool",
179
+ "--disable", "browser_use",
180
+ "--disable", "browser_use_external",
181
+ "--disable", "computer_use",
182
+ "--disable", "in_app_browser",
183
+ "--disable", "apps",
184
+ "--disable", "plugins",
185
+ "--disable", "hooks",
186
+ "--disable", "multi_agent",
187
+ "--disable", "multi_agent_v2",
188
+ "--disable", "memories",
189
+ "--disable", "image_generation",
190
+ "-c", "include_apply_patch_tool=false",
191
+ "-c", "tools.view_image=false",
192
+ "-c", `web_search="disabled"`,
193
+ "--sandbox", "read-only", // defense in depth behind the tool strip
194
+ "--skip-git-repo-check", // the judge workdir is not a repository
195
+ "-C", workdir,
196
+ }
197
+ }
198
+
199
+ // claudeEnvelope mirrors the parts of `claude -p --output-format json` output
200
+ //.synapse needs: the reply text and per-model usage.
201
+ type claudeEnvelope struct {
202
+ Result string `json:"result"`
203
+ ModelUsage map[string]struct {
204
+ InputTokens int64 `json:"inputTokens"`
205
+ CacheReadInputTokens int64 `json:"cacheReadInputTokens"`
206
+ CacheCreationInputTokens int64 `json:"cacheCreationInputTokens"`
207
+ } `json:"modelUsage"`
208
+ }
209
+
210
+ // parseClaudeEnvelope unwraps the result envelope. The main model is the one
211
+ // that consumed the most input — the CLI also runs a small helper model for
212
+ // housekeeping, but only the judge reads the full evidence document. An
213
+ // unparseable envelope falls back to the raw text with no model recorded.
214
+ func parseClaudeEnvelope(raw string) RunResult {
215
+ var envelope claudeEnvelope
216
+ if err := json.Unmarshal([]byte(strings.TrimSpace(raw)), &envelope); err != nil || envelope.Result == "" {
217
+ return RunResult{Text: raw}
218
+ }
219
+ model := ""
220
+ var maxInput int64 = -1
221
+ for name, usage := range envelope.ModelUsage {
222
+ total := usage.InputTokens + usage.CacheReadInputTokens + usage.CacheCreationInputTokens
223
+ if total > maxInput {
224
+ maxInput = total
225
+ model = name
226
+ }
227
+ }
228
+ return RunResult{Text: envelope.Result, Model: model}
229
+ }
230
+
231
+ var codexModelLine = regexp.MustCompile(`(?m)^model:\s+(\S+)`)
232
+
233
+ // codexModel pulls the model name from the config preamble codex exec prints
234
+ // before the reply; absent a match the model stays unrecorded.
235
+ func codexModel(raw string) string {
236
+ if match := codexModelLine.FindStringSubmatch(raw); match != nil {
237
+ return match[1]
238
+ }
239
+ return ""
240
+ }
@@ -0,0 +1,68 @@
1
+ package judge
2
+
3
+ import (
4
+ "slices"
5
+ "strings"
6
+ "testing"
7
+ "unicode/utf8"
8
+ )
9
+
10
+ func TestParseClaudeEnvelopePicksMainModel(t *testing.T) {
11
+ raw := `{"type":"result","result":"{\"ok\":true}","modelUsage":{
12
+ "claude-haiku-4-5":{"inputTokens":523,"outputTokens":13},
13
+ "claude-sonnet-5":{"inputTokens":1,"cacheReadInputTokens":3289,"cacheCreationInputTokens":5563}}}`
14
+ got := parseClaudeEnvelope(raw)
15
+ if got.Text != `{"ok":true}` {
16
+ t.Fatalf("text = %q", got.Text)
17
+ }
18
+ // The helper model wrote more output tokens, but only the judge read the
19
+ // full evidence document — input volume picks the right one.
20
+ if got.Model != "claude-sonnet-5" {
21
+ t.Fatalf("model = %q, want claude-sonnet-5", got.Model)
22
+ }
23
+ }
24
+
25
+ func TestParseClaudeEnvelopeFallsBackToRawText(t *testing.T) {
26
+ raw := `plain text, not an envelope {"ok":true}`
27
+ got := parseClaudeEnvelope(raw)
28
+ if got.Text != raw || got.Model != "" {
29
+ t.Fatalf("fallback = %#v", got)
30
+ }
31
+ }
32
+
33
+ func TestCodexModelReadsPreamble(t *testing.T) {
34
+ raw := "OpenAI Codex v0.143.0\n--------\nworkdir: /tmp\nmodel: gpt-5.6-sol\nprovider: openai\n--------\n{\"ok\":true}\n"
35
+ if got := codexModel(raw); got != "gpt-5.6-sol" {
36
+ t.Fatalf("model = %q", got)
37
+ }
38
+ if got := codexModel("no preamble at all"); got != "" {
39
+ t.Fatalf("expected empty model, got %q", got)
40
+ }
41
+ }
42
+
43
+ func TestCodexExecArgsExcludeRemovedFeatures(t *testing.T) {
44
+ args := codexExecArgs("/tmp/judge")
45
+ if slices.Contains(args, "browser_use_full_cdp_access") {
46
+ t.Fatal("codex args include removed browser_use_full_cdp_access feature")
47
+ }
48
+ }
49
+
50
+ func TestTruncateFailureDetailPreservesUTF8(t *testing.T) {
51
+ detail := strings.Repeat("a", 499) + "界tail"
52
+ got := truncateFailureDetail(detail)
53
+
54
+ if !utf8.ValidString(got) {
55
+ t.Fatalf("truncated detail is not valid UTF-8: %q", got)
56
+ }
57
+ want := strings.Repeat("a", 499) + "界"
58
+ if got != want {
59
+ t.Fatalf("truncated detail = %q, want %q", got, want)
60
+ }
61
+ }
62
+
63
+ func TestTruncateFailureDetailRepairsInvalidUTF8(t *testing.T) {
64
+ got := truncateFailureDetail(string([]byte{'o', 'k', 0xff}))
65
+ if !utf8.ValidString(got) {
66
+ t.Fatalf("failure detail contains invalid UTF-8: %q", got)
67
+ }
68
+ }
@@ -0,0 +1,37 @@
1
+ package judge
2
+
3
+ import (
4
+ "testing"
5
+
6
+ "github.com/hybridlabor-api/bdb-synapse/internal/model"
7
+ )
8
+
9
+ func TestFreshAgainstSummary(t *testing.T) {
10
+ report := &model.Report{
11
+ Session: model.ReportSession{EventCount: 3, UserTurns: 2},
12
+ Judge: model.ReportJudge{PromptVersion: PromptVersion, InputDigest: "d"},
13
+ }
14
+ meta := model.SessionMeta{EventCount: 3, UserTurns: 2}
15
+ if !FreshAgainstSummary(report, meta) {
16
+ t.Fatal("matching counts and current prompt graded stale")
17
+ }
18
+ if FreshAgainstSummary(report, model.SessionMeta{EventCount: 4, UserTurns: 2}) {
19
+ t.Fatal("event growth graded fresh")
20
+ }
21
+ if FreshAgainstSummary(report, model.SessionMeta{EventCount: 3, UserTurns: 3}) {
22
+ t.Fatal("message-only growth graded fresh")
23
+ }
24
+ noDigest := *report
25
+ noDigest.Judge.InputDigest = ""
26
+ if FreshAgainstSummary(&noDigest, meta) {
27
+ t.Fatal("pre-digest report graded fresh; the panel would disagree")
28
+ }
29
+ oldPrompt := *report
30
+ oldPrompt.Judge.PromptVersion = PromptVersion - 1
31
+ if FreshAgainstSummary(&oldPrompt, meta) {
32
+ t.Fatal("outdated prompt graded fresh")
33
+ }
34
+ if FreshAgainstSummary(nil, meta) {
35
+ t.Fatal("nil report graded fresh")
36
+ }
37
+ }