@hybridlabor-api/bdb-synapse 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +173 -0
- package/bin/synapse +0 -0
- package/cmd/rubriceval/main.go +308 -0
- package/cmd/synapse/main.go +280 -0
- package/cmd/synapse/main_test.go +16 -0
- package/go.mod +7 -0
- package/go.sum +6 -0
- package/internal/adapter/adapter.go +1117 -0
- package/internal/adapter/adapter_test.go +518 -0
- package/internal/adapter/agy/adapter.go +193 -0
- package/internal/adapter/claudecode/adapter.go +415 -0
- package/internal/adapter/claudecode/adapter_test.go +260 -0
- package/internal/adapter/claudecode/agents.go +387 -0
- package/internal/adapter/claudecode/agents_test.go +480 -0
- package/internal/adapter/claudecode/summary_inputs_test.go +19 -0
- package/internal/adapter/codex/adapter.go +919 -0
- package/internal/adapter/codex/adapter_test.go +920 -0
- package/internal/adapter/codex/agents.go +401 -0
- package/internal/adapter/codex/agents_test.go +610 -0
- package/internal/adapter/codex/summary_inputs_test.go +20 -0
- package/internal/adapter/pi/adapter.go +483 -0
- package/internal/adapter/pi/adapter_test.go +517 -0
- package/internal/citymap/builder.go +1124 -0
- package/internal/citymap/builder_test.go +818 -0
- package/internal/judge/cache.go +180 -0
- package/internal/judge/cli.go +240 -0
- package/internal/judge/cli_test.go +68 -0
- package/internal/judge/fresh_summary_test.go +37 -0
- package/internal/judge/input.go +233 -0
- package/internal/judge/judge.go +416 -0
- package/internal/judge/judge_test.go +288 -0
- package/internal/judge/prompt.go +129 -0
- package/internal/judge/rubric.go +275 -0
- package/internal/judge/rubric_test.go +642 -0
- package/internal/model/agent.go +63 -0
- package/internal/model/agent_schema_test.go +86 -0
- package/internal/model/agent_test.go +64 -0
- package/internal/model/model.go +175 -0
- package/internal/model/report.go +166 -0
- package/internal/model/stats.go +151 -0
- package/internal/model/stats_test.go +89 -0
- package/internal/model/trace_schema_test.go +67 -0
- package/internal/server/analyze.go +286 -0
- package/internal/server/analyze_test.go +297 -0
- package/internal/server/codex_index_test.go +40 -0
- package/internal/server/hardening_test.go +147 -0
- package/internal/server/reportindex.go +91 -0
- package/internal/server/reportindex_test.go +116 -0
- package/internal/server/server.go +1099 -0
- package/internal/server/server_test.go +1389 -0
- package/internal/server/static/assets/fraunces-latin-ext-standard-italic-CGbN9UgK.woff2 +0 -0
- package/internal/server/static/assets/fraunces-latin-ext-standard-normal-CJcjJNj7.woff2 +0 -0
- package/internal/server/static/assets/fraunces-latin-standard-italic-lSdLDfvT.woff2 +0 -0
- package/internal/server/static/assets/fraunces-latin-standard-normal-DihXLNYH.woff2 +0 -0
- package/internal/server/static/assets/fraunces-vietnamese-standard-italic-DxWqP7Ku.woff2 +0 -0
- package/internal/server/static/assets/fraunces-vietnamese-standard-normal-Czevyj-6.woff2 +0 -0
- package/internal/server/static/assets/index-BNoY_BiB.css +1 -0
- package/internal/server/static/assets/index-C_adLrJr.js +3 -0
- package/internal/server/static/assets/react-gcHzaSmV.js +10 -0
- package/internal/server/static/assets/schibsted-grotesk-latin-ext-wght-normal-hsMS0n0O.woff2 +0 -0
- package/internal/server/static/assets/schibsted-grotesk-latin-wght-normal-Bb8VGrTG.woff2 +0 -0
- package/internal/server/static/assets/three-DnGjZfD1.js +4012 -0
- package/internal/server/static/index.html +26 -0
- package/internal/server/tracestore.go +173 -0
- package/internal/server/tracestore_test.go +51 -0
- package/internal/textutil/truncate.go +30 -0
- package/internal/textutil/truncate_test.go +39 -0
- package/package.json +35 -0
- package/web/e2e/agent-lens.spec.ts +688 -0
- package/web/index.html +23 -0
- package/web/package-lock.json +1933 -0
- package/web/package.json +33 -0
- package/web/playwright.config.ts +24 -0
- package/web/src/App.tsx +876 -0
- package/web/src/api/client.ts +74 -0
- package/web/src/main.tsx +12 -0
- package/web/src/playback/recorder.ts +160 -0
- package/web/src/playback/reducer.ts +91 -0
- package/web/src/scene/CityScene.tsx +638 -0
- package/web/src/scene/TreeScene.tsx +656 -0
- package/web/src/scene/dirLabels.ts +145 -0
- package/web/src/scene/sceneUtils.ts +144 -0
- package/web/src/scene/textures.ts +60 -0
- package/web/src/scene/trail.ts +79 -0
- package/web/src/scene/treeLayout.ts +169 -0
- package/web/src/state/filters.ts +40 -0
- package/web/src/state/store.ts +83 -0
- package/web/src/styles.css +2565 -0
- package/web/src/types.ts +315 -0
- package/web/src/ui/AgentsPanel.tsx +376 -0
- package/web/src/ui/Dock.tsx +104 -0
- package/web/src/ui/Hud.tsx +335 -0
- package/web/src/ui/Inspector.tsx +107 -0
- package/web/src/ui/LogoMark.tsx +38 -0
- package/web/src/ui/ReportPanel.tsx +491 -0
- package/web/src/ui/SessionRail.tsx +316 -0
- package/web/src/ui/Timeline.tsx +458 -0
- package/web/src/ui/ViewPanel.tsx +45 -0
- package/web/src/ui/shortcuts.ts +4 -0
- package/web/tsconfig.json +21 -0
- package/web/vite.config.ts +28 -0
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
package judge
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"encoding/json"
|
|
5
|
+
"os"
|
|
6
|
+
"path/filepath"
|
|
7
|
+
|
|
8
|
+
"github.com/hybridlabor-api/bdb-synapse/internal/model"
|
|
9
|
+
)
|
|
10
|
+
|
|
11
|
+
// Cache persists reports under one file per session key so re-opening a
|
|
12
|
+
// session never re-runs the judge. Reports are expensive; traces are not.
|
|
13
|
+
type Cache struct {
|
|
14
|
+
Dir string
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
// DefaultCacheDir is ~/.synapse/reports —.synapse's own data directory,
|
|
18
|
+
// never inside ~/.claude, ~/.codex, or the inspected repository.
|
|
19
|
+
func DefaultCacheDir() string {
|
|
20
|
+
home, err := os.UserHomeDir()
|
|
21
|
+
if err != nil {
|
|
22
|
+
return ""
|
|
23
|
+
}
|
|
24
|
+
return filepath.Join(home, ".synapse", "reports")
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
func (c Cache) path(sessionKey string) string {
|
|
28
|
+
return filepath.Join(c.Dir, sessionKey+".json")
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
// Path returns the on-disk location of the session's report file, for
|
|
32
|
+
// callers that fingerprint reports without loading them.
|
|
33
|
+
func (c Cache) Path(sessionKey string) string {
|
|
34
|
+
return c.path(sessionKey)
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
// Load returns the cached report for the session key, or nil when absent or
|
|
38
|
+
// unreadable (a corrupt cache entry is treated as a miss, not an error).
|
|
39
|
+
func (c Cache) Load(sessionKey string) *model.Report {
|
|
40
|
+
if c.Dir == "" || sessionKey == "" {
|
|
41
|
+
return nil
|
|
42
|
+
}
|
|
43
|
+
data, err := os.ReadFile(c.path(sessionKey))
|
|
44
|
+
if err != nil {
|
|
45
|
+
return nil
|
|
46
|
+
}
|
|
47
|
+
var report model.Report
|
|
48
|
+
if json.Unmarshal(data, &report) != nil {
|
|
49
|
+
return nil
|
|
50
|
+
}
|
|
51
|
+
// Syntactically valid but hollow payloads ("null", "{}", hand-edited
|
|
52
|
+
// files) must read as a miss: the UI dereferences dimensions and judge
|
|
53
|
+
// unconditionally.
|
|
54
|
+
if report.Version < 1 || len(report.Dimensions) == 0 || report.Judge.CLI == "" {
|
|
55
|
+
return nil
|
|
56
|
+
}
|
|
57
|
+
// A dimension's nil findings serializes as JSON null, which the panel
|
|
58
|
+
// maps over unconditionally; normalize rather than reject. Rubric
|
|
59
|
+
// criteria get the same treatment.
|
|
60
|
+
for i := range report.Dimensions {
|
|
61
|
+
if report.Dimensions[i].Findings == nil {
|
|
62
|
+
report.Dimensions[i].Findings = []model.ReportFinding{}
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
if report.Rubric != nil {
|
|
66
|
+
for i := range report.Rubric.Tasks {
|
|
67
|
+
for j := range report.Rubric.Tasks[i].Criteria {
|
|
68
|
+
if report.Rubric.Tasks[i].Criteria[j].Findings == nil {
|
|
69
|
+
report.Rubric.Tasks[i].Criteria[j].Findings = []model.ReportFinding{}
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
return &report
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
// FreshAgainstTrace reports whether a cached report still matches the trace
|
|
78
|
+
// it would be regenerated from: same prompt version and the same judge input
|
|
79
|
+
// digest — event counts alone miss user messages (stored as marks) and
|
|
80
|
+
// content edits. The rubric layer is checked against the current task
|
|
81
|
+
// evidence separately, because its input window is wider than the scoring
|
|
82
|
+
// document's. The judge CLI is deliberately not part of freshness — a valid
|
|
83
|
+
// report stays valid. Reports from before the digest existed are stale by
|
|
84
|
+
// construction.
|
|
85
|
+
func FreshAgainstTrace(report *model.Report, trace *model.Trace) bool {
|
|
86
|
+
if report == nil ||
|
|
87
|
+
report.Judge.PromptVersion != PromptVersion ||
|
|
88
|
+
report.Judge.InputDigest != InputDigest(trace) {
|
|
89
|
+
return false
|
|
90
|
+
}
|
|
91
|
+
return rubricFresh(report, trace)
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
// FreshAgainstSummary is the cheap approximation of FreshAgainstTrace for
|
|
95
|
+
// callers holding only a session summary (the list view): it shares the
|
|
96
|
+
// prompt-version and digest-presence preconditions but compares the
|
|
97
|
+
// summary's event and user-turn counts instead of recomputing the input
|
|
98
|
+
// digest, so no trace parse is needed. It may call a report fresh that
|
|
99
|
+
// FreshAgainstTrace grades stale (a content edit that keeps both counts);
|
|
100
|
+
// the panel's full check corrects that. User turns catch message-only
|
|
101
|
+
// growth the event count is blind to.
|
|
102
|
+
func FreshAgainstSummary(report *model.Report, meta model.SessionMeta) bool {
|
|
103
|
+
return report != nil &&
|
|
104
|
+
report.Judge.PromptVersion == PromptVersion &&
|
|
105
|
+
report.Judge.InputDigest != "" &&
|
|
106
|
+
report.Session.EventCount == meta.EventCount &&
|
|
107
|
+
report.Session.UserTurns == meta.UserTurns
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
// rubricFresh verifies the rubric layer against the CURRENT task evidence.
|
|
111
|
+
// InputDigest reads the scoring document, whose message window is tighter
|
|
112
|
+
// than the rubric phase's: a mid-window message revision can leave the
|
|
113
|
+
// scoring digest untouched while the task evidence — and therefore the
|
|
114
|
+
// rubric that would be regenerated — has changed.
|
|
115
|
+
func rubricFresh(report *model.Report, trace *model.Trace) bool {
|
|
116
|
+
rubric := report.Rubric
|
|
117
|
+
if rubric == nil {
|
|
118
|
+
return true
|
|
119
|
+
}
|
|
120
|
+
switch rubric.Status {
|
|
121
|
+
case model.RubricStatusScored:
|
|
122
|
+
if report.Judge.RubricPromptVersion != RubricPromptVersion {
|
|
123
|
+
return false
|
|
124
|
+
}
|
|
125
|
+
// A scored rubric must name a valid source and still fingerprint the
|
|
126
|
+
// task evidence it would be regenerated from.
|
|
127
|
+
if rubric.Source != model.RubricSourceFull && rubric.Source != model.RubricSourceTask {
|
|
128
|
+
return false
|
|
129
|
+
}
|
|
130
|
+
return rubric.TaskDigest == TaskDigest(trace, rubric.Source)
|
|
131
|
+
case model.RubricStatusUnavailable:
|
|
132
|
+
// Deterministic skips stay fresh only while their condition still
|
|
133
|
+
// holds for the current task evidence. no-events needs no recheck
|
|
134
|
+
// (events appearing changes the narrative, which InputDigest covers),
|
|
135
|
+
// and generation-failed deliberately stays fresh — re-running it is
|
|
136
|
+
// the user's explicit call, and RubricSatisfied already refuses to
|
|
137
|
+
// treat it as settled.
|
|
138
|
+
switch rubric.Reason {
|
|
139
|
+
case model.RubricReasonNoTaskText:
|
|
140
|
+
return len(taskMessages(trace.Marks)) == 0
|
|
141
|
+
case model.RubricReasonWeakTaskText:
|
|
142
|
+
return len(taskMessages(trace.Marks)) > 0 && taskTextRunes(trace.Marks) < weakTaskTextRunes
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
return true
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
func (c Cache) Store(sessionKey string, report *model.Report) error {
|
|
149
|
+
if c.Dir == "" || sessionKey == "" {
|
|
150
|
+
return nil
|
|
151
|
+
}
|
|
152
|
+
if err := os.MkdirAll(c.Dir, 0o755); err != nil {
|
|
153
|
+
return err
|
|
154
|
+
}
|
|
155
|
+
data, err := json.MarshalIndent(report, "", " ")
|
|
156
|
+
if err != nil {
|
|
157
|
+
return err
|
|
158
|
+
}
|
|
159
|
+
// A unique temp file per writer: the CLI and the server may finish
|
|
160
|
+
// evaluating the same session concurrently, and a shared name would let
|
|
161
|
+
// them truncate each other mid-write. Last rename wins, atomically.
|
|
162
|
+
tmp, err := os.CreateTemp(c.Dir, sessionKey+"-*.tmp")
|
|
163
|
+
if err != nil {
|
|
164
|
+
return err
|
|
165
|
+
}
|
|
166
|
+
if _, err := tmp.Write(data); err != nil {
|
|
167
|
+
tmp.Close()
|
|
168
|
+
os.Remove(tmp.Name())
|
|
169
|
+
return err
|
|
170
|
+
}
|
|
171
|
+
if err := tmp.Close(); err != nil {
|
|
172
|
+
os.Remove(tmp.Name())
|
|
173
|
+
return err
|
|
174
|
+
}
|
|
175
|
+
if err := os.Rename(tmp.Name(), c.path(sessionKey)); err != nil {
|
|
176
|
+
os.Remove(tmp.Name())
|
|
177
|
+
return err
|
|
178
|
+
}
|
|
179
|
+
return nil
|
|
180
|
+
}
|
|
@@ -0,0 +1,240 @@
|
|
|
1
|
+
package judge
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"bytes"
|
|
5
|
+
"context"
|
|
6
|
+
"encoding/json"
|
|
7
|
+
"fmt"
|
|
8
|
+
"os"
|
|
9
|
+
"os/exec"
|
|
10
|
+
"path/filepath"
|
|
11
|
+
"regexp"
|
|
12
|
+
"strings"
|
|
13
|
+
|
|
14
|
+
"github.com/hybridlabor-api/bdb-synapse/internal/textutil"
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
// RunResult carries the judge's raw text plus which model produced it. The
|
|
18
|
+
// model is recorded on the report — verdicts from different judges are only
|
|
19
|
+
// comparable when the report says who judged.
|
|
20
|
+
type RunResult struct {
|
|
21
|
+
Text string
|
|
22
|
+
Model string
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
// Runner abstracts the judge CLI subprocess so tests can stub it.
|
|
26
|
+
type Runner interface {
|
|
27
|
+
Run(ctx context.Context, prompt, input string) (RunResult, error)
|
|
28
|
+
Name() string
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
// SupportedCLIs lists judge CLIs in detection preference order.
|
|
32
|
+
var SupportedCLIs = []string{"claude", "codex"}
|
|
33
|
+
|
|
34
|
+
// DetectCLI returns the first supported judge CLI found on PATH.
|
|
35
|
+
func DetectCLI() (string, error) {
|
|
36
|
+
if clis := DetectCLIs(); len(clis) > 0 {
|
|
37
|
+
return clis[0], nil
|
|
38
|
+
}
|
|
39
|
+
return "", fmt.Errorf("no judge CLI found on PATH (looked for %s)", strings.Join(SupportedCLIs, ", "))
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
// DetectCLIs returns every supported judge CLI found on PATH, in preference
|
|
43
|
+
// order; the UI offers the full list so the user can pick the judge.
|
|
44
|
+
func DetectCLIs() []string {
|
|
45
|
+
var clis []string
|
|
46
|
+
for _, cli := range SupportedCLIs {
|
|
47
|
+
if _, err := exec.LookPath(cli); err == nil {
|
|
48
|
+
clis = append(clis, cli)
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
return clis
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
// WorkDir returns ~/.synapse/judge, the neutral directory judge subprocesses
|
|
55
|
+
// run in. It holds no repository and no project instructions, and adapters use
|
|
56
|
+
// IsWorkDir to recognize sessions recorded there as.synapse's own judge runs
|
|
57
|
+
// (a fallback for codex CLIs that predate --ephemeral).
|
|
58
|
+
func WorkDir() string {
|
|
59
|
+
home, err := os.UserHomeDir()
|
|
60
|
+
if err != nil {
|
|
61
|
+
return ""
|
|
62
|
+
}
|
|
63
|
+
return filepath.Join(home, ".synapse", "judge")
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
// IsWorkDir reports whether path is the judge working directory.
|
|
67
|
+
func IsWorkDir(path string) bool {
|
|
68
|
+
dir := WorkDir()
|
|
69
|
+
return dir != "" && path != "" && filepath.Clean(path) == dir
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
func ensureWorkDir() (string, error) {
|
|
73
|
+
dir := WorkDir()
|
|
74
|
+
if dir == "" {
|
|
75
|
+
return "", fmt.Errorf("judge workdir: cannot resolve home directory")
|
|
76
|
+
}
|
|
77
|
+
if err := os.MkdirAll(dir, 0o755); err != nil {
|
|
78
|
+
return "", fmt.Errorf("judge workdir: %w", err)
|
|
79
|
+
}
|
|
80
|
+
return dir, nil
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
// CLIRunner shells out to a local agent CLI in non-interactive mode.
|
|
84
|
+
//
|
|
85
|
+
// The trace under evaluation is untrusted input (a prompt injection in the
|
|
86
|
+
// evaluated session must not reach tools), so the judge runs sealed: no
|
|
87
|
+
// tools, no MCP servers, no user or project settings, and nothing the judge
|
|
88
|
+
// produces may surface as a session for.synapse itself to scan.
|
|
89
|
+
type CLIRunner struct {
|
|
90
|
+
CLI string
|
|
91
|
+
// Model overrides the CLI's default model when set (an alias like
|
|
92
|
+
// "sonnet" or a full name like "gpt-5.6-sol").
|
|
93
|
+
Model string
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
func (r CLIRunner) Name() string { return r.CLI }
|
|
97
|
+
|
|
98
|
+
func (r CLIRunner) Run(ctx context.Context, prompt, input string) (RunResult, error) {
|
|
99
|
+
workdir, err := ensureWorkDir()
|
|
100
|
+
if err != nil {
|
|
101
|
+
return RunResult{}, err
|
|
102
|
+
}
|
|
103
|
+
var cmd *exec.Cmd
|
|
104
|
+
switch r.CLI {
|
|
105
|
+
case "claude":
|
|
106
|
+
// --output-format json wraps the reply in a result envelope whose
|
|
107
|
+
// modelUsage names the model that actually answered; see
|
|
108
|
+
// parseClaudeEnvelope.
|
|
109
|
+
args := []string{"-p",
|
|
110
|
+
"--no-session-persistence", // never a session file for.synapse to re-scan
|
|
111
|
+
"--tools", "",
|
|
112
|
+
"--strict-mcp-config", // with no --mcp-config: zero MCP servers
|
|
113
|
+
"--setting-sources", "", // no user/project settings, hooks, or allowlists
|
|
114
|
+
"--output-format", "json",
|
|
115
|
+
}
|
|
116
|
+
if r.Model != "" {
|
|
117
|
+
args = append(args, "--model", r.Model)
|
|
118
|
+
}
|
|
119
|
+
cmd = exec.CommandContext(ctx, "claude", append(args, prompt)...)
|
|
120
|
+
cmd.Stdin = strings.NewReader(input)
|
|
121
|
+
case "codex":
|
|
122
|
+
// "-" reads the whole prompt from stdin — as an argv it would hit
|
|
123
|
+
// per-argument size limits on big traces. codex interleaves its own
|
|
124
|
+
// logging on stdout; extractJSON in the parser copes with the noise,
|
|
125
|
+
// and the preamble's "model:" line names the model. The run is still
|
|
126
|
+
// pinned to the judge workdir, and the codex adapter marks sessions
|
|
127
|
+
// recorded there auxiliary — a belt for CLIs that predate
|
|
128
|
+
// --ephemeral and for sessions old versions already wrote.
|
|
129
|
+
//
|
|
130
|
+
// The judge must be a pure text function: the feature flags and
|
|
131
|
+
// config below strip every tool a prompt injection in the evaluated
|
|
132
|
+
// trace could reach for. Ground-truth verified (not model
|
|
133
|
+
// self-reports): with this set the judge cannot read local files,
|
|
134
|
+
// fetch URLs, apply patches, or spawn collaboration agents — each
|
|
135
|
+
// attempt fails at the tool router or sandbox.
|
|
136
|
+
args := codexExecArgs(workdir)
|
|
137
|
+
if r.Model != "" {
|
|
138
|
+
args = append(args, "-c", "model="+r.Model)
|
|
139
|
+
}
|
|
140
|
+
cmd = exec.CommandContext(ctx, "codex", append(args, "-")...)
|
|
141
|
+
cmd.Stdin = strings.NewReader(prompt + "\n\n" + input)
|
|
142
|
+
default:
|
|
143
|
+
return RunResult{}, fmt.Errorf("unsupported judge CLI %q", r.CLI)
|
|
144
|
+
}
|
|
145
|
+
cmd.Dir = workdir
|
|
146
|
+
var stdout, stderr bytes.Buffer
|
|
147
|
+
cmd.Stdout = &stdout
|
|
148
|
+
cmd.Stderr = &stderr
|
|
149
|
+
if err := cmd.Run(); err != nil {
|
|
150
|
+
detail := strings.TrimSpace(stderr.String())
|
|
151
|
+
if detail == "" {
|
|
152
|
+
detail = strings.TrimSpace(stdout.String())
|
|
153
|
+
}
|
|
154
|
+
detail = truncateFailureDetail(detail)
|
|
155
|
+
return RunResult{}, fmt.Errorf("%s failed: %w: %s", r.CLI, err, detail)
|
|
156
|
+
}
|
|
157
|
+
if r.CLI == "claude" {
|
|
158
|
+
return parseClaudeEnvelope(stdout.String()), nil
|
|
159
|
+
}
|
|
160
|
+
// codex prints its config preamble (with the "model:" line) on stderr;
|
|
161
|
+
// older versions used stdout, so check both.
|
|
162
|
+
model := codexModel(stderr.String())
|
|
163
|
+
if model == "" {
|
|
164
|
+
model = codexModel(stdout.String())
|
|
165
|
+
}
|
|
166
|
+
return RunResult{Text: stdout.String(), Model: model}, nil
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
func truncateFailureDetail(detail string) string {
|
|
170
|
+
return textutil.TruncateRunes(detail, 500, "")
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
func codexExecArgs(workdir string) []string {
|
|
174
|
+
return []string{"exec",
|
|
175
|
+
"--ephemeral", // no session file for.synapse to re-scan
|
|
176
|
+
"--ignore-user-config", // no user MCP servers or profiles; auth stays
|
|
177
|
+
"--ignore-rules", // no user/project execpolicy rules
|
|
178
|
+
"--disable", "shell_tool",
|
|
179
|
+
"--disable", "browser_use",
|
|
180
|
+
"--disable", "browser_use_external",
|
|
181
|
+
"--disable", "computer_use",
|
|
182
|
+
"--disable", "in_app_browser",
|
|
183
|
+
"--disable", "apps",
|
|
184
|
+
"--disable", "plugins",
|
|
185
|
+
"--disable", "hooks",
|
|
186
|
+
"--disable", "multi_agent",
|
|
187
|
+
"--disable", "multi_agent_v2",
|
|
188
|
+
"--disable", "memories",
|
|
189
|
+
"--disable", "image_generation",
|
|
190
|
+
"-c", "include_apply_patch_tool=false",
|
|
191
|
+
"-c", "tools.view_image=false",
|
|
192
|
+
"-c", `web_search="disabled"`,
|
|
193
|
+
"--sandbox", "read-only", // defense in depth behind the tool strip
|
|
194
|
+
"--skip-git-repo-check", // the judge workdir is not a repository
|
|
195
|
+
"-C", workdir,
|
|
196
|
+
}
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
// claudeEnvelope mirrors the parts of `claude -p --output-format json` output
|
|
200
|
+
//.synapse needs: the reply text and per-model usage.
|
|
201
|
+
type claudeEnvelope struct {
|
|
202
|
+
Result string `json:"result"`
|
|
203
|
+
ModelUsage map[string]struct {
|
|
204
|
+
InputTokens int64 `json:"inputTokens"`
|
|
205
|
+
CacheReadInputTokens int64 `json:"cacheReadInputTokens"`
|
|
206
|
+
CacheCreationInputTokens int64 `json:"cacheCreationInputTokens"`
|
|
207
|
+
} `json:"modelUsage"`
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
// parseClaudeEnvelope unwraps the result envelope. The main model is the one
|
|
211
|
+
// that consumed the most input — the CLI also runs a small helper model for
|
|
212
|
+
// housekeeping, but only the judge reads the full evidence document. An
|
|
213
|
+
// unparseable envelope falls back to the raw text with no model recorded.
|
|
214
|
+
func parseClaudeEnvelope(raw string) RunResult {
|
|
215
|
+
var envelope claudeEnvelope
|
|
216
|
+
if err := json.Unmarshal([]byte(strings.TrimSpace(raw)), &envelope); err != nil || envelope.Result == "" {
|
|
217
|
+
return RunResult{Text: raw}
|
|
218
|
+
}
|
|
219
|
+
model := ""
|
|
220
|
+
var maxInput int64 = -1
|
|
221
|
+
for name, usage := range envelope.ModelUsage {
|
|
222
|
+
total := usage.InputTokens + usage.CacheReadInputTokens + usage.CacheCreationInputTokens
|
|
223
|
+
if total > maxInput {
|
|
224
|
+
maxInput = total
|
|
225
|
+
model = name
|
|
226
|
+
}
|
|
227
|
+
}
|
|
228
|
+
return RunResult{Text: envelope.Result, Model: model}
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
var codexModelLine = regexp.MustCompile(`(?m)^model:\s+(\S+)`)
|
|
232
|
+
|
|
233
|
+
// codexModel pulls the model name from the config preamble codex exec prints
|
|
234
|
+
// before the reply; absent a match the model stays unrecorded.
|
|
235
|
+
func codexModel(raw string) string {
|
|
236
|
+
if match := codexModelLine.FindStringSubmatch(raw); match != nil {
|
|
237
|
+
return match[1]
|
|
238
|
+
}
|
|
239
|
+
return ""
|
|
240
|
+
}
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
package judge
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"slices"
|
|
5
|
+
"strings"
|
|
6
|
+
"testing"
|
|
7
|
+
"unicode/utf8"
|
|
8
|
+
)
|
|
9
|
+
|
|
10
|
+
func TestParseClaudeEnvelopePicksMainModel(t *testing.T) {
|
|
11
|
+
raw := `{"type":"result","result":"{\"ok\":true}","modelUsage":{
|
|
12
|
+
"claude-haiku-4-5":{"inputTokens":523,"outputTokens":13},
|
|
13
|
+
"claude-sonnet-5":{"inputTokens":1,"cacheReadInputTokens":3289,"cacheCreationInputTokens":5563}}}`
|
|
14
|
+
got := parseClaudeEnvelope(raw)
|
|
15
|
+
if got.Text != `{"ok":true}` {
|
|
16
|
+
t.Fatalf("text = %q", got.Text)
|
|
17
|
+
}
|
|
18
|
+
// The helper model wrote more output tokens, but only the judge read the
|
|
19
|
+
// full evidence document — input volume picks the right one.
|
|
20
|
+
if got.Model != "claude-sonnet-5" {
|
|
21
|
+
t.Fatalf("model = %q, want claude-sonnet-5", got.Model)
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
func TestParseClaudeEnvelopeFallsBackToRawText(t *testing.T) {
|
|
26
|
+
raw := `plain text, not an envelope {"ok":true}`
|
|
27
|
+
got := parseClaudeEnvelope(raw)
|
|
28
|
+
if got.Text != raw || got.Model != "" {
|
|
29
|
+
t.Fatalf("fallback = %#v", got)
|
|
30
|
+
}
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
func TestCodexModelReadsPreamble(t *testing.T) {
|
|
34
|
+
raw := "OpenAI Codex v0.143.0\n--------\nworkdir: /tmp\nmodel: gpt-5.6-sol\nprovider: openai\n--------\n{\"ok\":true}\n"
|
|
35
|
+
if got := codexModel(raw); got != "gpt-5.6-sol" {
|
|
36
|
+
t.Fatalf("model = %q", got)
|
|
37
|
+
}
|
|
38
|
+
if got := codexModel("no preamble at all"); got != "" {
|
|
39
|
+
t.Fatalf("expected empty model, got %q", got)
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
func TestCodexExecArgsExcludeRemovedFeatures(t *testing.T) {
|
|
44
|
+
args := codexExecArgs("/tmp/judge")
|
|
45
|
+
if slices.Contains(args, "browser_use_full_cdp_access") {
|
|
46
|
+
t.Fatal("codex args include removed browser_use_full_cdp_access feature")
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
func TestTruncateFailureDetailPreservesUTF8(t *testing.T) {
|
|
51
|
+
detail := strings.Repeat("a", 499) + "界tail"
|
|
52
|
+
got := truncateFailureDetail(detail)
|
|
53
|
+
|
|
54
|
+
if !utf8.ValidString(got) {
|
|
55
|
+
t.Fatalf("truncated detail is not valid UTF-8: %q", got)
|
|
56
|
+
}
|
|
57
|
+
want := strings.Repeat("a", 499) + "界"
|
|
58
|
+
if got != want {
|
|
59
|
+
t.Fatalf("truncated detail = %q, want %q", got, want)
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
func TestTruncateFailureDetailRepairsInvalidUTF8(t *testing.T) {
|
|
64
|
+
got := truncateFailureDetail(string([]byte{'o', 'k', 0xff}))
|
|
65
|
+
if !utf8.ValidString(got) {
|
|
66
|
+
t.Fatalf("failure detail contains invalid UTF-8: %q", got)
|
|
67
|
+
}
|
|
68
|
+
}
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
package judge
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"testing"
|
|
5
|
+
|
|
6
|
+
"github.com/hybridlabor-api/bdb-synapse/internal/model"
|
|
7
|
+
)
|
|
8
|
+
|
|
9
|
+
func TestFreshAgainstSummary(t *testing.T) {
|
|
10
|
+
report := &model.Report{
|
|
11
|
+
Session: model.ReportSession{EventCount: 3, UserTurns: 2},
|
|
12
|
+
Judge: model.ReportJudge{PromptVersion: PromptVersion, InputDigest: "d"},
|
|
13
|
+
}
|
|
14
|
+
meta := model.SessionMeta{EventCount: 3, UserTurns: 2}
|
|
15
|
+
if !FreshAgainstSummary(report, meta) {
|
|
16
|
+
t.Fatal("matching counts and current prompt graded stale")
|
|
17
|
+
}
|
|
18
|
+
if FreshAgainstSummary(report, model.SessionMeta{EventCount: 4, UserTurns: 2}) {
|
|
19
|
+
t.Fatal("event growth graded fresh")
|
|
20
|
+
}
|
|
21
|
+
if FreshAgainstSummary(report, model.SessionMeta{EventCount: 3, UserTurns: 3}) {
|
|
22
|
+
t.Fatal("message-only growth graded fresh")
|
|
23
|
+
}
|
|
24
|
+
noDigest := *report
|
|
25
|
+
noDigest.Judge.InputDigest = ""
|
|
26
|
+
if FreshAgainstSummary(&noDigest, meta) {
|
|
27
|
+
t.Fatal("pre-digest report graded fresh; the panel would disagree")
|
|
28
|
+
}
|
|
29
|
+
oldPrompt := *report
|
|
30
|
+
oldPrompt.Judge.PromptVersion = PromptVersion - 1
|
|
31
|
+
if FreshAgainstSummary(&oldPrompt, meta) {
|
|
32
|
+
t.Fatal("outdated prompt graded fresh")
|
|
33
|
+
}
|
|
34
|
+
if FreshAgainstSummary(nil, meta) {
|
|
35
|
+
t.Fatal("nil report graded fresh")
|
|
36
|
+
}
|
|
37
|
+
}
|