@hybridlabor-api/bdb-synapse 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (102) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +173 -0
  3. package/bin/synapse +0 -0
  4. package/cmd/rubriceval/main.go +308 -0
  5. package/cmd/synapse/main.go +280 -0
  6. package/cmd/synapse/main_test.go +16 -0
  7. package/go.mod +7 -0
  8. package/go.sum +6 -0
  9. package/internal/adapter/adapter.go +1117 -0
  10. package/internal/adapter/adapter_test.go +518 -0
  11. package/internal/adapter/agy/adapter.go +193 -0
  12. package/internal/adapter/claudecode/adapter.go +415 -0
  13. package/internal/adapter/claudecode/adapter_test.go +260 -0
  14. package/internal/adapter/claudecode/agents.go +387 -0
  15. package/internal/adapter/claudecode/agents_test.go +480 -0
  16. package/internal/adapter/claudecode/summary_inputs_test.go +19 -0
  17. package/internal/adapter/codex/adapter.go +919 -0
  18. package/internal/adapter/codex/adapter_test.go +920 -0
  19. package/internal/adapter/codex/agents.go +401 -0
  20. package/internal/adapter/codex/agents_test.go +610 -0
  21. package/internal/adapter/codex/summary_inputs_test.go +20 -0
  22. package/internal/adapter/pi/adapter.go +483 -0
  23. package/internal/adapter/pi/adapter_test.go +517 -0
  24. package/internal/citymap/builder.go +1124 -0
  25. package/internal/citymap/builder_test.go +818 -0
  26. package/internal/judge/cache.go +180 -0
  27. package/internal/judge/cli.go +240 -0
  28. package/internal/judge/cli_test.go +68 -0
  29. package/internal/judge/fresh_summary_test.go +37 -0
  30. package/internal/judge/input.go +233 -0
  31. package/internal/judge/judge.go +416 -0
  32. package/internal/judge/judge_test.go +288 -0
  33. package/internal/judge/prompt.go +129 -0
  34. package/internal/judge/rubric.go +275 -0
  35. package/internal/judge/rubric_test.go +642 -0
  36. package/internal/model/agent.go +63 -0
  37. package/internal/model/agent_schema_test.go +86 -0
  38. package/internal/model/agent_test.go +64 -0
  39. package/internal/model/model.go +175 -0
  40. package/internal/model/report.go +166 -0
  41. package/internal/model/stats.go +151 -0
  42. package/internal/model/stats_test.go +89 -0
  43. package/internal/model/trace_schema_test.go +67 -0
  44. package/internal/server/analyze.go +286 -0
  45. package/internal/server/analyze_test.go +297 -0
  46. package/internal/server/codex_index_test.go +40 -0
  47. package/internal/server/hardening_test.go +147 -0
  48. package/internal/server/reportindex.go +91 -0
  49. package/internal/server/reportindex_test.go +116 -0
  50. package/internal/server/server.go +1099 -0
  51. package/internal/server/server_test.go +1389 -0
  52. package/internal/server/static/assets/fraunces-latin-ext-standard-italic-CGbN9UgK.woff2 +0 -0
  53. package/internal/server/static/assets/fraunces-latin-ext-standard-normal-CJcjJNj7.woff2 +0 -0
  54. package/internal/server/static/assets/fraunces-latin-standard-italic-lSdLDfvT.woff2 +0 -0
  55. package/internal/server/static/assets/fraunces-latin-standard-normal-DihXLNYH.woff2 +0 -0
  56. package/internal/server/static/assets/fraunces-vietnamese-standard-italic-DxWqP7Ku.woff2 +0 -0
  57. package/internal/server/static/assets/fraunces-vietnamese-standard-normal-Czevyj-6.woff2 +0 -0
  58. package/internal/server/static/assets/index-BNoY_BiB.css +1 -0
  59. package/internal/server/static/assets/index-C_adLrJr.js +3 -0
  60. package/internal/server/static/assets/react-gcHzaSmV.js +10 -0
  61. package/internal/server/static/assets/schibsted-grotesk-latin-ext-wght-normal-hsMS0n0O.woff2 +0 -0
  62. package/internal/server/static/assets/schibsted-grotesk-latin-wght-normal-Bb8VGrTG.woff2 +0 -0
  63. package/internal/server/static/assets/three-DnGjZfD1.js +4012 -0
  64. package/internal/server/static/index.html +26 -0
  65. package/internal/server/tracestore.go +173 -0
  66. package/internal/server/tracestore_test.go +51 -0
  67. package/internal/textutil/truncate.go +30 -0
  68. package/internal/textutil/truncate_test.go +39 -0
  69. package/package.json +35 -0
  70. package/web/e2e/agent-lens.spec.ts +688 -0
  71. package/web/index.html +23 -0
  72. package/web/package-lock.json +1933 -0
  73. package/web/package.json +33 -0
  74. package/web/playwright.config.ts +24 -0
  75. package/web/src/App.tsx +876 -0
  76. package/web/src/api/client.ts +74 -0
  77. package/web/src/main.tsx +12 -0
  78. package/web/src/playback/recorder.ts +160 -0
  79. package/web/src/playback/reducer.ts +91 -0
  80. package/web/src/scene/CityScene.tsx +638 -0
  81. package/web/src/scene/TreeScene.tsx +656 -0
  82. package/web/src/scene/dirLabels.ts +145 -0
  83. package/web/src/scene/sceneUtils.ts +144 -0
  84. package/web/src/scene/textures.ts +60 -0
  85. package/web/src/scene/trail.ts +79 -0
  86. package/web/src/scene/treeLayout.ts +169 -0
  87. package/web/src/state/filters.ts +40 -0
  88. package/web/src/state/store.ts +83 -0
  89. package/web/src/styles.css +2565 -0
  90. package/web/src/types.ts +315 -0
  91. package/web/src/ui/AgentsPanel.tsx +376 -0
  92. package/web/src/ui/Dock.tsx +104 -0
  93. package/web/src/ui/Hud.tsx +335 -0
  94. package/web/src/ui/Inspector.tsx +107 -0
  95. package/web/src/ui/LogoMark.tsx +38 -0
  96. package/web/src/ui/ReportPanel.tsx +491 -0
  97. package/web/src/ui/SessionRail.tsx +316 -0
  98. package/web/src/ui/Timeline.tsx +458 -0
  99. package/web/src/ui/ViewPanel.tsx +45 -0
  100. package/web/src/ui/shortcuts.ts +4 -0
  101. package/web/tsconfig.json +21 -0
  102. package/web/vite.config.ts +28 -0
@@ -0,0 +1,151 @@
1
+ package model
2
+
3
+ // ComputeStats derives session facts from a parsed trace. errorSignal is the
4
+ // adapter's grade for its own error detection (ObservabilityExact when the
5
+ // source log flags failures structurally, ObservabilityEstimated when they
6
+ // are inferred from output text); an empty value falls back to estimated.
7
+ func ComputeStats(trace *Trace, filesInRepo int, errorSignal string) Stats {
8
+ state := map[string]string{}
9
+ lastReadVersion := map[string]int{}
10
+ editVersion := map[string]int{}
11
+ readEvents := 0
12
+ weakReads := 0
13
+ repeatedReads := 0
14
+ errors := 0
15
+ unknownOutcomes := false
16
+ firstEdit := -1
17
+
18
+ stats := Stats{FilesInRepo: filesInRepo}
19
+
20
+ for _, event := range trace.Events {
21
+ countAction(&stats.Actions, event.Action)
22
+ if event.IsError {
23
+ errors++
24
+ countAction(&stats.Errors, event.Action)
25
+ } else if !event.OutcomeKnown {
26
+ unknownOutcomes = true
27
+ }
28
+ stats.ResultBytes += int64(event.ResultBytes)
29
+ switch event.Action {
30
+ case "verify":
31
+ stats.EditsAfterLastVerify = 0
32
+ case "edit":
33
+ stats.EditsAfterLastVerify++
34
+ }
35
+ for _, target := range event.Targets {
36
+ if target.Path == "" {
37
+ continue
38
+ }
39
+ prev := state[target.Path]
40
+ if RankTouch(target.Touch) > RankTouch(prev) {
41
+ state[target.Path] = target.Touch
42
+ }
43
+ if target.Touch == "edit" {
44
+ editVersion[target.Path]++
45
+ }
46
+ if target.Touch == "read" {
47
+ readEvents++
48
+ if target.Weak {
49
+ weakReads++
50
+ }
51
+ if version, ok := lastReadVersion[target.Path]; ok && version == editVersion[target.Path] {
52
+ repeatedReads++
53
+ }
54
+ lastReadVersion[target.Path] = editVersion[target.Path]
55
+ }
56
+ if target.Touch == "edit" && firstEdit == -1 {
57
+ firstEdit = event.Seq
58
+ }
59
+ }
60
+ }
61
+
62
+ if firstEdit >= 0 {
63
+ stats.EventsBeforeFirstEdit = firstEdit
64
+ } else {
65
+ stats.EventsBeforeFirstEdit = len(trace.Events)
66
+ }
67
+
68
+ for _, touch := range state {
69
+ switch touch {
70
+ case "edit":
71
+ stats.Edited++
72
+ stats.Fovea++
73
+ case "read":
74
+ stats.Fovea++
75
+ case "hit":
76
+ stats.Parafovea++
77
+ }
78
+ }
79
+ for _, count := range editVersion {
80
+ if count > stats.MaxEditsPerFile {
81
+ stats.MaxEditsPerFile = count
82
+ }
83
+ if count >= 3 {
84
+ stats.ChurnFiles++
85
+ }
86
+ }
87
+ for _, mark := range trace.Marks {
88
+ switch mark.Type {
89
+ case "user-message":
90
+ stats.UserTurns++
91
+ case "compaction":
92
+ stats.Compactions++
93
+ case "subagent":
94
+ stats.Subagents++
95
+ }
96
+ }
97
+ if readEvents > 0 {
98
+ stats.RegressionRate = float64(repeatedReads) / float64(readEvents)
99
+ }
100
+ if len(trace.Events) > 0 {
101
+ stats.ErrorRate = float64(errors) / float64(len(trace.Events))
102
+ }
103
+ // Weak read targets are inferred from command text, so any of them in the
104
+ // mix downgrades the re-read rate; no reads at all leaves it undefined.
105
+ switch {
106
+ case readEvents == 0:
107
+ stats.Observability.Reads = ObservabilityUnavailable
108
+ case weakReads == 0:
109
+ stats.Observability.Reads = ObservabilityExact
110
+ default:
111
+ stats.Observability.Reads = ObservabilityEstimated
112
+ }
113
+ if errorSignal == "" {
114
+ errorSignal = ObservabilityEstimated
115
+ }
116
+ if errorSignal == ObservabilityExact && unknownOutcomes {
117
+ errorSignal = ObservabilityEstimated
118
+ }
119
+ stats.Observability.Errors = errorSignal
120
+ return stats
121
+ }
122
+
123
+ func countAction(counts *ActionCounts, action string) {
124
+ switch action {
125
+ case "search":
126
+ counts.Search++
127
+ case "read":
128
+ counts.Read++
129
+ case "edit":
130
+ counts.Edit++
131
+ case "exec":
132
+ counts.Exec++
133
+ case "verify":
134
+ counts.Verify++
135
+ default:
136
+ counts.Other++
137
+ }
138
+ }
139
+
140
+ func RankTouch(touch string) int {
141
+ switch touch {
142
+ case "edit":
143
+ return 3
144
+ case "read":
145
+ return 2
146
+ case "hit":
147
+ return 1
148
+ default:
149
+ return 0
150
+ }
151
+ }
@@ -0,0 +1,89 @@
1
+ package model
2
+
3
+ import "testing"
4
+
5
+ func TestComputeStatsFactCounters(t *testing.T) {
6
+ trace := &Trace{
7
+ Events: []Event{
8
+ {Seq: 0, Action: "search", Targets: []Target{{Path: "a.go", Touch: "hit"}}, ResultBytes: 10},
9
+ {Seq: 1, Action: "read", Targets: []Target{{Path: "a.go", Touch: "read"}}, ResultBytes: 20},
10
+ {Seq: 2, Action: "edit", Targets: []Target{{Path: "a.go", Touch: "edit"}}},
11
+ {Seq: 3, Action: "edit", IsError: true, Targets: []Target{{Path: "a.go", Touch: "edit"}}},
12
+ {Seq: 4, Action: "edit", Targets: []Target{{Path: "a.go", Touch: "edit"}}},
13
+ {Seq: 5, Action: "verify"},
14
+ {Seq: 6, Action: "edit", Targets: []Target{{Path: "b.go", Touch: "edit"}}},
15
+ {Seq: 7, Action: "exec", IsError: true},
16
+ },
17
+ Marks: []Mark{
18
+ {Seq: 0, Type: "user-message"},
19
+ {Seq: 3, Type: "user-message"},
20
+ {Seq: 4, Type: "compaction"},
21
+ {Seq: 5, Type: "subagent"},
22
+ },
23
+ }
24
+
25
+ stats := ComputeStats(trace, 10, ObservabilityExact)
26
+
27
+ if stats.Actions != (ActionCounts{Search: 1, Read: 1, Edit: 4, Exec: 1, Verify: 1}) {
28
+ t.Fatalf("actions = %#v", stats.Actions)
29
+ }
30
+ if stats.Errors != (ActionCounts{Edit: 1, Exec: 1}) {
31
+ t.Fatalf("errors = %#v", stats.Errors)
32
+ }
33
+ if stats.MaxEditsPerFile != 3 || stats.ChurnFiles != 1 {
34
+ t.Fatalf("maxEditsPerFile = %d, churnFiles = %d", stats.MaxEditsPerFile, stats.ChurnFiles)
35
+ }
36
+ if stats.UserTurns != 2 || stats.Compactions != 1 || stats.Subagents != 1 {
37
+ t.Fatalf("marks = %d/%d/%d", stats.UserTurns, stats.Compactions, stats.Subagents)
38
+ }
39
+ if stats.ResultBytes != 30 {
40
+ t.Fatalf("resultBytes = %d", stats.ResultBytes)
41
+ }
42
+ if stats.EditsAfterLastVerify != 1 {
43
+ t.Fatalf("editsAfterLastVerify = %d", stats.EditsAfterLastVerify)
44
+ }
45
+ }
46
+
47
+ func TestComputeStatsEditsAfterLastVerifyWithoutVerify(t *testing.T) {
48
+ trace := &Trace{
49
+ Events: []Event{
50
+ {Seq: 0, Action: "edit", Targets: []Target{{Path: "a.go", Touch: "edit"}}},
51
+ {Seq: 1, Action: "edit", Targets: []Target{{Path: "b.go", Touch: "edit"}}},
52
+ },
53
+ }
54
+ if stats := ComputeStats(trace, 0, ObservabilityExact); stats.EditsAfterLastVerify != 2 {
55
+ t.Fatalf("editsAfterLastVerify = %d", stats.EditsAfterLastVerify)
56
+ }
57
+ }
58
+
59
+ func TestComputeStatsObservability(t *testing.T) {
60
+ strongRead := Event{Action: "read", OutcomeKnown: true, Targets: []Target{{Path: "a.go", Touch: "read"}}}
61
+ weakRead := Event{Action: "read", OutcomeKnown: true, Targets: []Target{{Path: "b.go", Touch: "read", Weak: true}}}
62
+ hitOnly := Event{Action: "search", Targets: []Target{{Path: "c.go", Touch: "hit"}}}
63
+ pending := Event{Action: "exec"}
64
+ legacyFailure := Event{Action: "exec", IsError: true}
65
+
66
+ tests := []struct {
67
+ name string
68
+ events []Event
69
+ errorSignal string
70
+ wantReads string
71
+ wantErrors string
72
+ }{
73
+ {"strong reads are exact", []Event{strongRead}, ObservabilityExact, ObservabilityExact, ObservabilityExact},
74
+ {"any weak read downgrades", []Event{strongRead, weakRead}, ObservabilityExact, ObservabilityEstimated, ObservabilityExact},
75
+ {"unknown outcome downgrades exact errors", []Event{strongRead, pending}, ObservabilityExact, ObservabilityExact, ObservabilityEstimated},
76
+ {"legacy failure remains known", []Event{legacyFailure}, ObservabilityExact, ObservabilityUnavailable, ObservabilityExact},
77
+ {"no reads is unavailable", []Event{hitOnly}, ObservabilityEstimated, ObservabilityUnavailable, ObservabilityEstimated},
78
+ {"empty error signal falls back to estimated", []Event{strongRead}, "", ObservabilityExact, ObservabilityEstimated},
79
+ }
80
+
81
+ for _, tt := range tests {
82
+ t.Run(tt.name, func(t *testing.T) {
83
+ stats := ComputeStats(&Trace{Events: tt.events}, 0, tt.errorSignal)
84
+ if stats.Observability.Reads != tt.wantReads || stats.Observability.Errors != tt.wantErrors {
85
+ t.Fatalf("observability = %#v, want reads %q errors %q", stats.Observability, tt.wantReads, tt.wantErrors)
86
+ }
87
+ })
88
+ }
89
+ }
@@ -0,0 +1,67 @@
1
+ package model
2
+
3
+ import (
4
+ "encoding/json"
5
+ "testing"
6
+
7
+ "github.com/santhosh-tekuri/jsonschema/v6"
8
+ )
9
+
10
+ func TestTraceSchemaAcceptsOutcomeCertainty(t *testing.T) {
11
+ trace := Trace{
12
+ Version: 1,
13
+ Session: TraceSession{
14
+ ID: "session",
15
+ Harness: "codex",
16
+ EventCount: 3,
17
+ },
18
+ Events: []Event{
19
+ {
20
+ Seq: 0,
21
+ Tool: "exec_command",
22
+ Action: "verify",
23
+ Targets: []Target{},
24
+ OutcomeKnown: true,
25
+ Summary: "run tests",
26
+ },
27
+ {
28
+ Seq: 1,
29
+ Tool: "exec_command",
30
+ Action: "verify",
31
+ Targets: []Target{},
32
+ IsError: true,
33
+ OutcomeKnown: true,
34
+ Summary: "tests failed",
35
+ },
36
+ {
37
+ Seq: 2,
38
+ Tool: "exec_command",
39
+ Action: "exec",
40
+ Targets: []Target{},
41
+ Summary: "still running",
42
+ },
43
+ },
44
+ Marks: []Mark{},
45
+ Stats: ComputeStats(&Trace{}, 0, ObservabilityEstimated),
46
+ }
47
+ document, err := json.Marshal(trace)
48
+ if err != nil {
49
+ t.Fatal(err)
50
+ }
51
+ var value any
52
+ if err := json.Unmarshal(document, &value); err != nil {
53
+ t.Fatal(err)
54
+ }
55
+ events := value.(map[string]any)["events"].([]any)
56
+ if _, found := events[2].(map[string]any)["outcomeKnown"]; found {
57
+ t.Fatal("unknown outcome serialized outcomeKnown")
58
+ }
59
+ compiler := jsonschema.NewCompiler()
60
+ schema, err := compiler.Compile("../../schema/trace.schema.json")
61
+ if err != nil {
62
+ t.Fatal(err)
63
+ }
64
+ if err := schema.Validate(value); err != nil {
65
+ t.Fatalf("trace with outcome certainty violates schema: %v\n%s", err, document)
66
+ }
67
+ }
@@ -0,0 +1,286 @@
1
+ package server
2
+
3
+ import (
4
+ "context"
5
+ "encoding/json"
6
+ "errors"
7
+ "fmt"
8
+ "io"
9
+ "net/http"
10
+ "slices"
11
+ "sync"
12
+
13
+ "github.com/hybridlabor-api/bdb-synapse/internal/judge"
14
+ "github.com/hybridlabor-api/bdb-synapse/internal/model"
15
+ )
16
+
17
+ // maxConcurrentJudges bounds simultaneous judge subprocesses: each one is a
18
+ // full agent-CLI run costing tokens and about a minute, and nothing stops a
19
+ // user from clicking evaluate across many sessions.
20
+ const maxConcurrentJudges = 2
21
+
22
+ // analyzeJob tracks one in-flight or finished judge run, keyed by session
23
+ // key. Evaluation only ever starts from an explicit POST — never from
24
+ // session scanning — because a judge run costs tokens and about a minute.
25
+ type analyzeJob struct {
26
+ done bool
27
+ report *model.Report
28
+ err string
29
+ // config identifies what this run was asked to produce (judge CLI, model,
30
+ // rubric on/off). A concurrent request for the same session with a
31
+ // different configuration must conflict, not silently receive this run.
32
+ config string
33
+ }
34
+
35
+ // jobConfig renders a request's evaluation configuration as the job identity.
36
+ func jobConfig(cli, model string, noRubric bool) string {
37
+ return fmt.Sprintf("cli=%s|model=%s|rubric=%t", cli, model, !noRubric)
38
+ }
39
+
40
+ type analyzeState struct {
41
+ mu sync.Mutex
42
+ jobs map[string]*analyzeJob
43
+ // active counts in-flight judge subprocesses across all sessions.
44
+ active int
45
+ // runner overrides the judge subprocess in tests; nil auto-detects a CLI.
46
+ runner judge.Runner
47
+ }
48
+
49
+ // snapshot returns a consistent copy of the session's job state. Job fields
50
+ // are written by the analyze goroutine under mu, so every read must happen
51
+ // inside the lock too — callers get a copy, never the live pointer.
52
+ func (a *analyzeState) snapshot(key string) (analyzeJob, bool) {
53
+ a.mu.Lock()
54
+ defer a.mu.Unlock()
55
+ job, ok := a.jobs[key]
56
+ if !ok {
57
+ return analyzeJob{}, false
58
+ }
59
+ return *job, true
60
+ }
61
+
62
+ // reportStateFor grades one session for the list view: "running" while a
63
+ // judge job is in flight, then "done" / "stale" / "failed". The badge uses
64
+ // the summary approximation of freshness (no trace parse, no per-session
65
+ // disk probe); the panel's FreshAgainstTrace check stays the precise one.
66
+ func (s *Server) reportStateFor(meta model.SessionMeta) string {
67
+ job, ok := s.analyze.snapshot(meta.Key)
68
+ var report *model.Report
69
+ switch {
70
+ case ok && !job.done:
71
+ return "running"
72
+ case ok && job.err != "":
73
+ return "failed"
74
+ case ok && job.report != nil:
75
+ report = job.report
76
+ default:
77
+ report = s.reportIndex.load(s.reportCache, meta.Key)
78
+ }
79
+ if report == nil {
80
+ return ""
81
+ }
82
+ if !judge.FreshAgainstSummary(report, meta) {
83
+ return "stale"
84
+ }
85
+ return "done"
86
+ }
87
+
88
+ // judgeInfo lists the judge CLIs the user can pick from, preference order
89
+ // first. A test runner narrows the list to itself.
90
+ func (s *Server) judgeInfo() ([]string, bool) {
91
+ if s.analyze.runner != nil {
92
+ return []string{s.analyze.runner.Name()}, true
93
+ }
94
+ clis := judge.DetectCLIs()
95
+ return clis, len(clis) > 0
96
+ }
97
+
98
+ type reportStatus struct {
99
+ State string `json:"state"` // none | running | done | failed
100
+ // Stale marks a done report generated from fewer events than the trace
101
+ // now has (or an older prompt); the UI offers re-evaluation.
102
+ Stale bool `json:"stale"`
103
+ Report *model.Report `json:"report,omitempty"`
104
+ Error string `json:"error,omitempty"`
105
+ JudgeAvailable bool `json:"judgeAvailable"`
106
+ // JudgeCLI is the default judge (first available); JudgeCLIs lists every
107
+ // installed CLI so the panel can offer a choice.
108
+ JudgeCLI string `json:"judgeCli,omitempty"`
109
+ JudgeCLIs []string `json:"judgeClis,omitempty"`
110
+ }
111
+
112
+ func (s *Server) handleSessionReport(w http.ResponseWriter, r *http.Request, selector string) {
113
+ if r.Method != http.MethodGet {
114
+ http.Error(w, "method not allowed", http.StatusMethodNotAllowed)
115
+ return
116
+ }
117
+ meta, err := s.findSession(selector)
118
+ if err != nil {
119
+ http.Error(w, err.Error(), http.StatusNotFound)
120
+ return
121
+ }
122
+ trace, _, err := s.traceAndMap(selector)
123
+ if err != nil {
124
+ http.Error(w, err.Error(), http.StatusNotFound)
125
+ return
126
+ }
127
+
128
+ status := reportStatus{State: "none"}
129
+ status.JudgeCLIs, status.JudgeAvailable = s.judgeInfo()
130
+ if status.JudgeAvailable {
131
+ status.JudgeCLI = status.JudgeCLIs[0]
132
+ }
133
+
134
+ job, ok := s.analyze.snapshot(meta.Key)
135
+ switch {
136
+ case ok && !job.done:
137
+ status.State = "running"
138
+ case ok && job.err != "":
139
+ status.State = "failed"
140
+ status.Error = job.err
141
+ case ok && job.report != nil:
142
+ status.State = "done"
143
+ status.Report = job.report
144
+ status.Stale = !judge.FreshAgainstTrace(job.report, trace)
145
+ default:
146
+ if cached := s.reportIndex.load(s.reportCache, meta.Key); cached != nil {
147
+ status.State = "done"
148
+ status.Report = cached
149
+ status.Stale = !judge.FreshAgainstTrace(cached, trace)
150
+ }
151
+ }
152
+ writeJSON(w, status)
153
+ }
154
+
155
+ func (s *Server) handleSessionAnalyze(w http.ResponseWriter, r *http.Request, selector string) {
156
+ if r.Method != http.MethodPost {
157
+ http.Error(w, "method not allowed", http.StatusMethodNotAllowed)
158
+ return
159
+ }
160
+ meta, err := s.findSession(selector)
161
+ if err != nil {
162
+ http.Error(w, err.Error(), http.StatusNotFound)
163
+ return
164
+ }
165
+ trace, _, err := s.traceAndMap(selector)
166
+ if err != nil {
167
+ http.Error(w, err.Error(), http.StatusNotFound)
168
+ return
169
+ }
170
+ clis, available := s.judgeInfo()
171
+ if !available {
172
+ http.Error(w, "no judge CLI found on PATH (looked for claude, codex)", http.StatusServiceUnavailable)
173
+ return
174
+ }
175
+
176
+ // Optional body: the panel's judge choice. An empty body keeps the
177
+ // default CLI and its default model; a malformed one is rejected — this
178
+ // request starts an expensive run, so a garbled choice must not silently
179
+ // fall back to defaults.
180
+ var req struct {
181
+ CLI string `json:"cli"`
182
+ Model string `json:"model"`
183
+ // Rubric false skips the task-rubric layer; absent or true keeps it.
184
+ Rubric *bool `json:"rubric"`
185
+ }
186
+ if r.Body != nil {
187
+ decoder := json.NewDecoder(r.Body)
188
+ if err := decoder.Decode(&req); err != nil {
189
+ if !errors.Is(err, io.EOF) {
190
+ http.Error(w, "invalid request body: "+err.Error(), http.StatusBadRequest)
191
+ return
192
+ }
193
+ } else if _, err := decoder.Token(); !errors.Is(err, io.EOF) {
194
+ // One JSON value and nothing after it — trailing garbage means a
195
+ // broken client, and this request starts an expensive run.
196
+ http.Error(w, "invalid request body: trailing data after JSON object", http.StatusBadRequest)
197
+ return
198
+ }
199
+ }
200
+ if req.CLI != "" && !slices.Contains(clis, req.CLI) {
201
+ http.Error(w, fmt.Sprintf("judge CLI %q is not available (installed: %v)", req.CLI, clis), http.StatusBadRequest)
202
+ return
203
+ }
204
+
205
+ noRubric := req.Rubric != nil && !*req.Rubric
206
+ config := jobConfig(req.CLI, req.Model, noRubric)
207
+ s.analyze.mu.Lock()
208
+ if job := s.analyze.jobs[meta.Key]; job != nil && !job.done {
209
+ sameConfig := job.config == config
210
+ s.analyze.mu.Unlock()
211
+ if !sameConfig {
212
+ // The API must not pretend to accept a configuration it will not
213
+ // run: the in-flight job was asked for something else.
214
+ http.Error(w, "an evaluation with a different judge configuration is already running for this session; wait for it to finish", http.StatusConflict)
215
+ return
216
+ }
217
+ w.WriteHeader(http.StatusAccepted)
218
+ writeJSON(w, reportStatus{State: "running", JudgeAvailable: true})
219
+ return
220
+ }
221
+ if s.analyze.active >= maxConcurrentJudges {
222
+ s.analyze.mu.Unlock()
223
+ http.Error(w, fmt.Sprintf("%d evaluations already running; wait for one to finish", maxConcurrentJudges), http.StatusTooManyRequests)
224
+ return
225
+ }
226
+ job := &analyzeJob{config: config}
227
+ s.analyze.jobs[meta.Key] = job
228
+ s.analyze.active++
229
+ s.analyze.mu.Unlock()
230
+
231
+ go s.runAnalyze(meta.Key, trace, job, req.CLI, req.Model, noRubric)
232
+
233
+ w.WriteHeader(http.StatusAccepted)
234
+ writeJSON(w, reportStatus{State: "running", JudgeAvailable: true})
235
+ }
236
+
237
+ func (s *Server) runAnalyze(key string, trace *model.Trace, job *analyzeJob, cli, judgeModel string, noRubric bool) {
238
+ ctx, cancel := context.WithTimeout(context.Background(), judge.DefaultTimeout)
239
+ defer cancel()
240
+ // The cached report seeds rubric reuse: a re-evaluation with unchanged
241
+ // task wording keeps the same criteria — including across judge CLIs,
242
+ // which is exactly what makes their verdicts comparable. A rubric:false
243
+ // run mirrors the CLI's --no-rubric semantics and bypasses the cache in
244
+ // both directions: it neither mines the cache nor overwrites a richer
245
+ // report with a dimensions-only one — its result lives only in the job.
246
+ var cached *model.Report
247
+ if !noRubric {
248
+ cached = s.reportCache.Load(key)
249
+ }
250
+ report, err := judge.Analyze(ctx, trace, judge.Options{
251
+ Runner: s.analyze.runner,
252
+ CLI: cli,
253
+ Model: judgeModel,
254
+ NoRubric: noRubric,
255
+ CachedReport: cached,
256
+ })
257
+
258
+ // Persist before publishing done, and outside the lock: once the job entry
259
+ // is dropped, polls must be able to find the report on disk.
260
+ persisted := false
261
+ if err == nil && !noRubric && s.reportCache.Dir != "" {
262
+ persisted = s.reportCache.Store(key, report) == nil
263
+ }
264
+ if persisted {
265
+ // Polls landing between the store and the index's next directory scan
266
+ // must find the report once the job entry is dropped below.
267
+ s.reportIndex.markPresent(key)
268
+ }
269
+
270
+ s.analyze.mu.Lock()
271
+ defer s.analyze.mu.Unlock()
272
+ s.analyze.active--
273
+ job.done = true
274
+ if err != nil {
275
+ job.err = err.Error()
276
+ return
277
+ }
278
+ job.report = report
279
+ if persisted {
280
+ // The cache owns the report now; dropping the entry keeps the jobs map
281
+ // bounded. When the disk write failed the entry stays as the only copy —
282
+ // losing it would cost a re-run — and failed jobs stay too (small, and
283
+ // the UI needs the error until a re-run replaces them).
284
+ delete(s.analyze.jobs, key)
285
+ }
286
+ }