@hybridlabor-api/bdb-synapse 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +173 -0
- package/bin/synapse +0 -0
- package/cmd/rubriceval/main.go +308 -0
- package/cmd/synapse/main.go +280 -0
- package/cmd/synapse/main_test.go +16 -0
- package/go.mod +7 -0
- package/go.sum +6 -0
- package/internal/adapter/adapter.go +1117 -0
- package/internal/adapter/adapter_test.go +518 -0
- package/internal/adapter/agy/adapter.go +193 -0
- package/internal/adapter/claudecode/adapter.go +415 -0
- package/internal/adapter/claudecode/adapter_test.go +260 -0
- package/internal/adapter/claudecode/agents.go +387 -0
- package/internal/adapter/claudecode/agents_test.go +480 -0
- package/internal/adapter/claudecode/summary_inputs_test.go +19 -0
- package/internal/adapter/codex/adapter.go +919 -0
- package/internal/adapter/codex/adapter_test.go +920 -0
- package/internal/adapter/codex/agents.go +401 -0
- package/internal/adapter/codex/agents_test.go +610 -0
- package/internal/adapter/codex/summary_inputs_test.go +20 -0
- package/internal/adapter/pi/adapter.go +483 -0
- package/internal/adapter/pi/adapter_test.go +517 -0
- package/internal/citymap/builder.go +1124 -0
- package/internal/citymap/builder_test.go +818 -0
- package/internal/judge/cache.go +180 -0
- package/internal/judge/cli.go +240 -0
- package/internal/judge/cli_test.go +68 -0
- package/internal/judge/fresh_summary_test.go +37 -0
- package/internal/judge/input.go +233 -0
- package/internal/judge/judge.go +416 -0
- package/internal/judge/judge_test.go +288 -0
- package/internal/judge/prompt.go +129 -0
- package/internal/judge/rubric.go +275 -0
- package/internal/judge/rubric_test.go +642 -0
- package/internal/model/agent.go +63 -0
- package/internal/model/agent_schema_test.go +86 -0
- package/internal/model/agent_test.go +64 -0
- package/internal/model/model.go +175 -0
- package/internal/model/report.go +166 -0
- package/internal/model/stats.go +151 -0
- package/internal/model/stats_test.go +89 -0
- package/internal/model/trace_schema_test.go +67 -0
- package/internal/server/analyze.go +286 -0
- package/internal/server/analyze_test.go +297 -0
- package/internal/server/codex_index_test.go +40 -0
- package/internal/server/hardening_test.go +147 -0
- package/internal/server/reportindex.go +91 -0
- package/internal/server/reportindex_test.go +116 -0
- package/internal/server/server.go +1099 -0
- package/internal/server/server_test.go +1389 -0
- package/internal/server/static/assets/fraunces-latin-ext-standard-italic-CGbN9UgK.woff2 +0 -0
- package/internal/server/static/assets/fraunces-latin-ext-standard-normal-CJcjJNj7.woff2 +0 -0
- package/internal/server/static/assets/fraunces-latin-standard-italic-lSdLDfvT.woff2 +0 -0
- package/internal/server/static/assets/fraunces-latin-standard-normal-DihXLNYH.woff2 +0 -0
- package/internal/server/static/assets/fraunces-vietnamese-standard-italic-DxWqP7Ku.woff2 +0 -0
- package/internal/server/static/assets/fraunces-vietnamese-standard-normal-Czevyj-6.woff2 +0 -0
- package/internal/server/static/assets/index-BNoY_BiB.css +1 -0
- package/internal/server/static/assets/index-C_adLrJr.js +3 -0
- package/internal/server/static/assets/react-gcHzaSmV.js +10 -0
- package/internal/server/static/assets/schibsted-grotesk-latin-ext-wght-normal-hsMS0n0O.woff2 +0 -0
- package/internal/server/static/assets/schibsted-grotesk-latin-wght-normal-Bb8VGrTG.woff2 +0 -0
- package/internal/server/static/assets/three-DnGjZfD1.js +4012 -0
- package/internal/server/static/index.html +26 -0
- package/internal/server/tracestore.go +173 -0
- package/internal/server/tracestore_test.go +51 -0
- package/internal/textutil/truncate.go +30 -0
- package/internal/textutil/truncate_test.go +39 -0
- package/package.json +35 -0
- package/web/e2e/agent-lens.spec.ts +688 -0
- package/web/index.html +23 -0
- package/web/package-lock.json +1933 -0
- package/web/package.json +33 -0
- package/web/playwright.config.ts +24 -0
- package/web/src/App.tsx +876 -0
- package/web/src/api/client.ts +74 -0
- package/web/src/main.tsx +12 -0
- package/web/src/playback/recorder.ts +160 -0
- package/web/src/playback/reducer.ts +91 -0
- package/web/src/scene/CityScene.tsx +638 -0
- package/web/src/scene/TreeScene.tsx +656 -0
- package/web/src/scene/dirLabels.ts +145 -0
- package/web/src/scene/sceneUtils.ts +144 -0
- package/web/src/scene/textures.ts +60 -0
- package/web/src/scene/trail.ts +79 -0
- package/web/src/scene/treeLayout.ts +169 -0
- package/web/src/state/filters.ts +40 -0
- package/web/src/state/store.ts +83 -0
- package/web/src/styles.css +2565 -0
- package/web/src/types.ts +315 -0
- package/web/src/ui/AgentsPanel.tsx +376 -0
- package/web/src/ui/Dock.tsx +104 -0
- package/web/src/ui/Hud.tsx +335 -0
- package/web/src/ui/Inspector.tsx +107 -0
- package/web/src/ui/LogoMark.tsx +38 -0
- package/web/src/ui/ReportPanel.tsx +491 -0
- package/web/src/ui/SessionRail.tsx +316 -0
- package/web/src/ui/Timeline.tsx +458 -0
- package/web/src/ui/ViewPanel.tsx +45 -0
- package/web/src/ui/shortcuts.ts +4 -0
- package/web/tsconfig.json +21 -0
- package/web/vite.config.ts +28 -0
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
package model
|
|
2
|
+
|
|
3
|
+
// ComputeStats derives session facts from a parsed trace. errorSignal is the
|
|
4
|
+
// adapter's grade for its own error detection (ObservabilityExact when the
|
|
5
|
+
// source log flags failures structurally, ObservabilityEstimated when they
|
|
6
|
+
// are inferred from output text); an empty value falls back to estimated.
|
|
7
|
+
func ComputeStats(trace *Trace, filesInRepo int, errorSignal string) Stats {
|
|
8
|
+
state := map[string]string{}
|
|
9
|
+
lastReadVersion := map[string]int{}
|
|
10
|
+
editVersion := map[string]int{}
|
|
11
|
+
readEvents := 0
|
|
12
|
+
weakReads := 0
|
|
13
|
+
repeatedReads := 0
|
|
14
|
+
errors := 0
|
|
15
|
+
unknownOutcomes := false
|
|
16
|
+
firstEdit := -1
|
|
17
|
+
|
|
18
|
+
stats := Stats{FilesInRepo: filesInRepo}
|
|
19
|
+
|
|
20
|
+
for _, event := range trace.Events {
|
|
21
|
+
countAction(&stats.Actions, event.Action)
|
|
22
|
+
if event.IsError {
|
|
23
|
+
errors++
|
|
24
|
+
countAction(&stats.Errors, event.Action)
|
|
25
|
+
} else if !event.OutcomeKnown {
|
|
26
|
+
unknownOutcomes = true
|
|
27
|
+
}
|
|
28
|
+
stats.ResultBytes += int64(event.ResultBytes)
|
|
29
|
+
switch event.Action {
|
|
30
|
+
case "verify":
|
|
31
|
+
stats.EditsAfterLastVerify = 0
|
|
32
|
+
case "edit":
|
|
33
|
+
stats.EditsAfterLastVerify++
|
|
34
|
+
}
|
|
35
|
+
for _, target := range event.Targets {
|
|
36
|
+
if target.Path == "" {
|
|
37
|
+
continue
|
|
38
|
+
}
|
|
39
|
+
prev := state[target.Path]
|
|
40
|
+
if RankTouch(target.Touch) > RankTouch(prev) {
|
|
41
|
+
state[target.Path] = target.Touch
|
|
42
|
+
}
|
|
43
|
+
if target.Touch == "edit" {
|
|
44
|
+
editVersion[target.Path]++
|
|
45
|
+
}
|
|
46
|
+
if target.Touch == "read" {
|
|
47
|
+
readEvents++
|
|
48
|
+
if target.Weak {
|
|
49
|
+
weakReads++
|
|
50
|
+
}
|
|
51
|
+
if version, ok := lastReadVersion[target.Path]; ok && version == editVersion[target.Path] {
|
|
52
|
+
repeatedReads++
|
|
53
|
+
}
|
|
54
|
+
lastReadVersion[target.Path] = editVersion[target.Path]
|
|
55
|
+
}
|
|
56
|
+
if target.Touch == "edit" && firstEdit == -1 {
|
|
57
|
+
firstEdit = event.Seq
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
if firstEdit >= 0 {
|
|
63
|
+
stats.EventsBeforeFirstEdit = firstEdit
|
|
64
|
+
} else {
|
|
65
|
+
stats.EventsBeforeFirstEdit = len(trace.Events)
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
for _, touch := range state {
|
|
69
|
+
switch touch {
|
|
70
|
+
case "edit":
|
|
71
|
+
stats.Edited++
|
|
72
|
+
stats.Fovea++
|
|
73
|
+
case "read":
|
|
74
|
+
stats.Fovea++
|
|
75
|
+
case "hit":
|
|
76
|
+
stats.Parafovea++
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
for _, count := range editVersion {
|
|
80
|
+
if count > stats.MaxEditsPerFile {
|
|
81
|
+
stats.MaxEditsPerFile = count
|
|
82
|
+
}
|
|
83
|
+
if count >= 3 {
|
|
84
|
+
stats.ChurnFiles++
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
for _, mark := range trace.Marks {
|
|
88
|
+
switch mark.Type {
|
|
89
|
+
case "user-message":
|
|
90
|
+
stats.UserTurns++
|
|
91
|
+
case "compaction":
|
|
92
|
+
stats.Compactions++
|
|
93
|
+
case "subagent":
|
|
94
|
+
stats.Subagents++
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
if readEvents > 0 {
|
|
98
|
+
stats.RegressionRate = float64(repeatedReads) / float64(readEvents)
|
|
99
|
+
}
|
|
100
|
+
if len(trace.Events) > 0 {
|
|
101
|
+
stats.ErrorRate = float64(errors) / float64(len(trace.Events))
|
|
102
|
+
}
|
|
103
|
+
// Weak read targets are inferred from command text, so any of them in the
|
|
104
|
+
// mix downgrades the re-read rate; no reads at all leaves it undefined.
|
|
105
|
+
switch {
|
|
106
|
+
case readEvents == 0:
|
|
107
|
+
stats.Observability.Reads = ObservabilityUnavailable
|
|
108
|
+
case weakReads == 0:
|
|
109
|
+
stats.Observability.Reads = ObservabilityExact
|
|
110
|
+
default:
|
|
111
|
+
stats.Observability.Reads = ObservabilityEstimated
|
|
112
|
+
}
|
|
113
|
+
if errorSignal == "" {
|
|
114
|
+
errorSignal = ObservabilityEstimated
|
|
115
|
+
}
|
|
116
|
+
if errorSignal == ObservabilityExact && unknownOutcomes {
|
|
117
|
+
errorSignal = ObservabilityEstimated
|
|
118
|
+
}
|
|
119
|
+
stats.Observability.Errors = errorSignal
|
|
120
|
+
return stats
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
func countAction(counts *ActionCounts, action string) {
|
|
124
|
+
switch action {
|
|
125
|
+
case "search":
|
|
126
|
+
counts.Search++
|
|
127
|
+
case "read":
|
|
128
|
+
counts.Read++
|
|
129
|
+
case "edit":
|
|
130
|
+
counts.Edit++
|
|
131
|
+
case "exec":
|
|
132
|
+
counts.Exec++
|
|
133
|
+
case "verify":
|
|
134
|
+
counts.Verify++
|
|
135
|
+
default:
|
|
136
|
+
counts.Other++
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
func RankTouch(touch string) int {
|
|
141
|
+
switch touch {
|
|
142
|
+
case "edit":
|
|
143
|
+
return 3
|
|
144
|
+
case "read":
|
|
145
|
+
return 2
|
|
146
|
+
case "hit":
|
|
147
|
+
return 1
|
|
148
|
+
default:
|
|
149
|
+
return 0
|
|
150
|
+
}
|
|
151
|
+
}
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
package model
|
|
2
|
+
|
|
3
|
+
import "testing"
|
|
4
|
+
|
|
5
|
+
func TestComputeStatsFactCounters(t *testing.T) {
|
|
6
|
+
trace := &Trace{
|
|
7
|
+
Events: []Event{
|
|
8
|
+
{Seq: 0, Action: "search", Targets: []Target{{Path: "a.go", Touch: "hit"}}, ResultBytes: 10},
|
|
9
|
+
{Seq: 1, Action: "read", Targets: []Target{{Path: "a.go", Touch: "read"}}, ResultBytes: 20},
|
|
10
|
+
{Seq: 2, Action: "edit", Targets: []Target{{Path: "a.go", Touch: "edit"}}},
|
|
11
|
+
{Seq: 3, Action: "edit", IsError: true, Targets: []Target{{Path: "a.go", Touch: "edit"}}},
|
|
12
|
+
{Seq: 4, Action: "edit", Targets: []Target{{Path: "a.go", Touch: "edit"}}},
|
|
13
|
+
{Seq: 5, Action: "verify"},
|
|
14
|
+
{Seq: 6, Action: "edit", Targets: []Target{{Path: "b.go", Touch: "edit"}}},
|
|
15
|
+
{Seq: 7, Action: "exec", IsError: true},
|
|
16
|
+
},
|
|
17
|
+
Marks: []Mark{
|
|
18
|
+
{Seq: 0, Type: "user-message"},
|
|
19
|
+
{Seq: 3, Type: "user-message"},
|
|
20
|
+
{Seq: 4, Type: "compaction"},
|
|
21
|
+
{Seq: 5, Type: "subagent"},
|
|
22
|
+
},
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
stats := ComputeStats(trace, 10, ObservabilityExact)
|
|
26
|
+
|
|
27
|
+
if stats.Actions != (ActionCounts{Search: 1, Read: 1, Edit: 4, Exec: 1, Verify: 1}) {
|
|
28
|
+
t.Fatalf("actions = %#v", stats.Actions)
|
|
29
|
+
}
|
|
30
|
+
if stats.Errors != (ActionCounts{Edit: 1, Exec: 1}) {
|
|
31
|
+
t.Fatalf("errors = %#v", stats.Errors)
|
|
32
|
+
}
|
|
33
|
+
if stats.MaxEditsPerFile != 3 || stats.ChurnFiles != 1 {
|
|
34
|
+
t.Fatalf("maxEditsPerFile = %d, churnFiles = %d", stats.MaxEditsPerFile, stats.ChurnFiles)
|
|
35
|
+
}
|
|
36
|
+
if stats.UserTurns != 2 || stats.Compactions != 1 || stats.Subagents != 1 {
|
|
37
|
+
t.Fatalf("marks = %d/%d/%d", stats.UserTurns, stats.Compactions, stats.Subagents)
|
|
38
|
+
}
|
|
39
|
+
if stats.ResultBytes != 30 {
|
|
40
|
+
t.Fatalf("resultBytes = %d", stats.ResultBytes)
|
|
41
|
+
}
|
|
42
|
+
if stats.EditsAfterLastVerify != 1 {
|
|
43
|
+
t.Fatalf("editsAfterLastVerify = %d", stats.EditsAfterLastVerify)
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
func TestComputeStatsEditsAfterLastVerifyWithoutVerify(t *testing.T) {
|
|
48
|
+
trace := &Trace{
|
|
49
|
+
Events: []Event{
|
|
50
|
+
{Seq: 0, Action: "edit", Targets: []Target{{Path: "a.go", Touch: "edit"}}},
|
|
51
|
+
{Seq: 1, Action: "edit", Targets: []Target{{Path: "b.go", Touch: "edit"}}},
|
|
52
|
+
},
|
|
53
|
+
}
|
|
54
|
+
if stats := ComputeStats(trace, 0, ObservabilityExact); stats.EditsAfterLastVerify != 2 {
|
|
55
|
+
t.Fatalf("editsAfterLastVerify = %d", stats.EditsAfterLastVerify)
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
func TestComputeStatsObservability(t *testing.T) {
|
|
60
|
+
strongRead := Event{Action: "read", OutcomeKnown: true, Targets: []Target{{Path: "a.go", Touch: "read"}}}
|
|
61
|
+
weakRead := Event{Action: "read", OutcomeKnown: true, Targets: []Target{{Path: "b.go", Touch: "read", Weak: true}}}
|
|
62
|
+
hitOnly := Event{Action: "search", Targets: []Target{{Path: "c.go", Touch: "hit"}}}
|
|
63
|
+
pending := Event{Action: "exec"}
|
|
64
|
+
legacyFailure := Event{Action: "exec", IsError: true}
|
|
65
|
+
|
|
66
|
+
tests := []struct {
|
|
67
|
+
name string
|
|
68
|
+
events []Event
|
|
69
|
+
errorSignal string
|
|
70
|
+
wantReads string
|
|
71
|
+
wantErrors string
|
|
72
|
+
}{
|
|
73
|
+
{"strong reads are exact", []Event{strongRead}, ObservabilityExact, ObservabilityExact, ObservabilityExact},
|
|
74
|
+
{"any weak read downgrades", []Event{strongRead, weakRead}, ObservabilityExact, ObservabilityEstimated, ObservabilityExact},
|
|
75
|
+
{"unknown outcome downgrades exact errors", []Event{strongRead, pending}, ObservabilityExact, ObservabilityExact, ObservabilityEstimated},
|
|
76
|
+
{"legacy failure remains known", []Event{legacyFailure}, ObservabilityExact, ObservabilityUnavailable, ObservabilityExact},
|
|
77
|
+
{"no reads is unavailable", []Event{hitOnly}, ObservabilityEstimated, ObservabilityUnavailable, ObservabilityEstimated},
|
|
78
|
+
{"empty error signal falls back to estimated", []Event{strongRead}, "", ObservabilityExact, ObservabilityEstimated},
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
for _, tt := range tests {
|
|
82
|
+
t.Run(tt.name, func(t *testing.T) {
|
|
83
|
+
stats := ComputeStats(&Trace{Events: tt.events}, 0, tt.errorSignal)
|
|
84
|
+
if stats.Observability.Reads != tt.wantReads || stats.Observability.Errors != tt.wantErrors {
|
|
85
|
+
t.Fatalf("observability = %#v, want reads %q errors %q", stats.Observability, tt.wantReads, tt.wantErrors)
|
|
86
|
+
}
|
|
87
|
+
})
|
|
88
|
+
}
|
|
89
|
+
}
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
package model
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"encoding/json"
|
|
5
|
+
"testing"
|
|
6
|
+
|
|
7
|
+
"github.com/santhosh-tekuri/jsonschema/v6"
|
|
8
|
+
)
|
|
9
|
+
|
|
10
|
+
func TestTraceSchemaAcceptsOutcomeCertainty(t *testing.T) {
|
|
11
|
+
trace := Trace{
|
|
12
|
+
Version: 1,
|
|
13
|
+
Session: TraceSession{
|
|
14
|
+
ID: "session",
|
|
15
|
+
Harness: "codex",
|
|
16
|
+
EventCount: 3,
|
|
17
|
+
},
|
|
18
|
+
Events: []Event{
|
|
19
|
+
{
|
|
20
|
+
Seq: 0,
|
|
21
|
+
Tool: "exec_command",
|
|
22
|
+
Action: "verify",
|
|
23
|
+
Targets: []Target{},
|
|
24
|
+
OutcomeKnown: true,
|
|
25
|
+
Summary: "run tests",
|
|
26
|
+
},
|
|
27
|
+
{
|
|
28
|
+
Seq: 1,
|
|
29
|
+
Tool: "exec_command",
|
|
30
|
+
Action: "verify",
|
|
31
|
+
Targets: []Target{},
|
|
32
|
+
IsError: true,
|
|
33
|
+
OutcomeKnown: true,
|
|
34
|
+
Summary: "tests failed",
|
|
35
|
+
},
|
|
36
|
+
{
|
|
37
|
+
Seq: 2,
|
|
38
|
+
Tool: "exec_command",
|
|
39
|
+
Action: "exec",
|
|
40
|
+
Targets: []Target{},
|
|
41
|
+
Summary: "still running",
|
|
42
|
+
},
|
|
43
|
+
},
|
|
44
|
+
Marks: []Mark{},
|
|
45
|
+
Stats: ComputeStats(&Trace{}, 0, ObservabilityEstimated),
|
|
46
|
+
}
|
|
47
|
+
document, err := json.Marshal(trace)
|
|
48
|
+
if err != nil {
|
|
49
|
+
t.Fatal(err)
|
|
50
|
+
}
|
|
51
|
+
var value any
|
|
52
|
+
if err := json.Unmarshal(document, &value); err != nil {
|
|
53
|
+
t.Fatal(err)
|
|
54
|
+
}
|
|
55
|
+
events := value.(map[string]any)["events"].([]any)
|
|
56
|
+
if _, found := events[2].(map[string]any)["outcomeKnown"]; found {
|
|
57
|
+
t.Fatal("unknown outcome serialized outcomeKnown")
|
|
58
|
+
}
|
|
59
|
+
compiler := jsonschema.NewCompiler()
|
|
60
|
+
schema, err := compiler.Compile("../../schema/trace.schema.json")
|
|
61
|
+
if err != nil {
|
|
62
|
+
t.Fatal(err)
|
|
63
|
+
}
|
|
64
|
+
if err := schema.Validate(value); err != nil {
|
|
65
|
+
t.Fatalf("trace with outcome certainty violates schema: %v\n%s", err, document)
|
|
66
|
+
}
|
|
67
|
+
}
|
|
@@ -0,0 +1,286 @@
|
|
|
1
|
+
package server
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"context"
|
|
5
|
+
"encoding/json"
|
|
6
|
+
"errors"
|
|
7
|
+
"fmt"
|
|
8
|
+
"io"
|
|
9
|
+
"net/http"
|
|
10
|
+
"slices"
|
|
11
|
+
"sync"
|
|
12
|
+
|
|
13
|
+
"github.com/hybridlabor-api/bdb-synapse/internal/judge"
|
|
14
|
+
"github.com/hybridlabor-api/bdb-synapse/internal/model"
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
// maxConcurrentJudges bounds simultaneous judge subprocesses: each one is a
|
|
18
|
+
// full agent-CLI run costing tokens and about a minute, and nothing stops a
|
|
19
|
+
// user from clicking evaluate across many sessions.
|
|
20
|
+
const maxConcurrentJudges = 2
|
|
21
|
+
|
|
22
|
+
// analyzeJob tracks one in-flight or finished judge run, keyed by session
|
|
23
|
+
// key. Evaluation only ever starts from an explicit POST — never from
|
|
24
|
+
// session scanning — because a judge run costs tokens and about a minute.
|
|
25
|
+
type analyzeJob struct {
|
|
26
|
+
done bool
|
|
27
|
+
report *model.Report
|
|
28
|
+
err string
|
|
29
|
+
// config identifies what this run was asked to produce (judge CLI, model,
|
|
30
|
+
// rubric on/off). A concurrent request for the same session with a
|
|
31
|
+
// different configuration must conflict, not silently receive this run.
|
|
32
|
+
config string
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
// jobConfig renders a request's evaluation configuration as the job identity.
|
|
36
|
+
func jobConfig(cli, model string, noRubric bool) string {
|
|
37
|
+
return fmt.Sprintf("cli=%s|model=%s|rubric=%t", cli, model, !noRubric)
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
type analyzeState struct {
|
|
41
|
+
mu sync.Mutex
|
|
42
|
+
jobs map[string]*analyzeJob
|
|
43
|
+
// active counts in-flight judge subprocesses across all sessions.
|
|
44
|
+
active int
|
|
45
|
+
// runner overrides the judge subprocess in tests; nil auto-detects a CLI.
|
|
46
|
+
runner judge.Runner
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
// snapshot returns a consistent copy of the session's job state. Job fields
|
|
50
|
+
// are written by the analyze goroutine under mu, so every read must happen
|
|
51
|
+
// inside the lock too — callers get a copy, never the live pointer.
|
|
52
|
+
func (a *analyzeState) snapshot(key string) (analyzeJob, bool) {
|
|
53
|
+
a.mu.Lock()
|
|
54
|
+
defer a.mu.Unlock()
|
|
55
|
+
job, ok := a.jobs[key]
|
|
56
|
+
if !ok {
|
|
57
|
+
return analyzeJob{}, false
|
|
58
|
+
}
|
|
59
|
+
return *job, true
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
// reportStateFor grades one session for the list view: "running" while a
|
|
63
|
+
// judge job is in flight, then "done" / "stale" / "failed". The badge uses
|
|
64
|
+
// the summary approximation of freshness (no trace parse, no per-session
|
|
65
|
+
// disk probe); the panel's FreshAgainstTrace check stays the precise one.
|
|
66
|
+
func (s *Server) reportStateFor(meta model.SessionMeta) string {
|
|
67
|
+
job, ok := s.analyze.snapshot(meta.Key)
|
|
68
|
+
var report *model.Report
|
|
69
|
+
switch {
|
|
70
|
+
case ok && !job.done:
|
|
71
|
+
return "running"
|
|
72
|
+
case ok && job.err != "":
|
|
73
|
+
return "failed"
|
|
74
|
+
case ok && job.report != nil:
|
|
75
|
+
report = job.report
|
|
76
|
+
default:
|
|
77
|
+
report = s.reportIndex.load(s.reportCache, meta.Key)
|
|
78
|
+
}
|
|
79
|
+
if report == nil {
|
|
80
|
+
return ""
|
|
81
|
+
}
|
|
82
|
+
if !judge.FreshAgainstSummary(report, meta) {
|
|
83
|
+
return "stale"
|
|
84
|
+
}
|
|
85
|
+
return "done"
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
// judgeInfo lists the judge CLIs the user can pick from, preference order
|
|
89
|
+
// first. A test runner narrows the list to itself.
|
|
90
|
+
func (s *Server) judgeInfo() ([]string, bool) {
|
|
91
|
+
if s.analyze.runner != nil {
|
|
92
|
+
return []string{s.analyze.runner.Name()}, true
|
|
93
|
+
}
|
|
94
|
+
clis := judge.DetectCLIs()
|
|
95
|
+
return clis, len(clis) > 0
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
type reportStatus struct {
|
|
99
|
+
State string `json:"state"` // none | running | done | failed
|
|
100
|
+
// Stale marks a done report generated from fewer events than the trace
|
|
101
|
+
// now has (or an older prompt); the UI offers re-evaluation.
|
|
102
|
+
Stale bool `json:"stale"`
|
|
103
|
+
Report *model.Report `json:"report,omitempty"`
|
|
104
|
+
Error string `json:"error,omitempty"`
|
|
105
|
+
JudgeAvailable bool `json:"judgeAvailable"`
|
|
106
|
+
// JudgeCLI is the default judge (first available); JudgeCLIs lists every
|
|
107
|
+
// installed CLI so the panel can offer a choice.
|
|
108
|
+
JudgeCLI string `json:"judgeCli,omitempty"`
|
|
109
|
+
JudgeCLIs []string `json:"judgeClis,omitempty"`
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
func (s *Server) handleSessionReport(w http.ResponseWriter, r *http.Request, selector string) {
|
|
113
|
+
if r.Method != http.MethodGet {
|
|
114
|
+
http.Error(w, "method not allowed", http.StatusMethodNotAllowed)
|
|
115
|
+
return
|
|
116
|
+
}
|
|
117
|
+
meta, err := s.findSession(selector)
|
|
118
|
+
if err != nil {
|
|
119
|
+
http.Error(w, err.Error(), http.StatusNotFound)
|
|
120
|
+
return
|
|
121
|
+
}
|
|
122
|
+
trace, _, err := s.traceAndMap(selector)
|
|
123
|
+
if err != nil {
|
|
124
|
+
http.Error(w, err.Error(), http.StatusNotFound)
|
|
125
|
+
return
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
status := reportStatus{State: "none"}
|
|
129
|
+
status.JudgeCLIs, status.JudgeAvailable = s.judgeInfo()
|
|
130
|
+
if status.JudgeAvailable {
|
|
131
|
+
status.JudgeCLI = status.JudgeCLIs[0]
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
job, ok := s.analyze.snapshot(meta.Key)
|
|
135
|
+
switch {
|
|
136
|
+
case ok && !job.done:
|
|
137
|
+
status.State = "running"
|
|
138
|
+
case ok && job.err != "":
|
|
139
|
+
status.State = "failed"
|
|
140
|
+
status.Error = job.err
|
|
141
|
+
case ok && job.report != nil:
|
|
142
|
+
status.State = "done"
|
|
143
|
+
status.Report = job.report
|
|
144
|
+
status.Stale = !judge.FreshAgainstTrace(job.report, trace)
|
|
145
|
+
default:
|
|
146
|
+
if cached := s.reportIndex.load(s.reportCache, meta.Key); cached != nil {
|
|
147
|
+
status.State = "done"
|
|
148
|
+
status.Report = cached
|
|
149
|
+
status.Stale = !judge.FreshAgainstTrace(cached, trace)
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
writeJSON(w, status)
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
func (s *Server) handleSessionAnalyze(w http.ResponseWriter, r *http.Request, selector string) {
|
|
156
|
+
if r.Method != http.MethodPost {
|
|
157
|
+
http.Error(w, "method not allowed", http.StatusMethodNotAllowed)
|
|
158
|
+
return
|
|
159
|
+
}
|
|
160
|
+
meta, err := s.findSession(selector)
|
|
161
|
+
if err != nil {
|
|
162
|
+
http.Error(w, err.Error(), http.StatusNotFound)
|
|
163
|
+
return
|
|
164
|
+
}
|
|
165
|
+
trace, _, err := s.traceAndMap(selector)
|
|
166
|
+
if err != nil {
|
|
167
|
+
http.Error(w, err.Error(), http.StatusNotFound)
|
|
168
|
+
return
|
|
169
|
+
}
|
|
170
|
+
clis, available := s.judgeInfo()
|
|
171
|
+
if !available {
|
|
172
|
+
http.Error(w, "no judge CLI found on PATH (looked for claude, codex)", http.StatusServiceUnavailable)
|
|
173
|
+
return
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
// Optional body: the panel's judge choice. An empty body keeps the
|
|
177
|
+
// default CLI and its default model; a malformed one is rejected — this
|
|
178
|
+
// request starts an expensive run, so a garbled choice must not silently
|
|
179
|
+
// fall back to defaults.
|
|
180
|
+
var req struct {
|
|
181
|
+
CLI string `json:"cli"`
|
|
182
|
+
Model string `json:"model"`
|
|
183
|
+
// Rubric false skips the task-rubric layer; absent or true keeps it.
|
|
184
|
+
Rubric *bool `json:"rubric"`
|
|
185
|
+
}
|
|
186
|
+
if r.Body != nil {
|
|
187
|
+
decoder := json.NewDecoder(r.Body)
|
|
188
|
+
if err := decoder.Decode(&req); err != nil {
|
|
189
|
+
if !errors.Is(err, io.EOF) {
|
|
190
|
+
http.Error(w, "invalid request body: "+err.Error(), http.StatusBadRequest)
|
|
191
|
+
return
|
|
192
|
+
}
|
|
193
|
+
} else if _, err := decoder.Token(); !errors.Is(err, io.EOF) {
|
|
194
|
+
// One JSON value and nothing after it — trailing garbage means a
|
|
195
|
+
// broken client, and this request starts an expensive run.
|
|
196
|
+
http.Error(w, "invalid request body: trailing data after JSON object", http.StatusBadRequest)
|
|
197
|
+
return
|
|
198
|
+
}
|
|
199
|
+
}
|
|
200
|
+
if req.CLI != "" && !slices.Contains(clis, req.CLI) {
|
|
201
|
+
http.Error(w, fmt.Sprintf("judge CLI %q is not available (installed: %v)", req.CLI, clis), http.StatusBadRequest)
|
|
202
|
+
return
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
noRubric := req.Rubric != nil && !*req.Rubric
|
|
206
|
+
config := jobConfig(req.CLI, req.Model, noRubric)
|
|
207
|
+
s.analyze.mu.Lock()
|
|
208
|
+
if job := s.analyze.jobs[meta.Key]; job != nil && !job.done {
|
|
209
|
+
sameConfig := job.config == config
|
|
210
|
+
s.analyze.mu.Unlock()
|
|
211
|
+
if !sameConfig {
|
|
212
|
+
// The API must not pretend to accept a configuration it will not
|
|
213
|
+
// run: the in-flight job was asked for something else.
|
|
214
|
+
http.Error(w, "an evaluation with a different judge configuration is already running for this session; wait for it to finish", http.StatusConflict)
|
|
215
|
+
return
|
|
216
|
+
}
|
|
217
|
+
w.WriteHeader(http.StatusAccepted)
|
|
218
|
+
writeJSON(w, reportStatus{State: "running", JudgeAvailable: true})
|
|
219
|
+
return
|
|
220
|
+
}
|
|
221
|
+
if s.analyze.active >= maxConcurrentJudges {
|
|
222
|
+
s.analyze.mu.Unlock()
|
|
223
|
+
http.Error(w, fmt.Sprintf("%d evaluations already running; wait for one to finish", maxConcurrentJudges), http.StatusTooManyRequests)
|
|
224
|
+
return
|
|
225
|
+
}
|
|
226
|
+
job := &analyzeJob{config: config}
|
|
227
|
+
s.analyze.jobs[meta.Key] = job
|
|
228
|
+
s.analyze.active++
|
|
229
|
+
s.analyze.mu.Unlock()
|
|
230
|
+
|
|
231
|
+
go s.runAnalyze(meta.Key, trace, job, req.CLI, req.Model, noRubric)
|
|
232
|
+
|
|
233
|
+
w.WriteHeader(http.StatusAccepted)
|
|
234
|
+
writeJSON(w, reportStatus{State: "running", JudgeAvailable: true})
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
func (s *Server) runAnalyze(key string, trace *model.Trace, job *analyzeJob, cli, judgeModel string, noRubric bool) {
|
|
238
|
+
ctx, cancel := context.WithTimeout(context.Background(), judge.DefaultTimeout)
|
|
239
|
+
defer cancel()
|
|
240
|
+
// The cached report seeds rubric reuse: a re-evaluation with unchanged
|
|
241
|
+
// task wording keeps the same criteria — including across judge CLIs,
|
|
242
|
+
// which is exactly what makes their verdicts comparable. A rubric:false
|
|
243
|
+
// run mirrors the CLI's --no-rubric semantics and bypasses the cache in
|
|
244
|
+
// both directions: it neither mines the cache nor overwrites a richer
|
|
245
|
+
// report with a dimensions-only one — its result lives only in the job.
|
|
246
|
+
var cached *model.Report
|
|
247
|
+
if !noRubric {
|
|
248
|
+
cached = s.reportCache.Load(key)
|
|
249
|
+
}
|
|
250
|
+
report, err := judge.Analyze(ctx, trace, judge.Options{
|
|
251
|
+
Runner: s.analyze.runner,
|
|
252
|
+
CLI: cli,
|
|
253
|
+
Model: judgeModel,
|
|
254
|
+
NoRubric: noRubric,
|
|
255
|
+
CachedReport: cached,
|
|
256
|
+
})
|
|
257
|
+
|
|
258
|
+
// Persist before publishing done, and outside the lock: once the job entry
|
|
259
|
+
// is dropped, polls must be able to find the report on disk.
|
|
260
|
+
persisted := false
|
|
261
|
+
if err == nil && !noRubric && s.reportCache.Dir != "" {
|
|
262
|
+
persisted = s.reportCache.Store(key, report) == nil
|
|
263
|
+
}
|
|
264
|
+
if persisted {
|
|
265
|
+
// Polls landing between the store and the index's next directory scan
|
|
266
|
+
// must find the report once the job entry is dropped below.
|
|
267
|
+
s.reportIndex.markPresent(key)
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
s.analyze.mu.Lock()
|
|
271
|
+
defer s.analyze.mu.Unlock()
|
|
272
|
+
s.analyze.active--
|
|
273
|
+
job.done = true
|
|
274
|
+
if err != nil {
|
|
275
|
+
job.err = err.Error()
|
|
276
|
+
return
|
|
277
|
+
}
|
|
278
|
+
job.report = report
|
|
279
|
+
if persisted {
|
|
280
|
+
// The cache owns the report now; dropping the entry keeps the jobs map
|
|
281
|
+
// bounded. When the disk write failed the entry stays as the only copy —
|
|
282
|
+
// losing it would cost a re-run — and failed jobs stay too (small, and
|
|
283
|
+
// the UI needs the error until a re-run replaces them).
|
|
284
|
+
delete(s.analyze.jobs, key)
|
|
285
|
+
}
|
|
286
|
+
}
|