@hybridlabor-api/bdb-synapse 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (102) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +173 -0
  3. package/bin/synapse +0 -0
  4. package/cmd/rubriceval/main.go +308 -0
  5. package/cmd/synapse/main.go +280 -0
  6. package/cmd/synapse/main_test.go +16 -0
  7. package/go.mod +7 -0
  8. package/go.sum +6 -0
  9. package/internal/adapter/adapter.go +1117 -0
  10. package/internal/adapter/adapter_test.go +518 -0
  11. package/internal/adapter/agy/adapter.go +193 -0
  12. package/internal/adapter/claudecode/adapter.go +415 -0
  13. package/internal/adapter/claudecode/adapter_test.go +260 -0
  14. package/internal/adapter/claudecode/agents.go +387 -0
  15. package/internal/adapter/claudecode/agents_test.go +480 -0
  16. package/internal/adapter/claudecode/summary_inputs_test.go +19 -0
  17. package/internal/adapter/codex/adapter.go +919 -0
  18. package/internal/adapter/codex/adapter_test.go +920 -0
  19. package/internal/adapter/codex/agents.go +401 -0
  20. package/internal/adapter/codex/agents_test.go +610 -0
  21. package/internal/adapter/codex/summary_inputs_test.go +20 -0
  22. package/internal/adapter/pi/adapter.go +483 -0
  23. package/internal/adapter/pi/adapter_test.go +517 -0
  24. package/internal/citymap/builder.go +1124 -0
  25. package/internal/citymap/builder_test.go +818 -0
  26. package/internal/judge/cache.go +180 -0
  27. package/internal/judge/cli.go +240 -0
  28. package/internal/judge/cli_test.go +68 -0
  29. package/internal/judge/fresh_summary_test.go +37 -0
  30. package/internal/judge/input.go +233 -0
  31. package/internal/judge/judge.go +416 -0
  32. package/internal/judge/judge_test.go +288 -0
  33. package/internal/judge/prompt.go +129 -0
  34. package/internal/judge/rubric.go +275 -0
  35. package/internal/judge/rubric_test.go +642 -0
  36. package/internal/model/agent.go +63 -0
  37. package/internal/model/agent_schema_test.go +86 -0
  38. package/internal/model/agent_test.go +64 -0
  39. package/internal/model/model.go +175 -0
  40. package/internal/model/report.go +166 -0
  41. package/internal/model/stats.go +151 -0
  42. package/internal/model/stats_test.go +89 -0
  43. package/internal/model/trace_schema_test.go +67 -0
  44. package/internal/server/analyze.go +286 -0
  45. package/internal/server/analyze_test.go +297 -0
  46. package/internal/server/codex_index_test.go +40 -0
  47. package/internal/server/hardening_test.go +147 -0
  48. package/internal/server/reportindex.go +91 -0
  49. package/internal/server/reportindex_test.go +116 -0
  50. package/internal/server/server.go +1099 -0
  51. package/internal/server/server_test.go +1389 -0
  52. package/internal/server/static/assets/fraunces-latin-ext-standard-italic-CGbN9UgK.woff2 +0 -0
  53. package/internal/server/static/assets/fraunces-latin-ext-standard-normal-CJcjJNj7.woff2 +0 -0
  54. package/internal/server/static/assets/fraunces-latin-standard-italic-lSdLDfvT.woff2 +0 -0
  55. package/internal/server/static/assets/fraunces-latin-standard-normal-DihXLNYH.woff2 +0 -0
  56. package/internal/server/static/assets/fraunces-vietnamese-standard-italic-DxWqP7Ku.woff2 +0 -0
  57. package/internal/server/static/assets/fraunces-vietnamese-standard-normal-Czevyj-6.woff2 +0 -0
  58. package/internal/server/static/assets/index-BNoY_BiB.css +1 -0
  59. package/internal/server/static/assets/index-C_adLrJr.js +3 -0
  60. package/internal/server/static/assets/react-gcHzaSmV.js +10 -0
  61. package/internal/server/static/assets/schibsted-grotesk-latin-ext-wght-normal-hsMS0n0O.woff2 +0 -0
  62. package/internal/server/static/assets/schibsted-grotesk-latin-wght-normal-Bb8VGrTG.woff2 +0 -0
  63. package/internal/server/static/assets/three-DnGjZfD1.js +4012 -0
  64. package/internal/server/static/index.html +26 -0
  65. package/internal/server/tracestore.go +173 -0
  66. package/internal/server/tracestore_test.go +51 -0
  67. package/internal/textutil/truncate.go +30 -0
  68. package/internal/textutil/truncate_test.go +39 -0
  69. package/package.json +35 -0
  70. package/web/e2e/agent-lens.spec.ts +688 -0
  71. package/web/index.html +23 -0
  72. package/web/package-lock.json +1933 -0
  73. package/web/package.json +33 -0
  74. package/web/playwright.config.ts +24 -0
  75. package/web/src/App.tsx +876 -0
  76. package/web/src/api/client.ts +74 -0
  77. package/web/src/main.tsx +12 -0
  78. package/web/src/playback/recorder.ts +160 -0
  79. package/web/src/playback/reducer.ts +91 -0
  80. package/web/src/scene/CityScene.tsx +638 -0
  81. package/web/src/scene/TreeScene.tsx +656 -0
  82. package/web/src/scene/dirLabels.ts +145 -0
  83. package/web/src/scene/sceneUtils.ts +144 -0
  84. package/web/src/scene/textures.ts +60 -0
  85. package/web/src/scene/trail.ts +79 -0
  86. package/web/src/scene/treeLayout.ts +169 -0
  87. package/web/src/state/filters.ts +40 -0
  88. package/web/src/state/store.ts +83 -0
  89. package/web/src/styles.css +2565 -0
  90. package/web/src/types.ts +315 -0
  91. package/web/src/ui/AgentsPanel.tsx +376 -0
  92. package/web/src/ui/Dock.tsx +104 -0
  93. package/web/src/ui/Hud.tsx +335 -0
  94. package/web/src/ui/Inspector.tsx +107 -0
  95. package/web/src/ui/LogoMark.tsx +38 -0
  96. package/web/src/ui/ReportPanel.tsx +491 -0
  97. package/web/src/ui/SessionRail.tsx +316 -0
  98. package/web/src/ui/Timeline.tsx +458 -0
  99. package/web/src/ui/ViewPanel.tsx +45 -0
  100. package/web/src/ui/shortcuts.ts +4 -0
  101. package/web/tsconfig.json +21 -0
  102. package/web/vite.config.ts +28 -0
@@ -0,0 +1,63 @@
1
+ package model
2
+
3
+ const AgentGraphVersion = 1
4
+
5
+ const (
6
+ AgentKindMain = "main"
7
+ AgentKindSubagent = "subagent"
8
+
9
+ AgentStatusMain = "main"
10
+ AgentStatusLaunched = "launched"
11
+ AgentStatusFailed = "failed"
12
+ AgentStatusUnknown = "unknown"
13
+
14
+ TraceAvailabilityAvailable = "available"
15
+ TraceAvailabilityMissing = "missing"
16
+ TraceAvailabilityUnavailable = "unavailable"
17
+
18
+ AgentLinkQualityExact = "exact"
19
+ AgentLinkQualityDerived = "derived"
20
+ AgentLinkQualityUnavailable = "unavailable"
21
+
22
+ AgentLinkMethodRoot = "root"
23
+ AgentLinkMethodCodexAgentID = "codex-agent-id"
24
+ AgentLinkMethodCodexParentThreadID = "codex-parent-thread-id"
25
+ AgentLinkMethodClaudeToolUseID = "claude-tool-use-id"
26
+ AgentLinkMethodClaudeSubagentsDirectory = "claude-subagents-directory"
27
+ AgentLinkMethodUnavailable = "unavailable"
28
+ )
29
+
30
+ type AgentGraph struct {
31
+ Version int `json:"version"`
32
+ RootSessionKey string `json:"rootSessionKey"`
33
+ Agents []AgentNode `json:"agents"`
34
+ }
35
+
36
+ type AgentNode struct {
37
+ ID string `json:"id"`
38
+ ParentID string `json:"parentId,omitempty"`
39
+ Depth int `json:"depth"`
40
+ Kind string `json:"kind"`
41
+ Label string `json:"label"`
42
+ Role string `json:"role,omitempty"`
43
+ InstructionPreview string `json:"instructionPreview,omitempty"`
44
+ LaunchSeq *int `json:"launchSeq,omitempty"`
45
+ LaunchCallID string `json:"launchCallId,omitempty"`
46
+ Status string `json:"status"`
47
+ TraceAvailability string `json:"traceAvailability"`
48
+ TraceSessionKey string `json:"traceSessionKey,omitempty"`
49
+ TraceEventCount int `json:"traceEventCount"`
50
+ LinkQuality string `json:"linkQuality"`
51
+ LinkMethod string `json:"linkMethod"`
52
+ }
53
+
54
+ type AgentSessionMeta struct {
55
+ SourceID string
56
+ RootSessionID string
57
+ ParentSessionID string
58
+ AgentPath string
59
+ Depth int
60
+ Label string
61
+ Role string
62
+ LaunchCallID string
63
+ }
@@ -0,0 +1,86 @@
1
+ package model
2
+
3
+ import (
4
+ "encoding/json"
5
+ "testing"
6
+
7
+ "github.com/santhosh-tekuri/jsonschema/v6"
8
+ )
9
+
10
+ func TestAgentGraphSchemaAcceptsRepresentativeGraph(t *testing.T) {
11
+ launchSeq := 2
12
+ graph := AgentGraph{
13
+ Version: AgentGraphVersion,
14
+ RootSessionKey: "root-key",
15
+ Agents: []AgentNode{
16
+ {
17
+ ID: "agt_main",
18
+ Kind: AgentKindMain,
19
+ Label: "Main",
20
+ Status: AgentStatusMain,
21
+ TraceAvailability: TraceAvailabilityAvailable,
22
+ TraceSessionKey: "root-key",
23
+ TraceEventCount: 12,
24
+ LinkQuality: AgentLinkQualityExact,
25
+ LinkMethod: AgentLinkMethodRoot,
26
+ },
27
+ {
28
+ ID: "agt_child",
29
+ ParentID: "agt_main",
30
+ Depth: 1,
31
+ Kind: AgentKindSubagent,
32
+ Label: "Reviewer",
33
+ Role: "reviewer",
34
+ InstructionPreview: "Review the change",
35
+ LaunchSeq: &launchSeq,
36
+ LaunchCallID: "call-review",
37
+ Status: AgentStatusLaunched,
38
+ TraceAvailability: TraceAvailabilityAvailable,
39
+ TraceSessionKey: "child-key",
40
+ TraceEventCount: 3,
41
+ LinkQuality: AgentLinkQualityExact,
42
+ LinkMethod: AgentLinkMethodCodexAgentID,
43
+ },
44
+ {
45
+ ID: "agt_missing",
46
+ ParentID: "agt_main",
47
+ Depth: 1,
48
+ Kind: AgentKindSubagent,
49
+ Label: "Missing",
50
+ Status: AgentStatusLaunched,
51
+ TraceAvailability: TraceAvailabilityMissing,
52
+ LinkQuality: AgentLinkQualityDerived,
53
+ LinkMethod: AgentLinkMethodClaudeSubagentsDirectory,
54
+ },
55
+ {
56
+ ID: "agt_failed",
57
+ ParentID: "agt_main",
58
+ Depth: 1,
59
+ Kind: AgentKindSubagent,
60
+ Label: "Failed",
61
+ Status: AgentStatusFailed,
62
+ TraceAvailability: TraceAvailabilityUnavailable,
63
+ LinkQuality: AgentLinkQualityUnavailable,
64
+ LinkMethod: AgentLinkMethodUnavailable,
65
+ },
66
+ },
67
+ }
68
+
69
+ document, err := json.Marshal(graph)
70
+ if err != nil {
71
+ t.Fatal(err)
72
+ }
73
+ var value any
74
+ if err := json.Unmarshal(document, &value); err != nil {
75
+ t.Fatal(err)
76
+ }
77
+
78
+ compiler := jsonschema.NewCompiler()
79
+ schema, err := compiler.Compile("../../schema/agent-graph.schema.json")
80
+ if err != nil {
81
+ t.Fatal(err)
82
+ }
83
+ if err := schema.Validate(value); err != nil {
84
+ t.Fatalf("representative AgentGraph violates schema: %v\n%s", err, document)
85
+ }
86
+ }
@@ -0,0 +1,64 @@
1
+ package model
2
+
3
+ import (
4
+ "encoding/json"
5
+ "testing"
6
+ )
7
+
8
+ func TestAgentNodeOmitsMainRelationshipFields(t *testing.T) {
9
+ node := AgentNode{
10
+ ID: "main",
11
+ Depth: 0,
12
+ Kind: AgentKindMain,
13
+ Label: "Main",
14
+ Status: AgentStatusMain,
15
+ TraceAvailability: TraceAvailabilityAvailable,
16
+ TraceSessionKey: "root-key",
17
+ TraceEventCount: 12,
18
+ LinkQuality: AgentLinkQualityExact,
19
+ LinkMethod: AgentLinkMethodRoot,
20
+ }
21
+
22
+ data, err := json.Marshal(node)
23
+ if err != nil {
24
+ t.Fatal(err)
25
+ }
26
+ var got map[string]any
27
+ if err := json.Unmarshal(data, &got); err != nil {
28
+ t.Fatal(err)
29
+ }
30
+ if _, ok := got["parentId"]; ok {
31
+ t.Fatalf("main node serialized parentId: %s", data)
32
+ }
33
+ if _, ok := got["launchSeq"]; ok {
34
+ t.Fatalf("main node serialized launchSeq: %s", data)
35
+ }
36
+ }
37
+
38
+ func TestAgentNodeSerializesZeroLaunchSeq(t *testing.T) {
39
+ zero := 0
40
+ node := AgentNode{
41
+ ID: "child",
42
+ ParentID: "main",
43
+ Depth: 1,
44
+ Kind: AgentKindSubagent,
45
+ Label: "Child",
46
+ LaunchSeq: &zero,
47
+ Status: AgentStatusLaunched,
48
+ TraceAvailability: TraceAvailabilityMissing,
49
+ LinkQuality: AgentLinkQualityExact,
50
+ LinkMethod: AgentLinkMethodCodexAgentID,
51
+ }
52
+
53
+ data, err := json.Marshal(node)
54
+ if err != nil {
55
+ t.Fatal(err)
56
+ }
57
+ var got map[string]any
58
+ if err := json.Unmarshal(data, &got); err != nil {
59
+ t.Fatal(err)
60
+ }
61
+ if got["launchSeq"] != float64(0) {
62
+ t.Fatalf("launchSeq = %#v, JSON = %s", got["launchSeq"], data)
63
+ }
64
+ }
@@ -0,0 +1,175 @@
1
+ package model
2
+
3
+ type Rect struct {
4
+ X float64 `json:"x"`
5
+ Z float64 `json:"z"`
6
+ W float64 `json:"w"`
7
+ D float64 `json:"d"`
8
+ }
9
+
10
+ type RepoMeta struct {
11
+ Root string `json:"root"`
12
+ Commit string `json:"commit,omitempty"`
13
+ Dirty bool `json:"dirty"`
14
+ GeneratedAt string `json:"generatedAt"`
15
+ // Truncated marks a map that hit a scan or size budget — the session's
16
+ // tree (or its trace targets) holds more than the citymap shows.
17
+ Truncated bool `json:"truncated,omitempty"`
18
+ }
19
+
20
+ type CityMap struct {
21
+ Version int `json:"version"`
22
+ Repo RepoMeta `json:"repo"`
23
+ Files []CityFile `json:"files"`
24
+ Dirs []CityDir `json:"dirs"`
25
+ Layout LayoutMeta `json:"layout"`
26
+ }
27
+
28
+ type CityFile struct {
29
+ ID int `json:"id"`
30
+ Path string `json:"path"`
31
+ Dir string `json:"dir"`
32
+ Lines int `json:"lines"`
33
+ Bytes int64 `json:"bytes"`
34
+ Lang string `json:"lang,omitempty"`
35
+ Rect Rect `json:"rect"`
36
+ Ghost bool `json:"ghost"`
37
+ }
38
+
39
+ type CityDir struct {
40
+ Path string `json:"path"`
41
+ Depth int `json:"depth"`
42
+ Rect Rect `json:"rect"`
43
+ FileCount int `json:"fileCount"`
44
+ Lines int `json:"lines"`
45
+ }
46
+
47
+ type LayoutMeta struct {
48
+ Algorithm string `json:"algorithm"`
49
+ Weight string `json:"weight"`
50
+ }
51
+
52
+ type Trace struct {
53
+ Version int `json:"version"`
54
+ Session TraceSession `json:"session"`
55
+ Events []Event `json:"events"`
56
+ Marks []Mark `json:"marks"`
57
+ Stats Stats `json:"stats"`
58
+ }
59
+
60
+ type TraceSession struct {
61
+ ID string `json:"id"`
62
+ Harness string `json:"harness"`
63
+ Model string `json:"model,omitempty"`
64
+ Title string `json:"title,omitempty"`
65
+ Cwd string `json:"cwd,omitempty"`
66
+ Commit string `json:"commit,omitempty"`
67
+ StartedAt string `json:"startedAt,omitempty"`
68
+ EndedAt string `json:"endedAt,omitempty"`
69
+ EventCount int `json:"eventCount"`
70
+ Path string `json:"path,omitempty"`
71
+ }
72
+
73
+ type Event struct {
74
+ Seq int `json:"seq"`
75
+ Timestamp string `json:"ts,omitempty"`
76
+ Tool string `json:"tool"`
77
+ Action string `json:"action"`
78
+ Targets []Target `json:"targets"`
79
+ Outside []OutsideTouch `json:"outside,omitempty"`
80
+ ResultBytes int `json:"resultBytes"`
81
+ IsError bool `json:"isError"`
82
+ // OutcomeKnown distinguishes a recorded success from an unknown non-error
83
+ // result. IsError remains sufficient to identify failures in older traces.
84
+ OutcomeKnown bool `json:"outcomeKnown,omitempty"`
85
+ Summary string `json:"summary"`
86
+ }
87
+
88
+ type Target struct {
89
+ Path string `json:"path"`
90
+ FileID *int `json:"fileId,omitempty"`
91
+ Touch string `json:"touch"`
92
+ Lines [][2]int `json:"lines,omitempty"`
93
+ Weak bool `json:"weak,omitempty"`
94
+ }
95
+
96
+ type OutsideTouch struct {
97
+ Scope string `json:"scope"`
98
+ Path string `json:"path"`
99
+ }
100
+
101
+ type Mark struct {
102
+ Seq int `json:"seq"`
103
+ Type string `json:"type"`
104
+ Note string `json:"note,omitempty"`
105
+ }
106
+
107
+ type Stats struct {
108
+ FilesInRepo int `json:"filesInRepo"`
109
+ Fovea int `json:"fovea"`
110
+ Parafovea int `json:"parafovea"`
111
+ Edited int `json:"edited"`
112
+ EventsBeforeFirstEdit int `json:"eventsBeforeFirstEdit"`
113
+ RegressionRate float64 `json:"regressionRate"`
114
+ ErrorRate float64 `json:"errorRate"`
115
+ Actions ActionCounts `json:"actions"`
116
+ Errors ActionCounts `json:"errors"`
117
+ MaxEditsPerFile int `json:"maxEditsPerFile"`
118
+ // ChurnFiles counts files edited in three or more events.
119
+ ChurnFiles int `json:"churnFiles"`
120
+ UserTurns int `json:"userTurns"`
121
+ Compactions int `json:"compactions"`
122
+ Subagents int `json:"subagents"`
123
+ ResultBytes int64 `json:"resultBytes"`
124
+ // EditsAfterLastVerify counts edit events after the last verify event;
125
+ // when the session never ran a verify it counts every edit event.
126
+ EditsAfterLastVerify int `json:"editsAfterLastVerify"`
127
+ // Observability grades each derived metric's source signal so the UI can
128
+ // tell a true zero from a blind spot in the session log.
129
+ Observability Observability `json:"observability"`
130
+ }
131
+
132
+ // Observability values: "exact" when the harness records the signal
133
+ // structurally, "estimated" when it is inferred from command or output text,
134
+ // "unavailable" when the log carries no usable signal.
135
+ const (
136
+ ObservabilityExact = "exact"
137
+ ObservabilityEstimated = "estimated"
138
+ ObservabilityUnavailable = "unavailable"
139
+ )
140
+
141
+ type Observability struct {
142
+ Reads string `json:"reads"`
143
+ Errors string `json:"errors"`
144
+ }
145
+
146
+ // ActionCounts tallies events per action class; as Stats.Errors it tallies
147
+ // only the events that returned an error.
148
+ type ActionCounts struct {
149
+ Search int `json:"search"`
150
+ Read int `json:"read"`
151
+ Edit int `json:"edit"`
152
+ Exec int `json:"exec"`
153
+ Verify int `json:"verify"`
154
+ Other int `json:"other"`
155
+ }
156
+
157
+ type SessionMeta struct {
158
+ Key string `json:"key"`
159
+ ID string `json:"id"`
160
+ Harness string `json:"harness"`
161
+ Title string `json:"title,omitempty"`
162
+ Path string `json:"path"`
163
+ Cwd string `json:"cwd,omitempty"`
164
+ Model string `json:"model,omitempty"`
165
+ GitBranch string `json:"gitBranch,omitempty"`
166
+ StartedAt string `json:"startedAt,omitempty"`
167
+ EndedAt string `json:"endedAt,omitempty"`
168
+ // EventCount and UserTurns together are the cheap staleness signal for
169
+ // report badges: user messages land on marks, not events, so the count
170
+ // alone misses exactly the follow-ups that matter most.
171
+ EventCount int `json:"eventCount"`
172
+ UserTurns int `json:"userTurns"`
173
+ Auxiliary bool `json:"-"`
174
+ Agent *AgentSessionMeta `json:"-"`
175
+ }
@@ -0,0 +1,166 @@
1
+ package model
2
+
3
+ // Report is the third first-class artifact next to CityMap and Trace: an
4
+ // LLM-assisted evaluation of one session trace. The LLM contributes findings
5
+ // and narrative; dimension verdicts are always derived mechanically from
6
+ // finding severities so two reports stay comparable.
7
+ type Report struct {
8
+ Version int `json:"version"`
9
+ Session ReportSession `json:"session"`
10
+ Judge ReportJudge `json:"judge"`
11
+ TaskSummary string `json:"taskSummary"`
12
+ Dimensions []ReportDimension `json:"dimensions"`
13
+ Rubric *Rubric `json:"rubric,omitempty"`
14
+ NotableMoments []ReportMoment `json:"notableMoments,omitempty"`
15
+ Narrative string `json:"narrative"`
16
+ }
17
+
18
+ // ReportSession pins the report to the trace state it was generated from;
19
+ // EventCount is a cheap display/badge signal — freshness is decided by
20
+ // ReportJudge.InputDigest, which also sees user messages and event content.
21
+ type ReportSession struct {
22
+ ID string `json:"id"`
23
+ Harness string `json:"harness"`
24
+ Model string `json:"model,omitempty"`
25
+ EventCount int `json:"eventCount"`
26
+ // UserTurns mirrors SessionMeta.UserTurns at generation time, giving the
27
+ // badge's cheap staleness check eyes on message-only session growth.
28
+ UserTurns int `json:"userTurns,omitempty"`
29
+ }
30
+
31
+ type ReportJudge struct {
32
+ CLI string `json:"cli"`
33
+ // Model names the LLM that actually judged (best-effort, reported by the
34
+ // CLI itself); display and comparability only — never part of freshness.
35
+ Model string `json:"model,omitempty"`
36
+ // RequestedModel keeps the alias the run was asked for (e.g. "sonnet"),
37
+ // so a repeated aliased request can recognize its own cached report.
38
+ RequestedModel string `json:"requestedModel,omitempty"`
39
+ PromptVersion int `json:"promptVersion"`
40
+ // RubricPromptVersion is set only when the report carries a scored rubric;
41
+ // deterministic skips (no/weak task text) stay fresh across rubric prompt
42
+ // revisions because no generation happened.
43
+ RubricPromptVersion int `json:"rubricPromptVersion,omitempty"`
44
+ GeneratedAt string `json:"generatedAt"`
45
+ // InputDigest fingerprints the exact evidence document the judge read;
46
+ // the report is fresh only while the trace still renders to this digest.
47
+ InputDigest string `json:"inputDigest,omitempty"`
48
+ }
49
+
50
+ // Dimension names, fixed set.
51
+ const (
52
+ DimensionExploration = "exploration"
53
+ DimensionScope = "scope"
54
+ DimensionWandering = "wandering"
55
+ DimensionVerification = "verification"
56
+ )
57
+
58
+ // DimensionNames lists the four evaluation dimensions in display order.
59
+ var DimensionNames = []string{DimensionExploration, DimensionScope, DimensionWandering, DimensionVerification}
60
+
61
+ // Verdict values; SeverityInfo maps to VerdictGood.
62
+ const (
63
+ VerdictGood = "good"
64
+ VerdictWarning = "warning"
65
+ VerdictProblem = "problem"
66
+ VerdictInsufficientData = "insufficient-data"
67
+ )
68
+
69
+ const (
70
+ SeverityInfo = "info"
71
+ SeverityWarning = "warning"
72
+ SeverityProblem = "problem"
73
+ )
74
+
75
+ type ReportDimension struct {
76
+ Name string `json:"name"`
77
+ Verdict string `json:"verdict"`
78
+ Findings []ReportFinding `json:"findings"`
79
+ }
80
+
81
+ type ReportFinding struct {
82
+ Claim string `json:"claim"`
83
+ Severity string `json:"severity"`
84
+ // Always at least one entry — evidence-less findings are dropped at
85
+ // parse time, and the schema marks the field required accordingly.
86
+ EvidenceSeqs []int `json:"evidenceSeqs"`
87
+ }
88
+
89
+ type ReportMoment struct {
90
+ Seq int `json:"seq"`
91
+ Note string `json:"note"`
92
+ }
93
+
94
+ // Rubric statuses and skip/degrade reasons.
95
+ const (
96
+ RubricStatusScored = "scored"
97
+ RubricStatusUnavailable = "unavailable"
98
+
99
+ RubricReasonGenerationFailed = "generation-failed"
100
+ RubricReasonNoTaskText = "no-task-text"
101
+ RubricReasonWeakTaskText = "weak-task-text"
102
+ // RubricReasonNoEvents skips traces with no tool events: with nothing to
103
+ // cite, every finding would be dropped and criteria would default to
104
+ // good verdicts on zero evidence.
105
+ RubricReasonNoEvents = "no-events"
106
+ )
107
+
108
+ // Rubric generation input modes. A rubric generated with the full evidence
109
+ // document may absorb this attempt's implementation choices into its anchors;
110
+ // comparison across agents must only ever reuse task-sourced rubrics.
111
+ const (
112
+ RubricSourceFull = "full"
113
+ RubricSourceTask = "task"
114
+ )
115
+
116
+ // Criterion evidence coverage. Unlike severities these never feed warnings:
117
+ // a criterion the log cannot evidence rolls up to insufficient-data instead
118
+ // of counting against the agent.
119
+ const (
120
+ CoverageSufficient = "sufficient"
121
+ CoveragePartial = "partial"
122
+ CoverageNone = "none"
123
+ )
124
+
125
+ // Rubric is the task-accounting layer of a report: session-specific criteria
126
+ // generated before scoring, grouped by the independent tasks the judge
127
+ // enumerated from the user's messages. The fixed dimensions never depend on
128
+ // it — a rubric failure degrades to a dimensions-only report.
129
+ type Rubric struct {
130
+ Status string `json:"status"`
131
+ // Reason qualifies an unavailable rubric: generation-failed after retry,
132
+ // or a deterministic skip (no-task-text, weak-task-text).
133
+ Reason string `json:"reason,omitempty"`
134
+ Source string `json:"source,omitempty"`
135
+ // TaskDigest fingerprints the task wording the rubric was derived from;
136
+ // a re-evaluation whose digest still matches reuses the rubric unchanged.
137
+ TaskDigest string `json:"taskDigest,omitempty"`
138
+ Tasks []RubricTask `json:"tasks,omitempty"`
139
+ // Note carries what the scorer felt the rubric did not let it express.
140
+ Note string `json:"note,omitempty"`
141
+ }
142
+
143
+ type RubricTask struct {
144
+ Title string `json:"title"`
145
+ Type string `json:"type,omitempty"`
146
+ // AnchorUserMessages are [user #N] ordinals from the evidence document,
147
+ // validated against the ordinals actually rendered there.
148
+ AnchorUserMessages []int `json:"anchorUserMessages"`
149
+ // AnchorSeqs are the mark seqs those ordinals resolve to — derived in Go,
150
+ // never taken from the judge — so the UI can jump to a task's start.
151
+ AnchorSeqs []int `json:"anchorSeqs,omitempty"`
152
+ Criteria []RubricCriterion `json:"criteria"`
153
+ }
154
+
155
+ type RubricCriterion struct {
156
+ ID string `json:"id"`
157
+ Title string `json:"title"`
158
+ Why string `json:"why,omitempty"`
159
+ Good string `json:"good,omitempty"`
160
+ Bad string `json:"bad,omitempty"`
161
+ // Coverage and findings come from the scoring pass; verdict is rolled up
162
+ // mechanically (coverage none forces insufficient-data).
163
+ Coverage string `json:"coverage,omitempty"`
164
+ Verdict string `json:"verdict"`
165
+ Findings []ReportFinding `json:"findings"`
166
+ }