@hybridlabor-api/bdb-synapse 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (102) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +173 -0
  3. package/bin/synapse +0 -0
  4. package/cmd/rubriceval/main.go +308 -0
  5. package/cmd/synapse/main.go +280 -0
  6. package/cmd/synapse/main_test.go +16 -0
  7. package/go.mod +7 -0
  8. package/go.sum +6 -0
  9. package/internal/adapter/adapter.go +1117 -0
  10. package/internal/adapter/adapter_test.go +518 -0
  11. package/internal/adapter/agy/adapter.go +193 -0
  12. package/internal/adapter/claudecode/adapter.go +415 -0
  13. package/internal/adapter/claudecode/adapter_test.go +260 -0
  14. package/internal/adapter/claudecode/agents.go +387 -0
  15. package/internal/adapter/claudecode/agents_test.go +480 -0
  16. package/internal/adapter/claudecode/summary_inputs_test.go +19 -0
  17. package/internal/adapter/codex/adapter.go +919 -0
  18. package/internal/adapter/codex/adapter_test.go +920 -0
  19. package/internal/adapter/codex/agents.go +401 -0
  20. package/internal/adapter/codex/agents_test.go +610 -0
  21. package/internal/adapter/codex/summary_inputs_test.go +20 -0
  22. package/internal/adapter/pi/adapter.go +483 -0
  23. package/internal/adapter/pi/adapter_test.go +517 -0
  24. package/internal/citymap/builder.go +1124 -0
  25. package/internal/citymap/builder_test.go +818 -0
  26. package/internal/judge/cache.go +180 -0
  27. package/internal/judge/cli.go +240 -0
  28. package/internal/judge/cli_test.go +68 -0
  29. package/internal/judge/fresh_summary_test.go +37 -0
  30. package/internal/judge/input.go +233 -0
  31. package/internal/judge/judge.go +416 -0
  32. package/internal/judge/judge_test.go +288 -0
  33. package/internal/judge/prompt.go +129 -0
  34. package/internal/judge/rubric.go +275 -0
  35. package/internal/judge/rubric_test.go +642 -0
  36. package/internal/model/agent.go +63 -0
  37. package/internal/model/agent_schema_test.go +86 -0
  38. package/internal/model/agent_test.go +64 -0
  39. package/internal/model/model.go +175 -0
  40. package/internal/model/report.go +166 -0
  41. package/internal/model/stats.go +151 -0
  42. package/internal/model/stats_test.go +89 -0
  43. package/internal/model/trace_schema_test.go +67 -0
  44. package/internal/server/analyze.go +286 -0
  45. package/internal/server/analyze_test.go +297 -0
  46. package/internal/server/codex_index_test.go +40 -0
  47. package/internal/server/hardening_test.go +147 -0
  48. package/internal/server/reportindex.go +91 -0
  49. package/internal/server/reportindex_test.go +116 -0
  50. package/internal/server/server.go +1099 -0
  51. package/internal/server/server_test.go +1389 -0
  52. package/internal/server/static/assets/fraunces-latin-ext-standard-italic-CGbN9UgK.woff2 +0 -0
  53. package/internal/server/static/assets/fraunces-latin-ext-standard-normal-CJcjJNj7.woff2 +0 -0
  54. package/internal/server/static/assets/fraunces-latin-standard-italic-lSdLDfvT.woff2 +0 -0
  55. package/internal/server/static/assets/fraunces-latin-standard-normal-DihXLNYH.woff2 +0 -0
  56. package/internal/server/static/assets/fraunces-vietnamese-standard-italic-DxWqP7Ku.woff2 +0 -0
  57. package/internal/server/static/assets/fraunces-vietnamese-standard-normal-Czevyj-6.woff2 +0 -0
  58. package/internal/server/static/assets/index-BNoY_BiB.css +1 -0
  59. package/internal/server/static/assets/index-C_adLrJr.js +3 -0
  60. package/internal/server/static/assets/react-gcHzaSmV.js +10 -0
  61. package/internal/server/static/assets/schibsted-grotesk-latin-ext-wght-normal-hsMS0n0O.woff2 +0 -0
  62. package/internal/server/static/assets/schibsted-grotesk-latin-wght-normal-Bb8VGrTG.woff2 +0 -0
  63. package/internal/server/static/assets/three-DnGjZfD1.js +4012 -0
  64. package/internal/server/static/index.html +26 -0
  65. package/internal/server/tracestore.go +173 -0
  66. package/internal/server/tracestore_test.go +51 -0
  67. package/internal/textutil/truncate.go +30 -0
  68. package/internal/textutil/truncate_test.go +39 -0
  69. package/package.json +35 -0
  70. package/web/e2e/agent-lens.spec.ts +688 -0
  71. package/web/index.html +23 -0
  72. package/web/package-lock.json +1933 -0
  73. package/web/package.json +33 -0
  74. package/web/playwright.config.ts +24 -0
  75. package/web/src/App.tsx +876 -0
  76. package/web/src/api/client.ts +74 -0
  77. package/web/src/main.tsx +12 -0
  78. package/web/src/playback/recorder.ts +160 -0
  79. package/web/src/playback/reducer.ts +91 -0
  80. package/web/src/scene/CityScene.tsx +638 -0
  81. package/web/src/scene/TreeScene.tsx +656 -0
  82. package/web/src/scene/dirLabels.ts +145 -0
  83. package/web/src/scene/sceneUtils.ts +144 -0
  84. package/web/src/scene/textures.ts +60 -0
  85. package/web/src/scene/trail.ts +79 -0
  86. package/web/src/scene/treeLayout.ts +169 -0
  87. package/web/src/state/filters.ts +40 -0
  88. package/web/src/state/store.ts +83 -0
  89. package/web/src/styles.css +2565 -0
  90. package/web/src/types.ts +315 -0
  91. package/web/src/ui/AgentsPanel.tsx +376 -0
  92. package/web/src/ui/Dock.tsx +104 -0
  93. package/web/src/ui/Hud.tsx +335 -0
  94. package/web/src/ui/Inspector.tsx +107 -0
  95. package/web/src/ui/LogoMark.tsx +38 -0
  96. package/web/src/ui/ReportPanel.tsx +491 -0
  97. package/web/src/ui/SessionRail.tsx +316 -0
  98. package/web/src/ui/Timeline.tsx +458 -0
  99. package/web/src/ui/ViewPanel.tsx +45 -0
  100. package/web/src/ui/shortcuts.ts +4 -0
  101. package/web/tsconfig.json +21 -0
  102. package/web/vite.config.ts +28 -0
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Ricko Yu
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
package/README.md ADDED
@@ -0,0 +1,173 @@
1
+ # <img src="assets/logo.svg" alt="" width="30" /> BDB Synapse
2
+
3
+ A visualization tool that replays coding-agent sessions on a 3D map of your codebase.
4
+
5
+ **BDB Synapse** is an extended fork of the excellent [cosmtrek/mindwalk](https://github.com/cosmtrek/mindwalk) (MIT License), created to support the BDB ecosystem of agents. It adds support for the Antigravity (agy) CLI alongside the existing supported agents.
6
+
7
+ ## The problem
8
+
9
+ A session log records what an agent did, but not how it understood the task:
10
+ which parts of the repo it treated as relevant, where it explored before it
11
+ acted, whether its footprint matched the scope you had in mind. Reading the
12
+ raw JSONL line by line doesn't answer any of that.
13
+
14
+ ## The idea
15
+
16
+ Draw the repository as a night map, and play the session back as light moving
17
+ through it: where the agent searched, read, and edited, the map glows —
18
+ everything else stays dark. The agent's understanding of the task becomes a
19
+ shape you can see at a glance. One Go binary reads Claude Code, Codex, pi, and Antigravity (agy)
20
+ session logs, fully local; viewing sends nothing anywhere. The one exception
21
+ is the optional session evaluation: when you explicitly run it, a summary of
22
+ that session (task wording, file paths, event digests) is sent to the model
23
+ behind your own `claude` or `codex` CLI — see
24
+ [Session evaluation](#session-evaluation).
25
+
26
+ ## Quick start
27
+
28
+ To build from source: `make setup && make build` → `bin/synapse`.
29
+
30
+ With no arguments, Synapse scans `~/.claude/projects`, `~/.codex/sessions`,
31
+ `~/.pi/agent/sessions`, and `~/.gemini/antigravity-cli/brain`, serves the UI on a random local port, and opens a
32
+ browser:
33
+
34
+ ```text
35
+ synapse serve [--port N] [--no-open] [--claude-dir DIR] [--codex-dir DIR] [--pi-dir DIR] [--agy-dir DIR]
36
+ synapse open [--no-open] <session.jsonl> open one specific session
37
+ synapse map [--no-open] <repo> open a repository map, no session needed
38
+ synapse build <repo> [-o out] write the repository citymap JSON
39
+ synapse trace <session> [-o out] write the normalized trace JSON
40
+ synapse analyze <session> [--judge claude|codex] [--model name] [--no-rubric]
41
+ evaluate one session (see below)
42
+ ```
43
+
44
+ ## Reading the picture
45
+
46
+ - **Tree / Terrain views** — the repo as a radial tree or a treemap plain;
47
+ glow ∝ how deeply and how often a file was touched.
48
+ - **Touch states** — each file keeps its deepest touch: seen (moss green),
49
+ read (moonlight blue), edited (warm amber), unvisited (dark). Files the
50
+ session touched that are no longer in the repo linger as wireframe ghosts.
51
+ The HUD folds friction signals — error rate, churned files, edits after the
52
+ last verify — into a review strip.
53
+ - **Playback deck** — scrub or play the session over a bucketed histogram of
54
+ the run. Bars sit on a cool/warm spectrum: observation stays cool (search,
55
+ read, exec), mutation glows warm (edit, verify), so editing phases jump out
56
+ at a glance. Restart, speed, and video export fold into the deck's `⋯` menu;
57
+ export records the playback to a `.webm` entirely client-side.
58
+ - **Timeline marks** — `◇` context compactions, `○` subagent launches,
59
+ `›` user turns; every mark is a click-to-jump target.
60
+ - **Agent lenses** — when a session launched subagents, the HUD carries a
61
+ subagent count and an agents panel: pick a lens to replay any subagent's
62
+ trace on the same map, then step back out to the main trace.
63
+ - **Inspector** — click a file to pin its visit history; click a visit row to
64
+ jump the playhead to that moment.
65
+ - **Evaluate** — ask a local agent CLI to judge the session's trajectory,
66
+ scored against criteria drafted from your own request; session rows carry
67
+ the evaluation state as a quiet badge. See
68
+ [Session evaluation](#session-evaluation).
69
+ - **Repo map** — `mindwalk map <repo>` (or the folder icon in the session
70
+ rail) renders any repository's citymap with no session attached; height
71
+ encodes lines of code instead of attention.
72
+
73
+ ![the same session on the terrain view](assets/screenshot-terrain.png)
74
+
75
+ ![agent lenses over the same session](assets/screenshot-agents.png)
76
+
77
+ Keyboard: `Space` play/pause · `←`/`→` step (`⇧` ×10) · `Home`/`End` ends ·
78
+ `S` speed · `V` view · `E` next edit · `X` next error · `M` next mark ·
79
+ `⌘B` session rail.
80
+
81
+ ## Session evaluation
82
+
83
+ The evaluate panel (and `mindwalk analyze`) asks a local agent CLI to judge
84
+ how the session went. A report has two layers:
85
+
86
+ - **Process dimensions** — exploration, scope, wandering, verification: four
87
+ fixed lenses, the same for every session, so reports stay comparable.
88
+ - **Task scorecard** — before scoring, the judge drafts criteria from your
89
+ own request wording: what would count as done for *this* task, grouped per
90
+ task when the session carried several. Each criterion is then scored
91
+ against the session, alongside the dimensions, in one pass.
92
+
93
+ Every finding in either layer must cite timeline events you can click
94
+ through to, and no verdict is the model's to decide: dimension and criterion
95
+ verdicts are rolled up mechanically from finding severities. When the log
96
+ simply can't show whether a criterion was met, its coverage drops and the
97
+ verdict reads "no signal" — an unverifiable criterion is a blind spot, not a
98
+ failure. Pick the judge (any installed CLI) and its model in the panel; the
99
+ report records who actually judged.
100
+
101
+ The scorecard steps aside rather than getting in the way: sessions with no
102
+ tool events or too little task text skip it, and a failed criteria draft
103
+ degrades to a dimensions-only report. `--no-rubric` (or `"rubric": false` on
104
+ the analyze API) skips it explicitly, in a single judge call. How the
105
+ scorecard is built — and why it is shaped the way it is — is covered in
106
+ [docs/dynamic-rubric-evaluation.md](docs/dynamic-rubric-evaluation.md).
107
+
108
+ **What leaves your machine, and only when you ask:** evaluation runs your own
109
+ `claude` or `codex` CLI — up to two sealed calls, one drafting criteria and
110
+ one scoring. Both send only that session's summary — the user messages'
111
+ wording, file paths, and one-line event digests — to the model behind your
112
+ account. Nothing is sent while viewing sessions, and no other session is
113
+ included. The judge subprocess runs sealed: no tools, no MCP servers, no user
114
+ or project settings, and no session persistence.
115
+
116
+ Reports are cached in `~/.mindwalk/reports`, one per session; a report goes
117
+ stale (never auto-reruns) when the session's content changes. Re-evaluating
118
+ a session whose task wording hasn't changed reuses the drafted criteria —
119
+ scores can move, the yardstick doesn't.
120
+
121
+ ## Under the hood
122
+
123
+ Three artifacts, kept deliberately separate:
124
+
125
+ 1. a **trace** — the session log normalized into an ordered stream of
126
+ file-touch events (`internal/adapter`, one adapter per agent format);
127
+ adapters also correlate subagent sessions into an agent graph, so each
128
+ subagent's trace can be replayed on its own;
129
+ 2. a **citymap** — a deterministic layout of the repository
130
+ (`internal/citymap`); the same tree always produces the same map, so
131
+ replays are comparable across sessions;
132
+ 3. a **report** — an LLM judge's evidence-anchored findings about one
133
+ session (`internal/judge`): four fixed process dimensions plus a
134
+ task-specific scorecard; the judge only contributes findings, verdicts
135
+ are always rolled up mechanically, so reports stay comparable too.
136
+
137
+ A local Go server (`internal/server`) joins them and serves the
138
+ React/Three.js frontend (`web`). `schema/` mirrors the exported JSON contracts.
139
+
140
+ ## Contributing
141
+
142
+ Issues and pull requests are welcome. To get a working dev setup:
143
+
144
+ ```sh
145
+ make setup # install frontend dependencies
146
+ make serve # dev server on :8765, serving web/dist from the working tree
147
+ make test # go test + frontend build — run before sending a PR
148
+ make build # regenerate embedded assets and bin/mindwalk
149
+ ```
150
+
151
+ Ground rules (see [AGENTS.md](AGENTS.md) for the full architecture notes):
152
+
153
+ - Keep the boundaries: adapters don't know about rendering, citymap generation
154
+ doesn't depend on playback, the judge reads only the normalized trace, and
155
+ the server just connects the pieces.
156
+ - Keep Go code `gofmt`-ed; never hand-edit `internal/server/static` —
157
+ regenerate it with `make build`.
158
+ - When trace, citymap, or report JSON shapes change, update `schema/` and the
159
+ relevant tests in the same change.
160
+
161
+ ## Star History
162
+
163
+ <a href="https://www.star-history.com/?repos=cosmtrek%2Fmindwalk&type=date&legend=top-left">
164
+ <picture>
165
+ <source media="(prefers-color-scheme: dark)" srcset="https://api.star-history.com/chart?repos=cosmtrek/mindwalk&type=date&theme=dark&legend=top-left&sealed_token=6ylPq85HVVSbxQtqpYdSNx2EFZXMTk4AhnMG197AQm7TDwfenvf415jqPnPRxRiXz4l_f7NRUM2OlNDptSLXC18Q7cX8CQpUBkJtepMUJg6gYhdNM9fTBqBN08fY19HNfmoCFjN2SThT9w81tO_WWCThVBZtf8tMRUC7Bmi3jJ3HFs-4734aDGFw-LOe" />
166
+ <source media="(prefers-color-scheme: light)" srcset="https://api.star-history.com/chart?repos=cosmtrek/mindwalk&type=date&legend=top-left&sealed_token=6ylPq85HVVSbxQtqpYdSNx2EFZXMTk4AhnMG197AQm7TDwfenvf415jqPnPRxRiXz4l_f7NRUM2OlNDptSLXC18Q7cX8CQpUBkJtepMUJg6gYhdNM9fTBqBN08fY19HNfmoCFjN2SThT9w81tO_WWCThVBZtf8tMRUC7Bmi3jJ3HFs-4734aDGFw-LOe" />
167
+ <img alt="Star History Chart" src="https://api.star-history.com/chart?repos=cosmtrek/mindwalk&type=date&legend=top-left&sealed_token=6ylPq85HVVSbxQtqpYdSNx2EFZXMTk4AhnMG197AQm7TDwfenvf415jqPnPRxRiXz4l_f7NRUM2OlNDptSLXC18Q7cX8CQpUBkJtepMUJg6gYhdNM9fTBqBN08fY19HNfmoCFjN2SThT9w81tO_WWCThVBZtf8tMRUC7Bmi3jJ3HFs-4734aDGFw-LOe" />
168
+ </picture>
169
+ </a>
170
+
171
+ ## License
172
+
173
+ [MIT](LICENSE) © 2026 Ricko Yu
package/bin/synapse ADDED
Binary file
@@ -0,0 +1,308 @@
1
+ // rubriceval is the offline evaluation bench for the rubric judge pipeline
2
+ // (design doc §15, M1.5). It drives the real judge.Analyze over a batch of
3
+ // historical sessions and reports the gate metrics: rubric outcome rates,
4
+ // coverage-sufficient rate, dead-criteria rate, and per-phase latency.
5
+ //
6
+ // Usage:
7
+ //
8
+ // go run ./cmd/rubriceval -o OUTDIR [-cli codex] [-workers 3] <session.jsonl>...
9
+ //
10
+ // Every session costs real judge-CLI calls; this is a bench, not a test.
11
+ package main
12
+
13
+ import (
14
+ "context"
15
+ "encoding/json"
16
+ "flag"
17
+ "fmt"
18
+ "os"
19
+ "path/filepath"
20
+ "sort"
21
+ "strings"
22
+ "sync"
23
+ "time"
24
+
25
+ "github.com/hybridlabor-api/bdb-synapse/internal/adapter"
26
+ "github.com/hybridlabor-api/bdb-synapse/internal/adapter/claudecode"
27
+ "github.com/hybridlabor-api/bdb-synapse/internal/adapter/codex"
28
+ "github.com/hybridlabor-api/bdb-synapse/internal/judge"
29
+ "github.com/hybridlabor-api/bdb-synapse/internal/model"
30
+ )
31
+
32
+ // timingRunner wraps the real CLI runner and records each sealed call. The
33
+ // prompt text distinguishes the phases: the generation prompt announces
34
+ // itself, and the unified scoring prompt names the RUBRIC input section.
35
+ type timingRunner struct {
36
+ inner judge.Runner
37
+ // dumpDir, when set, saves every call's raw output for failure analysis.
38
+ dumpDir string
39
+ session string
40
+ mu sync.Mutex
41
+ calls []callRecord
42
+ }
43
+
44
+ type callRecord struct {
45
+ Kind string `json:"kind"` // rubric | scoring-unified | scoring-legacy
46
+ DurationSec float64 `json:"durationSec"`
47
+ InputBytes int `json:"inputBytes"`
48
+ OutputBytes int `json:"outputBytes"`
49
+ }
50
+
51
+ func (t *timingRunner) Run(ctx context.Context, prompt, input string) (judge.RunResult, error) {
52
+ start := time.Now()
53
+ result, err := t.inner.Run(ctx, prompt, input)
54
+ kind := "scoring-legacy"
55
+ switch {
56
+ case strings.Contains(prompt, "designing an evaluation rubric"):
57
+ kind = "rubric"
58
+ case strings.Contains(prompt, "RUBRIC"):
59
+ kind = "scoring-unified"
60
+ }
61
+ t.mu.Lock()
62
+ t.calls = append(t.calls, callRecord{
63
+ Kind: kind,
64
+ DurationSec: time.Since(start).Seconds(),
65
+ InputBytes: len(input),
66
+ OutputBytes: len(result.Text),
67
+ })
68
+ call := len(t.calls)
69
+ t.mu.Unlock()
70
+ if t.dumpDir != "" {
71
+ name := fmt.Sprintf("%s.call%d.%s.txt", t.session, call, kind)
72
+ _ = os.WriteFile(filepath.Join(t.dumpDir, name), []byte(result.Text), 0o644)
73
+ }
74
+ return result, err
75
+ }
76
+
77
+ func (t *timingRunner) Name() string { return t.inner.Name() }
78
+
79
+ // sessionResult is one bench row; failures carry Error and nothing else.
80
+ type sessionResult struct {
81
+ Session string `json:"session"`
82
+ Harness string `json:"harness,omitempty"`
83
+ Events int `json:"events,omitempty"`
84
+ TaskRunes int `json:"taskRunes"`
85
+ Status string `json:"status"` // scored | unavailable | no-rubric-layer | error
86
+ Reason string `json:"reason,omitempty"`
87
+ Tasks int `json:"tasks,omitempty"`
88
+ Criteria int `json:"criteria,omitempty"`
89
+ Sufficient int `json:"sufficient,omitempty"`
90
+ Partial int `json:"partial,omitempty"`
91
+ None int `json:"none,omitempty"`
92
+ Dead int `json:"dead,omitempty"` // criteria with zero findings
93
+ Verdicts map[string]int `json:"verdicts,omitempty"`
94
+ Calls []callRecord `json:"calls,omitempty"`
95
+ TotalSec float64 `json:"totalSec"`
96
+ Error string `json:"error,omitempty"`
97
+ }
98
+
99
+ func main() {
100
+ outDir := flag.String("o", "", "output directory (required)")
101
+ cliName := flag.String("cli", "codex", "judge CLI")
102
+ modelName := flag.String("model", "", "judge model override")
103
+ workers := flag.Int("workers", 3, "concurrent sessions")
104
+ dumpRaw := flag.Bool("dump-raw", false, "save every judge call's raw output next to the reports")
105
+ flag.Parse()
106
+ if *outDir == "" || flag.NArg() == 0 {
107
+ fmt.Fprintln(os.Stderr, "usage: rubriceval -o OUTDIR [-cli codex] [-workers N] <session.jsonl>...")
108
+ os.Exit(2)
109
+ }
110
+ if err := os.MkdirAll(*outDir, 0o755); err != nil {
111
+ fmt.Fprintln(os.Stderr, err)
112
+ os.Exit(1)
113
+ }
114
+
115
+ sem := make(chan struct{}, *workers)
116
+ var wg sync.WaitGroup
117
+ var mu sync.Mutex
118
+ var results []sessionResult
119
+ for _, path := range flag.Args() {
120
+ wg.Add(1)
121
+ go func(path string) {
122
+ defer wg.Done()
123
+ sem <- struct{}{}
124
+ defer func() { <-sem }()
125
+ result := evalSession(path, *cliName, *modelName, *outDir, *dumpRaw)
126
+ mu.Lock()
127
+ results = append(results, result)
128
+ done := len(results)
129
+ mu.Unlock()
130
+ fmt.Printf("[%d/%d] %s → %s%s (%.0fs)\n", done, flag.NArg(), filepath.Base(path),
131
+ result.Status, reasonSuffix(result), result.TotalSec)
132
+ }(path)
133
+ }
134
+ wg.Wait()
135
+
136
+ sort.Slice(results, func(i, j int) bool { return results[i].Session < results[j].Session })
137
+ writeJSON(filepath.Join(*outDir, "results.json"), results)
138
+ printSummary(results)
139
+ }
140
+
141
+ func reasonSuffix(r sessionResult) string {
142
+ if r.Reason == "" {
143
+ return ""
144
+ }
145
+ return "/" + r.Reason
146
+ }
147
+
148
+ func evalSession(path, cliName, modelName, outDir string, dumpRaw bool) sessionResult {
149
+ result := sessionResult{Session: filepath.Base(path)}
150
+ trace, err := parseTrace(path)
151
+ if err != nil {
152
+ result.Status = "error"
153
+ result.Error = err.Error()
154
+ return result
155
+ }
156
+ result.Harness = trace.Session.Harness
157
+ result.Events = trace.Session.EventCount
158
+ result.TaskRunes = taskRunes(trace)
159
+
160
+ runner := &timingRunner{inner: judge.CLIRunner{CLI: cliName, Model: modelName}, session: strings.TrimSuffix(result.Session, ".jsonl")}
161
+ if dumpRaw {
162
+ runner.dumpDir = outDir
163
+ }
164
+ ctx, cancel := context.WithTimeout(context.Background(), judge.DefaultTimeout)
165
+ defer cancel()
166
+ start := time.Now()
167
+ report, err := judge.Analyze(ctx, trace, judge.Options{Runner: runner})
168
+ result.TotalSec = time.Since(start).Seconds()
169
+ result.Calls = runner.calls
170
+ if err != nil {
171
+ result.Status = "error"
172
+ result.Error = err.Error()
173
+ return result
174
+ }
175
+ writeJSON(filepath.Join(outDir, strings.TrimSuffix(result.Session, ".jsonl")+".report.json"), report)
176
+
177
+ if report.Rubric == nil {
178
+ result.Status = "no-rubric-layer"
179
+ return result
180
+ }
181
+ result.Status = report.Rubric.Status
182
+ result.Reason = report.Rubric.Reason
183
+ result.Tasks = len(report.Rubric.Tasks)
184
+ result.Verdicts = map[string]int{}
185
+ for _, task := range report.Rubric.Tasks {
186
+ for _, criterion := range task.Criteria {
187
+ result.Criteria++
188
+ switch criterion.Coverage {
189
+ case model.CoverageSufficient:
190
+ result.Sufficient++
191
+ case model.CoveragePartial:
192
+ result.Partial++
193
+ case model.CoverageNone:
194
+ result.None++
195
+ }
196
+ if len(criterion.Findings) == 0 {
197
+ result.Dead++
198
+ }
199
+ result.Verdicts[criterion.Verdict]++
200
+ }
201
+ }
202
+ return result
203
+ }
204
+
205
+ // taskRunes mirrors the pipeline's weak-task-text measurement so threshold
206
+ // calibration can plot outcomes against the signal the gate actually sees.
207
+ func taskRunes(trace *model.Trace) int {
208
+ total := 0
209
+ for _, mark := range trace.Marks {
210
+ if mark.Type != "user-message" {
211
+ continue
212
+ }
213
+ text := strings.TrimSpace(mark.Note)
214
+ if text == "" || adapter.InjectedUserMessage(text) {
215
+ continue
216
+ }
217
+ total += len([]rune(text))
218
+ }
219
+ return total
220
+ }
221
+
222
+ func printSummary(results []sessionResult) {
223
+ var scored, degraded, skipped, errored, noLayer int
224
+ var criteria, sufficient, partial, none, dead int
225
+ var rubricSecs, scoringSecs, totals []float64
226
+ taskCounts := map[int]int{}
227
+ for _, r := range results {
228
+ switch {
229
+ case r.Status == "error":
230
+ errored++
231
+ continue
232
+ case r.Status == "no-rubric-layer":
233
+ noLayer++
234
+ continue
235
+ case r.Status == model.RubricStatusScored:
236
+ scored++
237
+ case r.Reason == model.RubricReasonGenerationFailed:
238
+ degraded++
239
+ default:
240
+ skipped++
241
+ }
242
+ totals = append(totals, r.TotalSec)
243
+ criteria += r.Criteria
244
+ sufficient += r.Sufficient
245
+ partial += r.Partial
246
+ none += r.None
247
+ dead += r.Dead
248
+ if r.Status == model.RubricStatusScored {
249
+ taskCounts[r.Tasks]++
250
+ }
251
+ for _, call := range r.Calls {
252
+ switch call.Kind {
253
+ case "rubric":
254
+ rubricSecs = append(rubricSecs, call.DurationSec)
255
+ case "scoring-unified":
256
+ scoringSecs = append(scoringSecs, call.DurationSec)
257
+ }
258
+ }
259
+ }
260
+ fmt.Println("\n=== M1.5 gate summary ===")
261
+ fmt.Printf("sessions: %d — scored %d, skipped %d, degraded %d, no-layer %d, error %d\n",
262
+ len(results), scored, skipped, degraded, noLayer, errored)
263
+ if criteria > 0 {
264
+ fmt.Printf("criteria: %d — coverage sufficient %d (%.0f%%), partial %d, none %d; dead %d (%.0f%%)\n",
265
+ criteria, sufficient, 100*float64(sufficient)/float64(criteria), partial, none,
266
+ dead, 100*float64(dead)/float64(criteria))
267
+ }
268
+ fmt.Printf("task-count distribution (scored sessions): %v\n", taskCounts)
269
+ fmt.Printf("latency: rubric %s, unified scoring %s, session total %s\n",
270
+ stats(rubricSecs), stats(scoringSecs), stats(totals))
271
+ }
272
+
273
+ func stats(xs []float64) string {
274
+ if len(xs) == 0 {
275
+ return "n/a"
276
+ }
277
+ sort.Float64s(xs)
278
+ sum := 0.0
279
+ for _, x := range xs {
280
+ sum += x
281
+ }
282
+ return fmt.Sprintf("median %.0fs mean %.0fs max %.0fs (n=%d)",
283
+ xs[len(xs)/2], sum/float64(len(xs)), xs[len(xs)-1], len(xs))
284
+ }
285
+
286
+ func parseTrace(path string) (*model.Trace, error) {
287
+ abs, err := filepath.Abs(path)
288
+ if err != nil {
289
+ return nil, err
290
+ }
291
+ var lastErr error
292
+ for _, source := range []adapter.Source{claudecode.Adapter{}, codex.Adapter{}} {
293
+ trace, err := source.Parse(abs)
294
+ if err == nil {
295
+ return trace, nil
296
+ }
297
+ lastErr = err
298
+ }
299
+ return nil, lastErr
300
+ }
301
+
302
+ func writeJSON(path string, v any) {
303
+ data, err := json.MarshalIndent(v, "", " ")
304
+ if err != nil {
305
+ return
306
+ }
307
+ _ = os.WriteFile(path, data, 0o644)
308
+ }