@hybridlabor-api/bdb-synapse 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +173 -0
- package/bin/synapse +0 -0
- package/cmd/rubriceval/main.go +308 -0
- package/cmd/synapse/main.go +280 -0
- package/cmd/synapse/main_test.go +16 -0
- package/go.mod +7 -0
- package/go.sum +6 -0
- package/internal/adapter/adapter.go +1117 -0
- package/internal/adapter/adapter_test.go +518 -0
- package/internal/adapter/agy/adapter.go +193 -0
- package/internal/adapter/claudecode/adapter.go +415 -0
- package/internal/adapter/claudecode/adapter_test.go +260 -0
- package/internal/adapter/claudecode/agents.go +387 -0
- package/internal/adapter/claudecode/agents_test.go +480 -0
- package/internal/adapter/claudecode/summary_inputs_test.go +19 -0
- package/internal/adapter/codex/adapter.go +919 -0
- package/internal/adapter/codex/adapter_test.go +920 -0
- package/internal/adapter/codex/agents.go +401 -0
- package/internal/adapter/codex/agents_test.go +610 -0
- package/internal/adapter/codex/summary_inputs_test.go +20 -0
- package/internal/adapter/pi/adapter.go +483 -0
- package/internal/adapter/pi/adapter_test.go +517 -0
- package/internal/citymap/builder.go +1124 -0
- package/internal/citymap/builder_test.go +818 -0
- package/internal/judge/cache.go +180 -0
- package/internal/judge/cli.go +240 -0
- package/internal/judge/cli_test.go +68 -0
- package/internal/judge/fresh_summary_test.go +37 -0
- package/internal/judge/input.go +233 -0
- package/internal/judge/judge.go +416 -0
- package/internal/judge/judge_test.go +288 -0
- package/internal/judge/prompt.go +129 -0
- package/internal/judge/rubric.go +275 -0
- package/internal/judge/rubric_test.go +642 -0
- package/internal/model/agent.go +63 -0
- package/internal/model/agent_schema_test.go +86 -0
- package/internal/model/agent_test.go +64 -0
- package/internal/model/model.go +175 -0
- package/internal/model/report.go +166 -0
- package/internal/model/stats.go +151 -0
- package/internal/model/stats_test.go +89 -0
- package/internal/model/trace_schema_test.go +67 -0
- package/internal/server/analyze.go +286 -0
- package/internal/server/analyze_test.go +297 -0
- package/internal/server/codex_index_test.go +40 -0
- package/internal/server/hardening_test.go +147 -0
- package/internal/server/reportindex.go +91 -0
- package/internal/server/reportindex_test.go +116 -0
- package/internal/server/server.go +1099 -0
- package/internal/server/server_test.go +1389 -0
- package/internal/server/static/assets/fraunces-latin-ext-standard-italic-CGbN9UgK.woff2 +0 -0
- package/internal/server/static/assets/fraunces-latin-ext-standard-normal-CJcjJNj7.woff2 +0 -0
- package/internal/server/static/assets/fraunces-latin-standard-italic-lSdLDfvT.woff2 +0 -0
- package/internal/server/static/assets/fraunces-latin-standard-normal-DihXLNYH.woff2 +0 -0
- package/internal/server/static/assets/fraunces-vietnamese-standard-italic-DxWqP7Ku.woff2 +0 -0
- package/internal/server/static/assets/fraunces-vietnamese-standard-normal-Czevyj-6.woff2 +0 -0
- package/internal/server/static/assets/index-BNoY_BiB.css +1 -0
- package/internal/server/static/assets/index-C_adLrJr.js +3 -0
- package/internal/server/static/assets/react-gcHzaSmV.js +10 -0
- package/internal/server/static/assets/schibsted-grotesk-latin-ext-wght-normal-hsMS0n0O.woff2 +0 -0
- package/internal/server/static/assets/schibsted-grotesk-latin-wght-normal-Bb8VGrTG.woff2 +0 -0
- package/internal/server/static/assets/three-DnGjZfD1.js +4012 -0
- package/internal/server/static/index.html +26 -0
- package/internal/server/tracestore.go +173 -0
- package/internal/server/tracestore_test.go +51 -0
- package/internal/textutil/truncate.go +30 -0
- package/internal/textutil/truncate_test.go +39 -0
- package/package.json +35 -0
- package/web/e2e/agent-lens.spec.ts +688 -0
- package/web/index.html +23 -0
- package/web/package-lock.json +1933 -0
- package/web/package.json +33 -0
- package/web/playwright.config.ts +24 -0
- package/web/src/App.tsx +876 -0
- package/web/src/api/client.ts +74 -0
- package/web/src/main.tsx +12 -0
- package/web/src/playback/recorder.ts +160 -0
- package/web/src/playback/reducer.ts +91 -0
- package/web/src/scene/CityScene.tsx +638 -0
- package/web/src/scene/TreeScene.tsx +656 -0
- package/web/src/scene/dirLabels.ts +145 -0
- package/web/src/scene/sceneUtils.ts +144 -0
- package/web/src/scene/textures.ts +60 -0
- package/web/src/scene/trail.ts +79 -0
- package/web/src/scene/treeLayout.ts +169 -0
- package/web/src/state/filters.ts +40 -0
- package/web/src/state/store.ts +83 -0
- package/web/src/styles.css +2565 -0
- package/web/src/types.ts +315 -0
- package/web/src/ui/AgentsPanel.tsx +376 -0
- package/web/src/ui/Dock.tsx +104 -0
- package/web/src/ui/Hud.tsx +335 -0
- package/web/src/ui/Inspector.tsx +107 -0
- package/web/src/ui/LogoMark.tsx +38 -0
- package/web/src/ui/ReportPanel.tsx +491 -0
- package/web/src/ui/SessionRail.tsx +316 -0
- package/web/src/ui/Timeline.tsx +458 -0
- package/web/src/ui/ViewPanel.tsx +45 -0
- package/web/src/ui/shortcuts.ts +4 -0
- package/web/tsconfig.json +21 -0
- package/web/vite.config.ts +28 -0
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Ricko Yu
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/README.md
ADDED
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
# <img src="assets/logo.svg" alt="" width="30" /> BDB Synapse
|
|
2
|
+
|
|
3
|
+
A visualization tool that replays coding-agent sessions on a 3D map of your codebase.
|
|
4
|
+
|
|
5
|
+
**BDB Synapse** is an extended fork of the excellent [cosmtrek/mindwalk](https://github.com/cosmtrek/mindwalk) (MIT License), created to support the BDB ecosystem of agents. It adds support for the Antigravity (agy) CLI alongside the existing supported agents.
|
|
6
|
+
|
|
7
|
+
## The problem
|
|
8
|
+
|
|
9
|
+
A session log records what an agent did, but not how it understood the task:
|
|
10
|
+
which parts of the repo it treated as relevant, where it explored before it
|
|
11
|
+
acted, whether its footprint matched the scope you had in mind. Reading the
|
|
12
|
+
raw JSONL line by line doesn't answer any of that.
|
|
13
|
+
|
|
14
|
+
## The idea
|
|
15
|
+
|
|
16
|
+
Draw the repository as a night map, and play the session back as light moving
|
|
17
|
+
through it: where the agent searched, read, and edited, the map glows —
|
|
18
|
+
everything else stays dark. The agent's understanding of the task becomes a
|
|
19
|
+
shape you can see at a glance. One Go binary reads Claude Code, Codex, pi, and Antigravity (agy)
|
|
20
|
+
session logs, fully local; viewing sends nothing anywhere. The one exception
|
|
21
|
+
is the optional session evaluation: when you explicitly run it, a summary of
|
|
22
|
+
that session (task wording, file paths, event digests) is sent to the model
|
|
23
|
+
behind your own `claude` or `codex` CLI — see
|
|
24
|
+
[Session evaluation](#session-evaluation).
|
|
25
|
+
|
|
26
|
+
## Quick start
|
|
27
|
+
|
|
28
|
+
To build from source: `make setup && make build` → `bin/synapse`.
|
|
29
|
+
|
|
30
|
+
With no arguments, Synapse scans `~/.claude/projects`, `~/.codex/sessions`,
|
|
31
|
+
`~/.pi/agent/sessions`, and `~/.gemini/antigravity-cli/brain`, serves the UI on a random local port, and opens a
|
|
32
|
+
browser:
|
|
33
|
+
|
|
34
|
+
```text
|
|
35
|
+
synapse serve [--port N] [--no-open] [--claude-dir DIR] [--codex-dir DIR] [--pi-dir DIR] [--agy-dir DIR]
|
|
36
|
+
synapse open [--no-open] <session.jsonl> open one specific session
|
|
37
|
+
synapse map [--no-open] <repo> open a repository map, no session needed
|
|
38
|
+
synapse build <repo> [-o out] write the repository citymap JSON
|
|
39
|
+
synapse trace <session> [-o out] write the normalized trace JSON
|
|
40
|
+
synapse analyze <session> [--judge claude|codex] [--model name] [--no-rubric]
|
|
41
|
+
evaluate one session (see below)
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
## Reading the picture
|
|
45
|
+
|
|
46
|
+
- **Tree / Terrain views** — the repo as a radial tree or a treemap plain;
|
|
47
|
+
glow ∝ how deeply and how often a file was touched.
|
|
48
|
+
- **Touch states** — each file keeps its deepest touch: seen (moss green),
|
|
49
|
+
read (moonlight blue), edited (warm amber), unvisited (dark). Files the
|
|
50
|
+
session touched that are no longer in the repo linger as wireframe ghosts.
|
|
51
|
+
The HUD folds friction signals — error rate, churned files, edits after the
|
|
52
|
+
last verify — into a review strip.
|
|
53
|
+
- **Playback deck** — scrub or play the session over a bucketed histogram of
|
|
54
|
+
the run. Bars sit on a cool/warm spectrum: observation stays cool (search,
|
|
55
|
+
read, exec), mutation glows warm (edit, verify), so editing phases jump out
|
|
56
|
+
at a glance. Restart, speed, and video export fold into the deck's `⋯` menu;
|
|
57
|
+
export records the playback to a `.webm` entirely client-side.
|
|
58
|
+
- **Timeline marks** — `◇` context compactions, `○` subagent launches,
|
|
59
|
+
`›` user turns; every mark is a click-to-jump target.
|
|
60
|
+
- **Agent lenses** — when a session launched subagents, the HUD carries a
|
|
61
|
+
subagent count and an agents panel: pick a lens to replay any subagent's
|
|
62
|
+
trace on the same map, then step back out to the main trace.
|
|
63
|
+
- **Inspector** — click a file to pin its visit history; click a visit row to
|
|
64
|
+
jump the playhead to that moment.
|
|
65
|
+
- **Evaluate** — ask a local agent CLI to judge the session's trajectory,
|
|
66
|
+
scored against criteria drafted from your own request; session rows carry
|
|
67
|
+
the evaluation state as a quiet badge. See
|
|
68
|
+
[Session evaluation](#session-evaluation).
|
|
69
|
+
- **Repo map** — `mindwalk map <repo>` (or the folder icon in the session
|
|
70
|
+
rail) renders any repository's citymap with no session attached; height
|
|
71
|
+
encodes lines of code instead of attention.
|
|
72
|
+
|
|
73
|
+

|
|
74
|
+
|
|
75
|
+

|
|
76
|
+
|
|
77
|
+
Keyboard: `Space` play/pause · `←`/`→` step (`⇧` ×10) · `Home`/`End` ends ·
|
|
78
|
+
`S` speed · `V` view · `E` next edit · `X` next error · `M` next mark ·
|
|
79
|
+
`⌘B` session rail.
|
|
80
|
+
|
|
81
|
+
## Session evaluation
|
|
82
|
+
|
|
83
|
+
The evaluate panel (and `mindwalk analyze`) asks a local agent CLI to judge
|
|
84
|
+
how the session went. A report has two layers:
|
|
85
|
+
|
|
86
|
+
- **Process dimensions** — exploration, scope, wandering, verification: four
|
|
87
|
+
fixed lenses, the same for every session, so reports stay comparable.
|
|
88
|
+
- **Task scorecard** — before scoring, the judge drafts criteria from your
|
|
89
|
+
own request wording: what would count as done for *this* task, grouped per
|
|
90
|
+
task when the session carried several. Each criterion is then scored
|
|
91
|
+
against the session, alongside the dimensions, in one pass.
|
|
92
|
+
|
|
93
|
+
Every finding in either layer must cite timeline events you can click
|
|
94
|
+
through to, and no verdict is the model's to decide: dimension and criterion
|
|
95
|
+
verdicts are rolled up mechanically from finding severities. When the log
|
|
96
|
+
simply can't show whether a criterion was met, its coverage drops and the
|
|
97
|
+
verdict reads "no signal" — an unverifiable criterion is a blind spot, not a
|
|
98
|
+
failure. Pick the judge (any installed CLI) and its model in the panel; the
|
|
99
|
+
report records who actually judged.
|
|
100
|
+
|
|
101
|
+
The scorecard steps aside rather than getting in the way: sessions with no
|
|
102
|
+
tool events or too little task text skip it, and a failed criteria draft
|
|
103
|
+
degrades to a dimensions-only report. `--no-rubric` (or `"rubric": false` on
|
|
104
|
+
the analyze API) skips it explicitly, in a single judge call. How the
|
|
105
|
+
scorecard is built — and why it is shaped the way it is — is covered in
|
|
106
|
+
[docs/dynamic-rubric-evaluation.md](docs/dynamic-rubric-evaluation.md).
|
|
107
|
+
|
|
108
|
+
**What leaves your machine, and only when you ask:** evaluation runs your own
|
|
109
|
+
`claude` or `codex` CLI — up to two sealed calls, one drafting criteria and
|
|
110
|
+
one scoring. Both send only that session's summary — the user messages'
|
|
111
|
+
wording, file paths, and one-line event digests — to the model behind your
|
|
112
|
+
account. Nothing is sent while viewing sessions, and no other session is
|
|
113
|
+
included. The judge subprocess runs sealed: no tools, no MCP servers, no user
|
|
114
|
+
or project settings, and no session persistence.
|
|
115
|
+
|
|
116
|
+
Reports are cached in `~/.mindwalk/reports`, one per session; a report goes
|
|
117
|
+
stale (never auto-reruns) when the session's content changes. Re-evaluating
|
|
118
|
+
a session whose task wording hasn't changed reuses the drafted criteria —
|
|
119
|
+
scores can move, the yardstick doesn't.
|
|
120
|
+
|
|
121
|
+
## Under the hood
|
|
122
|
+
|
|
123
|
+
Three artifacts, kept deliberately separate:
|
|
124
|
+
|
|
125
|
+
1. a **trace** — the session log normalized into an ordered stream of
|
|
126
|
+
file-touch events (`internal/adapter`, one adapter per agent format);
|
|
127
|
+
adapters also correlate subagent sessions into an agent graph, so each
|
|
128
|
+
subagent's trace can be replayed on its own;
|
|
129
|
+
2. a **citymap** — a deterministic layout of the repository
|
|
130
|
+
(`internal/citymap`); the same tree always produces the same map, so
|
|
131
|
+
replays are comparable across sessions;
|
|
132
|
+
3. a **report** — an LLM judge's evidence-anchored findings about one
|
|
133
|
+
session (`internal/judge`): four fixed process dimensions plus a
|
|
134
|
+
task-specific scorecard; the judge only contributes findings, verdicts
|
|
135
|
+
are always rolled up mechanically, so reports stay comparable too.
|
|
136
|
+
|
|
137
|
+
A local Go server (`internal/server`) joins them and serves the
|
|
138
|
+
React/Three.js frontend (`web`). `schema/` mirrors the exported JSON contracts.
|
|
139
|
+
|
|
140
|
+
## Contributing
|
|
141
|
+
|
|
142
|
+
Issues and pull requests are welcome. To get a working dev setup:
|
|
143
|
+
|
|
144
|
+
```sh
|
|
145
|
+
make setup # install frontend dependencies
|
|
146
|
+
make serve # dev server on :8765, serving web/dist from the working tree
|
|
147
|
+
make test # go test + frontend build — run before sending a PR
|
|
148
|
+
make build # regenerate embedded assets and bin/mindwalk
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
Ground rules (see [AGENTS.md](AGENTS.md) for the full architecture notes):
|
|
152
|
+
|
|
153
|
+
- Keep the boundaries: adapters don't know about rendering, citymap generation
|
|
154
|
+
doesn't depend on playback, the judge reads only the normalized trace, and
|
|
155
|
+
the server just connects the pieces.
|
|
156
|
+
- Keep Go code `gofmt`-ed; never hand-edit `internal/server/static` —
|
|
157
|
+
regenerate it with `make build`.
|
|
158
|
+
- When trace, citymap, or report JSON shapes change, update `schema/` and the
|
|
159
|
+
relevant tests in the same change.
|
|
160
|
+
|
|
161
|
+
## Star History
|
|
162
|
+
|
|
163
|
+
<a href="https://www.star-history.com/?repos=cosmtrek%2Fmindwalk&type=date&legend=top-left">
|
|
164
|
+
<picture>
|
|
165
|
+
<source media="(prefers-color-scheme: dark)" srcset="https://api.star-history.com/chart?repos=cosmtrek/mindwalk&type=date&theme=dark&legend=top-left&sealed_token=6ylPq85HVVSbxQtqpYdSNx2EFZXMTk4AhnMG197AQm7TDwfenvf415jqPnPRxRiXz4l_f7NRUM2OlNDptSLXC18Q7cX8CQpUBkJtepMUJg6gYhdNM9fTBqBN08fY19HNfmoCFjN2SThT9w81tO_WWCThVBZtf8tMRUC7Bmi3jJ3HFs-4734aDGFw-LOe" />
|
|
166
|
+
<source media="(prefers-color-scheme: light)" srcset="https://api.star-history.com/chart?repos=cosmtrek/mindwalk&type=date&legend=top-left&sealed_token=6ylPq85HVVSbxQtqpYdSNx2EFZXMTk4AhnMG197AQm7TDwfenvf415jqPnPRxRiXz4l_f7NRUM2OlNDptSLXC18Q7cX8CQpUBkJtepMUJg6gYhdNM9fTBqBN08fY19HNfmoCFjN2SThT9w81tO_WWCThVBZtf8tMRUC7Bmi3jJ3HFs-4734aDGFw-LOe" />
|
|
167
|
+
<img alt="Star History Chart" src="https://api.star-history.com/chart?repos=cosmtrek/mindwalk&type=date&legend=top-left&sealed_token=6ylPq85HVVSbxQtqpYdSNx2EFZXMTk4AhnMG197AQm7TDwfenvf415jqPnPRxRiXz4l_f7NRUM2OlNDptSLXC18Q7cX8CQpUBkJtepMUJg6gYhdNM9fTBqBN08fY19HNfmoCFjN2SThT9w81tO_WWCThVBZtf8tMRUC7Bmi3jJ3HFs-4734aDGFw-LOe" />
|
|
168
|
+
</picture>
|
|
169
|
+
</a>
|
|
170
|
+
|
|
171
|
+
## License
|
|
172
|
+
|
|
173
|
+
[MIT](LICENSE) © 2026 Ricko Yu
|
package/bin/synapse
ADDED
|
Binary file
|
|
@@ -0,0 +1,308 @@
|
|
|
1
|
+
// rubriceval is the offline evaluation bench for the rubric judge pipeline
|
|
2
|
+
// (design doc §15, M1.5). It drives the real judge.Analyze over a batch of
|
|
3
|
+
// historical sessions and reports the gate metrics: rubric outcome rates,
|
|
4
|
+
// coverage-sufficient rate, dead-criteria rate, and per-phase latency.
|
|
5
|
+
//
|
|
6
|
+
// Usage:
|
|
7
|
+
//
|
|
8
|
+
// go run ./cmd/rubriceval -o OUTDIR [-cli codex] [-workers 3] <session.jsonl>...
|
|
9
|
+
//
|
|
10
|
+
// Every session costs real judge-CLI calls; this is a bench, not a test.
|
|
11
|
+
package main
|
|
12
|
+
|
|
13
|
+
import (
|
|
14
|
+
"context"
|
|
15
|
+
"encoding/json"
|
|
16
|
+
"flag"
|
|
17
|
+
"fmt"
|
|
18
|
+
"os"
|
|
19
|
+
"path/filepath"
|
|
20
|
+
"sort"
|
|
21
|
+
"strings"
|
|
22
|
+
"sync"
|
|
23
|
+
"time"
|
|
24
|
+
|
|
25
|
+
"github.com/hybridlabor-api/bdb-synapse/internal/adapter"
|
|
26
|
+
"github.com/hybridlabor-api/bdb-synapse/internal/adapter/claudecode"
|
|
27
|
+
"github.com/hybridlabor-api/bdb-synapse/internal/adapter/codex"
|
|
28
|
+
"github.com/hybridlabor-api/bdb-synapse/internal/judge"
|
|
29
|
+
"github.com/hybridlabor-api/bdb-synapse/internal/model"
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
// timingRunner wraps the real CLI runner and records each sealed call. The
|
|
33
|
+
// prompt text distinguishes the phases: the generation prompt announces
|
|
34
|
+
// itself, and the unified scoring prompt names the RUBRIC input section.
|
|
35
|
+
type timingRunner struct {
|
|
36
|
+
inner judge.Runner
|
|
37
|
+
// dumpDir, when set, saves every call's raw output for failure analysis.
|
|
38
|
+
dumpDir string
|
|
39
|
+
session string
|
|
40
|
+
mu sync.Mutex
|
|
41
|
+
calls []callRecord
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
type callRecord struct {
|
|
45
|
+
Kind string `json:"kind"` // rubric | scoring-unified | scoring-legacy
|
|
46
|
+
DurationSec float64 `json:"durationSec"`
|
|
47
|
+
InputBytes int `json:"inputBytes"`
|
|
48
|
+
OutputBytes int `json:"outputBytes"`
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
func (t *timingRunner) Run(ctx context.Context, prompt, input string) (judge.RunResult, error) {
|
|
52
|
+
start := time.Now()
|
|
53
|
+
result, err := t.inner.Run(ctx, prompt, input)
|
|
54
|
+
kind := "scoring-legacy"
|
|
55
|
+
switch {
|
|
56
|
+
case strings.Contains(prompt, "designing an evaluation rubric"):
|
|
57
|
+
kind = "rubric"
|
|
58
|
+
case strings.Contains(prompt, "RUBRIC"):
|
|
59
|
+
kind = "scoring-unified"
|
|
60
|
+
}
|
|
61
|
+
t.mu.Lock()
|
|
62
|
+
t.calls = append(t.calls, callRecord{
|
|
63
|
+
Kind: kind,
|
|
64
|
+
DurationSec: time.Since(start).Seconds(),
|
|
65
|
+
InputBytes: len(input),
|
|
66
|
+
OutputBytes: len(result.Text),
|
|
67
|
+
})
|
|
68
|
+
call := len(t.calls)
|
|
69
|
+
t.mu.Unlock()
|
|
70
|
+
if t.dumpDir != "" {
|
|
71
|
+
name := fmt.Sprintf("%s.call%d.%s.txt", t.session, call, kind)
|
|
72
|
+
_ = os.WriteFile(filepath.Join(t.dumpDir, name), []byte(result.Text), 0o644)
|
|
73
|
+
}
|
|
74
|
+
return result, err
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
func (t *timingRunner) Name() string { return t.inner.Name() }
|
|
78
|
+
|
|
79
|
+
// sessionResult is one bench row; failures carry Error and nothing else.
|
|
80
|
+
type sessionResult struct {
|
|
81
|
+
Session string `json:"session"`
|
|
82
|
+
Harness string `json:"harness,omitempty"`
|
|
83
|
+
Events int `json:"events,omitempty"`
|
|
84
|
+
TaskRunes int `json:"taskRunes"`
|
|
85
|
+
Status string `json:"status"` // scored | unavailable | no-rubric-layer | error
|
|
86
|
+
Reason string `json:"reason,omitempty"`
|
|
87
|
+
Tasks int `json:"tasks,omitempty"`
|
|
88
|
+
Criteria int `json:"criteria,omitempty"`
|
|
89
|
+
Sufficient int `json:"sufficient,omitempty"`
|
|
90
|
+
Partial int `json:"partial,omitempty"`
|
|
91
|
+
None int `json:"none,omitempty"`
|
|
92
|
+
Dead int `json:"dead,omitempty"` // criteria with zero findings
|
|
93
|
+
Verdicts map[string]int `json:"verdicts,omitempty"`
|
|
94
|
+
Calls []callRecord `json:"calls,omitempty"`
|
|
95
|
+
TotalSec float64 `json:"totalSec"`
|
|
96
|
+
Error string `json:"error,omitempty"`
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
func main() {
|
|
100
|
+
outDir := flag.String("o", "", "output directory (required)")
|
|
101
|
+
cliName := flag.String("cli", "codex", "judge CLI")
|
|
102
|
+
modelName := flag.String("model", "", "judge model override")
|
|
103
|
+
workers := flag.Int("workers", 3, "concurrent sessions")
|
|
104
|
+
dumpRaw := flag.Bool("dump-raw", false, "save every judge call's raw output next to the reports")
|
|
105
|
+
flag.Parse()
|
|
106
|
+
if *outDir == "" || flag.NArg() == 0 {
|
|
107
|
+
fmt.Fprintln(os.Stderr, "usage: rubriceval -o OUTDIR [-cli codex] [-workers N] <session.jsonl>...")
|
|
108
|
+
os.Exit(2)
|
|
109
|
+
}
|
|
110
|
+
if err := os.MkdirAll(*outDir, 0o755); err != nil {
|
|
111
|
+
fmt.Fprintln(os.Stderr, err)
|
|
112
|
+
os.Exit(1)
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
sem := make(chan struct{}, *workers)
|
|
116
|
+
var wg sync.WaitGroup
|
|
117
|
+
var mu sync.Mutex
|
|
118
|
+
var results []sessionResult
|
|
119
|
+
for _, path := range flag.Args() {
|
|
120
|
+
wg.Add(1)
|
|
121
|
+
go func(path string) {
|
|
122
|
+
defer wg.Done()
|
|
123
|
+
sem <- struct{}{}
|
|
124
|
+
defer func() { <-sem }()
|
|
125
|
+
result := evalSession(path, *cliName, *modelName, *outDir, *dumpRaw)
|
|
126
|
+
mu.Lock()
|
|
127
|
+
results = append(results, result)
|
|
128
|
+
done := len(results)
|
|
129
|
+
mu.Unlock()
|
|
130
|
+
fmt.Printf("[%d/%d] %s → %s%s (%.0fs)\n", done, flag.NArg(), filepath.Base(path),
|
|
131
|
+
result.Status, reasonSuffix(result), result.TotalSec)
|
|
132
|
+
}(path)
|
|
133
|
+
}
|
|
134
|
+
wg.Wait()
|
|
135
|
+
|
|
136
|
+
sort.Slice(results, func(i, j int) bool { return results[i].Session < results[j].Session })
|
|
137
|
+
writeJSON(filepath.Join(*outDir, "results.json"), results)
|
|
138
|
+
printSummary(results)
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
func reasonSuffix(r sessionResult) string {
|
|
142
|
+
if r.Reason == "" {
|
|
143
|
+
return ""
|
|
144
|
+
}
|
|
145
|
+
return "/" + r.Reason
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
func evalSession(path, cliName, modelName, outDir string, dumpRaw bool) sessionResult {
|
|
149
|
+
result := sessionResult{Session: filepath.Base(path)}
|
|
150
|
+
trace, err := parseTrace(path)
|
|
151
|
+
if err != nil {
|
|
152
|
+
result.Status = "error"
|
|
153
|
+
result.Error = err.Error()
|
|
154
|
+
return result
|
|
155
|
+
}
|
|
156
|
+
result.Harness = trace.Session.Harness
|
|
157
|
+
result.Events = trace.Session.EventCount
|
|
158
|
+
result.TaskRunes = taskRunes(trace)
|
|
159
|
+
|
|
160
|
+
runner := &timingRunner{inner: judge.CLIRunner{CLI: cliName, Model: modelName}, session: strings.TrimSuffix(result.Session, ".jsonl")}
|
|
161
|
+
if dumpRaw {
|
|
162
|
+
runner.dumpDir = outDir
|
|
163
|
+
}
|
|
164
|
+
ctx, cancel := context.WithTimeout(context.Background(), judge.DefaultTimeout)
|
|
165
|
+
defer cancel()
|
|
166
|
+
start := time.Now()
|
|
167
|
+
report, err := judge.Analyze(ctx, trace, judge.Options{Runner: runner})
|
|
168
|
+
result.TotalSec = time.Since(start).Seconds()
|
|
169
|
+
result.Calls = runner.calls
|
|
170
|
+
if err != nil {
|
|
171
|
+
result.Status = "error"
|
|
172
|
+
result.Error = err.Error()
|
|
173
|
+
return result
|
|
174
|
+
}
|
|
175
|
+
writeJSON(filepath.Join(outDir, strings.TrimSuffix(result.Session, ".jsonl")+".report.json"), report)
|
|
176
|
+
|
|
177
|
+
if report.Rubric == nil {
|
|
178
|
+
result.Status = "no-rubric-layer"
|
|
179
|
+
return result
|
|
180
|
+
}
|
|
181
|
+
result.Status = report.Rubric.Status
|
|
182
|
+
result.Reason = report.Rubric.Reason
|
|
183
|
+
result.Tasks = len(report.Rubric.Tasks)
|
|
184
|
+
result.Verdicts = map[string]int{}
|
|
185
|
+
for _, task := range report.Rubric.Tasks {
|
|
186
|
+
for _, criterion := range task.Criteria {
|
|
187
|
+
result.Criteria++
|
|
188
|
+
switch criterion.Coverage {
|
|
189
|
+
case model.CoverageSufficient:
|
|
190
|
+
result.Sufficient++
|
|
191
|
+
case model.CoveragePartial:
|
|
192
|
+
result.Partial++
|
|
193
|
+
case model.CoverageNone:
|
|
194
|
+
result.None++
|
|
195
|
+
}
|
|
196
|
+
if len(criterion.Findings) == 0 {
|
|
197
|
+
result.Dead++
|
|
198
|
+
}
|
|
199
|
+
result.Verdicts[criterion.Verdict]++
|
|
200
|
+
}
|
|
201
|
+
}
|
|
202
|
+
return result
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
// taskRunes mirrors the pipeline's weak-task-text measurement so threshold
|
|
206
|
+
// calibration can plot outcomes against the signal the gate actually sees.
|
|
207
|
+
func taskRunes(trace *model.Trace) int {
|
|
208
|
+
total := 0
|
|
209
|
+
for _, mark := range trace.Marks {
|
|
210
|
+
if mark.Type != "user-message" {
|
|
211
|
+
continue
|
|
212
|
+
}
|
|
213
|
+
text := strings.TrimSpace(mark.Note)
|
|
214
|
+
if text == "" || adapter.InjectedUserMessage(text) {
|
|
215
|
+
continue
|
|
216
|
+
}
|
|
217
|
+
total += len([]rune(text))
|
|
218
|
+
}
|
|
219
|
+
return total
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
func printSummary(results []sessionResult) {
|
|
223
|
+
var scored, degraded, skipped, errored, noLayer int
|
|
224
|
+
var criteria, sufficient, partial, none, dead int
|
|
225
|
+
var rubricSecs, scoringSecs, totals []float64
|
|
226
|
+
taskCounts := map[int]int{}
|
|
227
|
+
for _, r := range results {
|
|
228
|
+
switch {
|
|
229
|
+
case r.Status == "error":
|
|
230
|
+
errored++
|
|
231
|
+
continue
|
|
232
|
+
case r.Status == "no-rubric-layer":
|
|
233
|
+
noLayer++
|
|
234
|
+
continue
|
|
235
|
+
case r.Status == model.RubricStatusScored:
|
|
236
|
+
scored++
|
|
237
|
+
case r.Reason == model.RubricReasonGenerationFailed:
|
|
238
|
+
degraded++
|
|
239
|
+
default:
|
|
240
|
+
skipped++
|
|
241
|
+
}
|
|
242
|
+
totals = append(totals, r.TotalSec)
|
|
243
|
+
criteria += r.Criteria
|
|
244
|
+
sufficient += r.Sufficient
|
|
245
|
+
partial += r.Partial
|
|
246
|
+
none += r.None
|
|
247
|
+
dead += r.Dead
|
|
248
|
+
if r.Status == model.RubricStatusScored {
|
|
249
|
+
taskCounts[r.Tasks]++
|
|
250
|
+
}
|
|
251
|
+
for _, call := range r.Calls {
|
|
252
|
+
switch call.Kind {
|
|
253
|
+
case "rubric":
|
|
254
|
+
rubricSecs = append(rubricSecs, call.DurationSec)
|
|
255
|
+
case "scoring-unified":
|
|
256
|
+
scoringSecs = append(scoringSecs, call.DurationSec)
|
|
257
|
+
}
|
|
258
|
+
}
|
|
259
|
+
}
|
|
260
|
+
fmt.Println("\n=== M1.5 gate summary ===")
|
|
261
|
+
fmt.Printf("sessions: %d — scored %d, skipped %d, degraded %d, no-layer %d, error %d\n",
|
|
262
|
+
len(results), scored, skipped, degraded, noLayer, errored)
|
|
263
|
+
if criteria > 0 {
|
|
264
|
+
fmt.Printf("criteria: %d — coverage sufficient %d (%.0f%%), partial %d, none %d; dead %d (%.0f%%)\n",
|
|
265
|
+
criteria, sufficient, 100*float64(sufficient)/float64(criteria), partial, none,
|
|
266
|
+
dead, 100*float64(dead)/float64(criteria))
|
|
267
|
+
}
|
|
268
|
+
fmt.Printf("task-count distribution (scored sessions): %v\n", taskCounts)
|
|
269
|
+
fmt.Printf("latency: rubric %s, unified scoring %s, session total %s\n",
|
|
270
|
+
stats(rubricSecs), stats(scoringSecs), stats(totals))
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
func stats(xs []float64) string {
|
|
274
|
+
if len(xs) == 0 {
|
|
275
|
+
return "n/a"
|
|
276
|
+
}
|
|
277
|
+
sort.Float64s(xs)
|
|
278
|
+
sum := 0.0
|
|
279
|
+
for _, x := range xs {
|
|
280
|
+
sum += x
|
|
281
|
+
}
|
|
282
|
+
return fmt.Sprintf("median %.0fs mean %.0fs max %.0fs (n=%d)",
|
|
283
|
+
xs[len(xs)/2], sum/float64(len(xs)), xs[len(xs)-1], len(xs))
|
|
284
|
+
}
|
|
285
|
+
|
|
286
|
+
func parseTrace(path string) (*model.Trace, error) {
|
|
287
|
+
abs, err := filepath.Abs(path)
|
|
288
|
+
if err != nil {
|
|
289
|
+
return nil, err
|
|
290
|
+
}
|
|
291
|
+
var lastErr error
|
|
292
|
+
for _, source := range []adapter.Source{claudecode.Adapter{}, codex.Adapter{}} {
|
|
293
|
+
trace, err := source.Parse(abs)
|
|
294
|
+
if err == nil {
|
|
295
|
+
return trace, nil
|
|
296
|
+
}
|
|
297
|
+
lastErr = err
|
|
298
|
+
}
|
|
299
|
+
return nil, lastErr
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
func writeJSON(path string, v any) {
|
|
303
|
+
data, err := json.MarshalIndent(v, "", " ")
|
|
304
|
+
if err != nil {
|
|
305
|
+
return
|
|
306
|
+
}
|
|
307
|
+
_ = os.WriteFile(path, data, 0o644)
|
|
308
|
+
}
|