pi-ui-extend 1.0.39 → 1.0.40
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/external/pi-tools-suite/README.md +207 -0
- package/external/pi-tools-suite/docs/evals.md +684 -0
- package/external/pi-tools-suite/package.json +7 -3
- package/external/pi-tools-suite/src/coding-discipline/index.ts +41 -142
- package/external/pi-tools-suite/src/config.ts +0 -21
- package/external/pi-tools-suite/src/default-pi-tools-suite-config.ts +23 -6
- package/external/pi-tools-suite/src/tool-descriptions.ts +4 -3
- package/package.json +4 -5
|
@@ -0,0 +1,684 @@
|
|
|
1
|
+
# Evaluation framework
|
|
2
|
+
|
|
3
|
+
## Purpose
|
|
4
|
+
|
|
5
|
+
The eval framework answers four different questions that ordinary unit tests do
|
|
6
|
+
not answer equally well:
|
|
7
|
+
|
|
8
|
+
1. Does every extension and model-facing tool still satisfy its deterministic
|
|
9
|
+
contract?
|
|
10
|
+
2. Does a live model choose the right tool for the task instead of merely being
|
|
11
|
+
capable of calling it?
|
|
12
|
+
3. Does the model follow a good coding workflow from investigation through
|
|
13
|
+
verification, rather than reaching the right answer by a fragile sequence of
|
|
14
|
+
guesses and patches?
|
|
15
|
+
4. Does orchestration improve quality or cost without introducing unnecessary
|
|
16
|
+
delegation, retries, context growth, or latency?
|
|
17
|
+
|
|
18
|
+
The framework deliberately does **not** reduce all of these questions to one LLM
|
|
19
|
+
judge score. Correctness and safety invariants are machine-checked wherever
|
|
20
|
+
possible. Live-model evals are reserved for choices and workflows that cannot be
|
|
21
|
+
proved by deterministic tests alone.
|
|
22
|
+
|
|
23
|
+
The implementation lives under `test/evals/` and complements the existing unit,
|
|
24
|
+
integration, prompt-eval, browser-QA, and locate-benchmark suites.
|
|
25
|
+
|
|
26
|
+
## Design principles
|
|
27
|
+
|
|
28
|
+
### Deterministic facts are deterministic tests
|
|
29
|
+
|
|
30
|
+
Schemas, state transitions, routing precedence, persistence, retries, fallbacks,
|
|
31
|
+
security boundaries, parsing, cleanup, and other mechanical behavior should not
|
|
32
|
+
depend on a probabilistic judge. Existing focused tests remain the source of
|
|
33
|
+
truth for these contracts.
|
|
34
|
+
|
|
35
|
+
### Live evals test model behavior
|
|
36
|
+
|
|
37
|
+
Live cases are used for questions such as:
|
|
38
|
+
|
|
39
|
+
- should the model choose `repo_search` or a direct read?
|
|
40
|
+
- should a non-trivial task create a todo plan?
|
|
41
|
+
- should an expensive Sol parent delegate substantial implementation work?
|
|
42
|
+
- should Terra escalate a high-risk architecture decision?
|
|
43
|
+
- should a narrow local failure stay local instead of calling an oracle?
|
|
44
|
+
- did a coding agent reproduce the bug before editing and verify behavior after
|
|
45
|
+
editing?
|
|
46
|
+
|
|
47
|
+
### Outcome beats narration
|
|
48
|
+
|
|
49
|
+
Coding-quality cases use executable broken repositories and post-run checks.
|
|
50
|
+
Passing typecheck, build, or lint alone is insufficient when the requested
|
|
51
|
+
behavior is still wrong.
|
|
52
|
+
|
|
53
|
+
### Negative controls are first-class
|
|
54
|
+
|
|
55
|
+
The suite checks both when a tool **should** be called and when it **should
|
|
56
|
+
not**
|
|
57
|
+
be called. This is important for cost control: an agent that delegates every
|
|
58
|
+
task can look capable while still being materially worse.
|
|
59
|
+
|
|
60
|
+
### Cost is measured separately from correctness
|
|
61
|
+
|
|
62
|
+
Correctness and safety are hard gates. Token usage, model cost, tool calls, and
|
|
63
|
+
wall-clock time are comparison metrics. A cheap failure is still a failure, and
|
|
64
|
+
a correct solution that becomes dramatically more expensive should be visible
|
|
65
|
+
in the report rather than hidden inside one aggregate score.
|
|
66
|
+
|
|
67
|
+
## Framework layout
|
|
68
|
+
|
|
69
|
+
```text
|
|
70
|
+
test/evals/
|
|
71
|
+
cases.ts
|
|
72
|
+
coverage-manifest.ts
|
|
73
|
+
extension-contracts.test.ts
|
|
74
|
+
harness.test.ts
|
|
75
|
+
live-evals.test.ts
|
|
76
|
+
run-evals.ts
|
|
77
|
+
fixtures/
|
|
78
|
+
coding-hypotheses/
|
|
79
|
+
coding-regression/
|
|
80
|
+
coding-async/
|
|
81
|
+
harness/
|
|
82
|
+
types.ts
|
|
83
|
+
runner.ts
|
|
84
|
+
assertions.ts
|
|
85
|
+
metrics.ts
|
|
86
|
+
report.ts
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
The responsibilities are intentionally separated:
|
|
90
|
+
|
|
91
|
+
- `cases.ts` contains live behavioral scenarios and their assertions.
|
|
92
|
+
- `coverage-manifest.ts` maps every extension and model-facing tool to the tests
|
|
93
|
+
that cover it.
|
|
94
|
+
- `extension-contracts.test.ts` enforces coverage completeness and fills a few
|
|
95
|
+
previously missing deterministic registration contracts.
|
|
96
|
+
- `harness.test.ts` tests the eval machinery itself.
|
|
97
|
+
- `live-evals.test.ts` exposes the live cases through Bun's test runner.
|
|
98
|
+
- `run-evals.ts` runs a matrix and writes JSON/Markdown comparison artifacts.
|
|
99
|
+
- `runner.ts` creates isolated fixture projects, launches Pi, records events,
|
|
100
|
+
snapshots changed files, and collects metrics.
|
|
101
|
+
- `assertions.ts` converts case expectations into machine-checkable results.
|
|
102
|
+
- `metrics.ts` derives workflow, token, subagent, mutation, and timing metrics.
|
|
103
|
+
- `report.ts` produces the cross-model comparison report.
|
|
104
|
+
|
|
105
|
+
## Coverage gate
|
|
106
|
+
|
|
107
|
+
`coverage-manifest.ts` is a registry of deterministic and optional live coverage
|
|
108
|
+
for the suite.
|
|
109
|
+
|
|
110
|
+
Two invariants are enforced by `extension-contracts.test.ts`:
|
|
111
|
+
|
|
112
|
+
1. every extension registered in `src/index.ts` must have deterministic eval
|
|
113
|
+
coverage;
|
|
114
|
+
2. every model-facing tool described by the suite must have deterministic eval
|
|
115
|
+
coverage.
|
|
116
|
+
|
|
117
|
+
This is intentionally a maintenance gate. Adding a new extension or tool without
|
|
118
|
+
assigning it a contract causes the eval contract test to fail.
|
|
119
|
+
|
|
120
|
+
The current extension registry covers all 19 modules:
|
|
121
|
+
|
|
122
|
+
| Extension | Deterministic coverage | Representative live coverage |
|
|
123
|
+
| --- | --- | --- |
|
|
124
|
+
| `coding-discipline` | coding-discipline tests | coding-quality workflows |
|
|
125
|
+
| `ast-grep` | ast-grep tests | structural tool selection |
|
|
126
|
+
| `async-subagents` | core/tools/UI tests | Sol/Luna/Terra orchestration |
|
|
127
|
+
| `lsp` | LSP tests | deterministic only |
|
|
128
|
+
| `comment-checker` | comment-checker tests | deterministic only |
|
|
129
|
+
| `session-name` | session-name tests | deterministic only |
|
|
130
|
+
| `session-recovery` | recovery tests | overview-first recovery |
|
|
131
|
+
| `repo-discovery` | repo-discovery tests | search/architecture selection |
|
|
132
|
+
| `antigravity-auth` | provider/auth tests | deterministic only |
|
|
133
|
+
| `opencode-import` | import tests | deterministic only |
|
|
134
|
+
| `todo` | todo + persistence tests | plan creation / negative control |
|
|
135
|
+
| `model-tools` | alias/profile tests | direct small-task selection |
|
|
136
|
+
| `usage` | eval extension contracts | deterministic only |
|
|
137
|
+
| `web-search` | web-search tests | deterministic only |
|
|
138
|
+
| `dcp` | DCP prompt/pruning/state tests | existing prompt evals |
|
|
139
|
+
| `prompt-commands` | eval extension contracts | deterministic only |
|
|
140
|
+
| `skill-installer` | eval extension contracts | deterministic only |
|
|
141
|
+
| `credential-firewall` | firewall tests | deterministic only |
|
|
142
|
+
| `codex-reasoning-fix` | reasoning-fix tests | deterministic only |
|
|
143
|
+
|
|
144
|
+
The tool registry similarly covers `lookup`, `ast_grep`, `ast_apply`, all
|
|
145
|
+
subagent actions, all `repo_*` tools, `todo`, session tools, web tools, Claude
|
|
146
|
+
aliases, Codex aliases, and `compress`.
|
|
147
|
+
|
|
148
|
+
## Live case model
|
|
149
|
+
|
|
150
|
+
Each live case is an `EvalCase` with:
|
|
151
|
+
|
|
152
|
+
- a stable `id`;
|
|
153
|
+
- a category;
|
|
154
|
+
- a human description;
|
|
155
|
+
- a fixture project;
|
|
156
|
+
- a user prompt;
|
|
157
|
+
- optional model filters;
|
|
158
|
+
- optional fake indexed-repository support;
|
|
159
|
+
- optional tools that should be recorded but not executed;
|
|
160
|
+
- machine-checkable assertions;
|
|
161
|
+
- an optional custom validator for structured tool input.
|
|
162
|
+
|
|
163
|
+
Conceptually:
|
|
164
|
+
|
|
165
|
+
```ts
|
|
166
|
+
{
|
|
167
|
+
id: "quality.behavior-over-structural-check",
|
|
168
|
+
category: "coding-quality",
|
|
169
|
+
fixture: "coding-regression",
|
|
170
|
+
prompt: "Fix the rounding bug, reproduce it first, then verify behavior...",
|
|
171
|
+
assert: {
|
|
172
|
+
requireReproBeforeMutation: true,
|
|
173
|
+
requireVerificationAfterMutation: true,
|
|
174
|
+
maxMutations: 2,
|
|
175
|
+
maxFilesChanged: 1,
|
|
176
|
+
postRunCommands: [
|
|
177
|
+
{ command: "npm test" },
|
|
178
|
+
{ command: "node ...hidden-style boundary check..." },
|
|
179
|
+
],
|
|
180
|
+
},
|
|
181
|
+
}
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
The prompt should describe user intent rather than naming the expected tool
|
|
185
|
+
unless naming the tool is itself part of the contract. This prevents a case from
|
|
186
|
+
passing merely because the answer was embedded in the prompt.
|
|
187
|
+
|
|
188
|
+
## Initial 20-case corpus
|
|
189
|
+
|
|
190
|
+
### Tool selection
|
|
191
|
+
|
|
192
|
+
- `tool.semantic-repo-search`: an unknown behavior owner should select
|
|
193
|
+
`repo_search` before direct discovery.
|
|
194
|
+
- `tool.architecture-first`: a broad unfamiliar-repository overview should
|
|
195
|
+
select `repo_architecture`.
|
|
196
|
+
- `tool.exact-literal-direct`: an exact literal lookup should avoid semantic
|
|
197
|
+
and architecture-discovery overhead.
|
|
198
|
+
- `tool.ast-structural`: a syntax-aware structural query should select
|
|
199
|
+
`ast_grep`.
|
|
200
|
+
- `tool.todo-plan`: non-trivial four-stage work should initialize synchronized
|
|
201
|
+
todo state.
|
|
202
|
+
- `tool.session-recovery-overview`: lost context with no reliable search phrase
|
|
203
|
+
should begin with `session_overview`.
|
|
204
|
+
|
|
205
|
+
### Coding quality
|
|
206
|
+
|
|
207
|
+
- `quality.two-hypotheses-before-fix`: reproduce and distinguish plausible
|
|
208
|
+
causes before mutation.
|
|
209
|
+
- `quality.behavior-over-structural-check`: verify the behavioral claim instead
|
|
210
|
+
of stopping at structural validity.
|
|
211
|
+
- `quality.async-stale-state`: cover stale async completion behavior, not only
|
|
212
|
+
the success path.
|
|
213
|
+
- `quality.investigate-without-edit-probes`: investigate with evidence instead
|
|
214
|
+
of using edits as diagnostic probes.
|
|
215
|
+
- `quality.counterexample-preserves-validation`: preserve nearby validation and
|
|
216
|
+
explicitly check a boundary or counterexample.
|
|
217
|
+
|
|
218
|
+
### Orchestration and escalation
|
|
219
|
+
|
|
220
|
+
- `orchestration.sol-delegates-substantial` (Sol): the expensive parent should
|
|
221
|
+
delegate substantial multi-file work.
|
|
222
|
+
- `orchestration.sol-keeps-tiny-edit` (Sol): a tiny known-file edit should stay
|
|
223
|
+
in the parent.
|
|
224
|
+
- `orchestration.luna-delegates-substantial` (Luna): substantial implementation
|
|
225
|
+
or research should use the configured worker tier.
|
|
226
|
+
- `orchestration.luna-escalates-high-risk` (Luna): high-risk cross-module or
|
|
227
|
+
security work should trigger stronger review/deep escalation.
|
|
228
|
+
- `orchestration.terra-escalates-high-risk` (Terra): high-risk architecture or
|
|
229
|
+
security uncertainty should escalate to stronger roles.
|
|
230
|
+
|
|
231
|
+
### Negative controls
|
|
232
|
+
|
|
233
|
+
- `negative.trivial-chat-no-tools`: prevents todo/compress/subagent overhead for
|
|
234
|
+
trivial chat.
|
|
235
|
+
- `negative.known-file-read-no-oracle`: prevents planning/search/oracle
|
|
236
|
+
overhead for one known-file fact.
|
|
237
|
+
- `negative.routine-local-failure-no-oracle`: prevents escalation for a
|
|
238
|
+
straightforward locally verifiable test failure.
|
|
239
|
+
- `negative.small-known-edit-no-plan`: prevents todo/repo-search/subagent
|
|
240
|
+
overhead for a tiny known-file edit.
|
|
241
|
+
|
|
242
|
+
## Coding-quality fixtures
|
|
243
|
+
|
|
244
|
+
The quality fixtures are deliberately small enough to understand but broken in
|
|
245
|
+
ways that distinguish disciplined debugging from lucky patching.
|
|
246
|
+
|
|
247
|
+
### `coding-hypotheses`
|
|
248
|
+
|
|
249
|
+
Symptom: a second checkout for the same user can reuse the first checkout's
|
|
250
|
+
idempotency key.
|
|
251
|
+
|
|
252
|
+
At least two plausible explanations exist from the task description:
|
|
253
|
+
|
|
254
|
+
- the retry-key cache identity is too broad;
|
|
255
|
+
- callers may be passing the wrong checkout identifier.
|
|
256
|
+
|
|
257
|
+
The baseline test fails. The intended workflow is to inspect/reproduce, gather
|
|
258
|
+
evidence that separates the hypotheses, make the smallest fix, and verify both
|
|
259
|
+
stable retry behavior and separate checkout identities.
|
|
260
|
+
|
|
261
|
+
### `coding-regression`
|
|
262
|
+
|
|
263
|
+
Symptom: percentage discounts can produce a one-cent overcharge because the
|
|
264
|
+
fixture rounds where the contract requires flooring.
|
|
265
|
+
|
|
266
|
+
The function also contains input validation that must not be weakened. This
|
|
267
|
+
fixture catches agents that make a superficially plausible arithmetic change
|
|
268
|
+
without checking behavioral boundaries or regression safety.
|
|
269
|
+
|
|
270
|
+
### `coding-async`
|
|
271
|
+
|
|
272
|
+
Symptom: an older profile request can resolve after a newer request and
|
|
273
|
+
overwrite
|
|
274
|
+
the latest state.
|
|
275
|
+
|
|
276
|
+
The public contract still requires each `load(userId)` call to resolve with its
|
|
277
|
+
own profile. The fix therefore has to distinguish return-value correctness from
|
|
278
|
+
shared-state freshness and verify the stale completion path explicitly.
|
|
279
|
+
|
|
280
|
+
All three fixture test suites are expected to fail before the agent changes the
|
|
281
|
+
repository. A fixture that starts green is not a valid bug-fix eval.
|
|
282
|
+
|
|
283
|
+
## Isolation and runner lifecycle
|
|
284
|
+
|
|
285
|
+
For each `(case, model)` pair the runner:
|
|
286
|
+
|
|
287
|
+
1. copies the selected fixture into a fresh temporary project;
|
|
288
|
+
2. snapshots ordinary project files before execution;
|
|
289
|
+
3. creates an isolated `.pi` session directory;
|
|
290
|
+
4. optionally marks the project as indexed and installs a deterministic fake
|
|
291
|
+
`idx` executable for repository-selection cases;
|
|
292
|
+
5. writes a recorder extension that captures tool calls/results and final usage;
|
|
293
|
+
6. starts a real `pi` subprocess with the requested model and suite extension;
|
|
294
|
+
7. waits for completion or the per-case timeout;
|
|
295
|
+
8. snapshots the project again and derives changed files;
|
|
296
|
+
9. computes workflow/cost metrics;
|
|
297
|
+
10. evaluates assertions and post-run behavioral checks;
|
|
298
|
+
11. deletes the temporary project unless retention was requested.
|
|
299
|
+
|
|
300
|
+
The subprocess disables unrelated extension/skill/template/theme discovery so
|
|
301
|
+
the case measures this suite rather than ambient user configuration.
|
|
302
|
+
|
|
303
|
+
### Recorded-but-blocked tools
|
|
304
|
+
|
|
305
|
+
Some selection cases need to prove that a model **chose** a tool without paying
|
|
306
|
+
for or executing the next layer of work. `blockTools` records the call and then
|
|
307
|
+
returns a controlled block result.
|
|
308
|
+
|
|
309
|
+
Examples:
|
|
310
|
+
|
|
311
|
+
- `ast_grep` can be blocked after structural selection is observed;
|
|
312
|
+
- `subagents` can be blocked after delegation/escalation selection is observed.
|
|
313
|
+
|
|
314
|
+
This keeps selection evals focused and prevents nested model calls from making a
|
|
315
|
+
simple routing case unnecessarily expensive.
|
|
316
|
+
|
|
317
|
+
## Assertions
|
|
318
|
+
|
|
319
|
+
The shared assertion layer currently supports:
|
|
320
|
+
|
|
321
|
+
- required tools;
|
|
322
|
+
- forbidden tools;
|
|
323
|
+
- exact first tool;
|
|
324
|
+
- first tool from an allowed set;
|
|
325
|
+
- maximum tool-call count;
|
|
326
|
+
- maximum mutation count;
|
|
327
|
+
- maximum changed-file count;
|
|
328
|
+
- reproduction/verification before the first mutation;
|
|
329
|
+
- behavioral verification after the last mutation;
|
|
330
|
+
- required/forbidden output text;
|
|
331
|
+
- post-run commands with expected exit status;
|
|
332
|
+
- required post-run output text;
|
|
333
|
+
- custom case-specific validation.
|
|
334
|
+
|
|
335
|
+
The todo-plan case uses custom validation to check structured todo input: one
|
|
336
|
+
`batch_create`, four stages, exactly one `in_progress` item, and an explicit
|
|
337
|
+
final
|
|
338
|
+
report task.
|
|
339
|
+
|
|
340
|
+
## Workflow inference
|
|
341
|
+
|
|
342
|
+
The harness does not inspect private chain-of-thought. It infers observable
|
|
343
|
+
workflow from tool events.
|
|
344
|
+
|
|
345
|
+
Mutation tools currently include:
|
|
346
|
+
|
|
347
|
+
```text
|
|
348
|
+
edit / Edit
|
|
349
|
+
write / Write
|
|
350
|
+
apply_patch
|
|
351
|
+
ast_apply
|
|
352
|
+
```
|
|
353
|
+
|
|
354
|
+
Verification calls are shell-like tool calls whose commands visibly run tests,
|
|
355
|
+
checks, lint, typecheck, or similar verification commands.
|
|
356
|
+
|
|
357
|
+
This makes assertions such as `requireReproBeforeMutation` and
|
|
358
|
+
`requireVerificationAfterMutation` auditable without depending on model-written
|
|
359
|
+
prose.
|
|
360
|
+
|
|
361
|
+
The heuristic is intentionally conservative. If a new verification mechanism is
|
|
362
|
+
introduced, extend `metrics.ts` and add a harness regression test instead of
|
|
363
|
+
silently assuming the metric still means the same thing.
|
|
364
|
+
|
|
365
|
+
## Metrics
|
|
366
|
+
|
|
367
|
+
Each result records:
|
|
368
|
+
|
|
369
|
+
| Metric | Meaning |
|
|
370
|
+
| --- | --- |
|
|
371
|
+
| `elapsedMs` | wall-clock runtime for the model case |
|
|
372
|
+
| `toolCallCount` | total captured tool calls |
|
|
373
|
+
| `toolCalls` | ordered tool names |
|
|
374
|
+
| `failedToolResults` | tool results marked as errors |
|
|
375
|
+
| `mutationCount` | number of recognized mutation tool calls |
|
|
376
|
+
| `verificationCount` | number of recognized verification commands |
|
|
377
|
+
| `changedFiles` | project files whose content changed |
|
|
378
|
+
| `parentUsage` | parent token usage and provider-reported cost |
|
|
379
|
+
| `subagentUsage` | aggregate worker/subagent usage |
|
|
380
|
+
| `subagentCount` | number of detected subagent workspaces |
|
|
381
|
+
|
|
382
|
+
Parent usage is captured from the final `agent_end` event when available, with
|
|
383
|
+
session artifacts as a fallback. Subagent usage is collected from subagent
|
|
384
|
+
session artifacts under `.pi/subagents/`.
|
|
385
|
+
|
|
386
|
+
Provider cost fields are reported when the provider supplies them. A zero cost
|
|
387
|
+
does not necessarily mean a request was free; it may mean that the active
|
|
388
|
+
provider does not populate monetary cost metadata.
|
|
389
|
+
|
|
390
|
+
## Reports
|
|
391
|
+
|
|
392
|
+
`npm run evals:report` writes:
|
|
393
|
+
|
|
394
|
+
```text
|
|
395
|
+
test/evals/artifacts/<timestamp>/eval-report.json
|
|
396
|
+
test/evals/artifacts/<timestamp>/eval-report.md
|
|
397
|
+
```
|
|
398
|
+
|
|
399
|
+
The default artifact directory is ignored by Git.
|
|
400
|
+
|
|
401
|
+
The Markdown report contains a model summary table with:
|
|
402
|
+
|
|
403
|
+
- passed cases / total cases;
|
|
404
|
+
- parent tokens;
|
|
405
|
+
- worker tokens;
|
|
406
|
+
- provider-reported cost;
|
|
407
|
+
- tool calls;
|
|
408
|
+
- elapsed time.
|
|
409
|
+
|
|
410
|
+
It also lists every case with its tool sequence, changed files, token split, and
|
|
411
|
+
runtime. Failed assertions are expanded in a dedicated failure section.
|
|
412
|
+
|
|
413
|
+
The JSON report preserves the structured results for later statistical or CI
|
|
414
|
+
analysis.
|
|
415
|
+
|
|
416
|
+
## Running evals
|
|
417
|
+
|
|
418
|
+
### Deterministic coverage and harness tests
|
|
419
|
+
|
|
420
|
+
Run these on normal development changes:
|
|
421
|
+
|
|
422
|
+
```bash
|
|
423
|
+
npm run test:evals:contracts
|
|
424
|
+
```
|
|
425
|
+
|
|
426
|
+
This runs both the extension/tool coverage gate and tests for the harness
|
|
427
|
+
itself.
|
|
428
|
+
It does not call a live model.
|
|
429
|
+
|
|
430
|
+
The same tests are also included in the normal `npm test` traversal because they
|
|
431
|
+
live under `test/`.
|
|
432
|
+
|
|
433
|
+
### Live matrix through Bun tests
|
|
434
|
+
|
|
435
|
+
Set one or more models:
|
|
436
|
+
|
|
437
|
+
```bash
|
|
438
|
+
PI_TOOLS_SUITE_EVAL_MODELS='zai/glm-5.3' \
|
|
439
|
+
npm run test:evals:live
|
|
440
|
+
```
|
|
441
|
+
|
|
442
|
+
Multiple models are comma-, semicolon-, or newline-separated:
|
|
443
|
+
|
|
444
|
+
```bash
|
|
445
|
+
MODELS='zai/glm-5.3,'\
|
|
446
|
+
'openai-codex/gpt-5.6-luna,'\
|
|
447
|
+
'openai-codex/gpt-5.6-terra,'\
|
|
448
|
+
'openai-codex/gpt-5.6-sol'
|
|
449
|
+
PI_TOOLS_SUITE_EVAL_MODELS="$MODELS" \
|
|
450
|
+
npm run test:evals:live
|
|
451
|
+
```
|
|
452
|
+
|
|
453
|
+
Cases with model filters only run for matching models. For example, Sol-specific
|
|
454
|
+
orchestration cases do not run against GLM.
|
|
455
|
+
|
|
456
|
+
`PI_TOOLS_SUITE_EVAL_CASES` and `PI_TOOLS_SUITE_EVAL_CATEGORIES` (documented
|
|
457
|
+
below) filter the Bun live matrix the same way they filter the report runner.
|
|
458
|
+
|
|
459
|
+
### Comparative report runner
|
|
460
|
+
|
|
461
|
+
For model comparison, prefer the report runner:
|
|
462
|
+
|
|
463
|
+
```bash
|
|
464
|
+
MODELS='zai/glm-5.3,'\
|
|
465
|
+
'openai-codex/gpt-5.6-terra,'\
|
|
466
|
+
'openai-codex/gpt-5.6-sol'
|
|
467
|
+
PI_TOOLS_SUITE_EVAL_MODELS="$MODELS" \
|
|
468
|
+
npm run evals:report
|
|
469
|
+
```
|
|
470
|
+
|
|
471
|
+
The process exits non-zero if any selected case fails.
|
|
472
|
+
|
|
473
|
+
### Run only one category
|
|
474
|
+
|
|
475
|
+
```bash
|
|
476
|
+
PI_TOOLS_SUITE_EVAL_MODELS='zai/glm-5.3' \
|
|
477
|
+
PI_TOOLS_SUITE_EVAL_CATEGORIES='coding-quality,negative' \
|
|
478
|
+
npm run evals:report
|
|
479
|
+
```
|
|
480
|
+
|
|
481
|
+
Valid current categories are:
|
|
482
|
+
|
|
483
|
+
```text
|
|
484
|
+
tool-selection
|
|
485
|
+
coding-quality
|
|
486
|
+
orchestration
|
|
487
|
+
negative
|
|
488
|
+
```
|
|
489
|
+
|
|
490
|
+
### Run named cases
|
|
491
|
+
|
|
492
|
+
```bash
|
|
493
|
+
PI_TOOLS_SUITE_EVAL_MODELS='openai-codex/gpt-5.6-sol' \
|
|
494
|
+
CASES='orchestration.sol-delegates-substantial,'\
|
|
495
|
+
'orchestration.sol-keeps-tiny-edit'
|
|
496
|
+
PI_TOOLS_SUITE_EVAL_CASES="$CASES" \
|
|
497
|
+
npm run evals:report
|
|
498
|
+
```
|
|
499
|
+
|
|
500
|
+
### Other runner controls
|
|
501
|
+
|
|
502
|
+
| Variable | Purpose |
|
|
503
|
+
| --- | --- |
|
|
504
|
+
| `PI_TOOLS_SUITE_EVAL_TIMEOUT_MS` | per-case timeout; default 240 seconds |
|
|
505
|
+
| `PI_TOOLS_SUITE_EVAL_STREAM_IO=1` | stream child stdout/stderr while running |
|
|
506
|
+
| `PI_TOOLS_SUITE_EVAL_KEEP=1` | retain temporary projects for debugging |
|
|
507
|
+
| `PI_TOOLS_SUITE_EVAL_OUTPUT_DIR=/path` | choose report output directory |
|
|
508
|
+
|
|
509
|
+
## Existing prompt evals and benchmarks
|
|
510
|
+
|
|
511
|
+
The unified harness does not replace the older focused suites.
|
|
512
|
+
|
|
513
|
+
Use the existing prompt evals when changing the behavior they specifically
|
|
514
|
+
cover:
|
|
515
|
+
|
|
516
|
+
```bash
|
|
517
|
+
npm run test:prompt-evals
|
|
518
|
+
npm run test:prompt-evals:tool-selection
|
|
519
|
+
npm run test:prompt-evals:async
|
|
520
|
+
npm run test:prompt-evals:dcp
|
|
521
|
+
```
|
|
522
|
+
|
|
523
|
+
Use the locate benchmark for deeper repository-discovery efficiency comparisons:
|
|
524
|
+
|
|
525
|
+
```bash
|
|
526
|
+
npm run bench:locate
|
|
527
|
+
npm run bench:locate:analyze
|
|
528
|
+
```
|
|
529
|
+
|
|
530
|
+
The long-term direction is to share infrastructure where useful, but not to
|
|
531
|
+
discard specialized tests that measure a distinct contract well.
|
|
532
|
+
|
|
533
|
+
## Recommended model matrix
|
|
534
|
+
|
|
535
|
+
For changes to model discipline or orchestration, the useful comparison set is:
|
|
536
|
+
|
|
537
|
+
```text
|
|
538
|
+
zai/glm-5.3
|
|
539
|
+
openai-codex/gpt-5.6-luna
|
|
540
|
+
openai-codex/gpt-5.6-terra
|
|
541
|
+
openai-codex/gpt-5.6-sol
|
|
542
|
+
```
|
|
543
|
+
|
|
544
|
+
This matrix exposes several important regressions:
|
|
545
|
+
|
|
546
|
+
- a prompt improves GLM but makes Terra overthink simple tasks;
|
|
547
|
+
- Luna stops delegating substantial work;
|
|
548
|
+
- Terra escalates routine failures unnecessarily;
|
|
549
|
+
- Sol stops delegating expensive implementation work;
|
|
550
|
+
- Sol delegates tiny tasks and increases total cost;
|
|
551
|
+
- a cheaper route saves parent tokens but increases total worker tokens or
|
|
552
|
+
wall-clock time enough to erase the benefit.
|
|
553
|
+
|
|
554
|
+
Model availability and authentication are environment-dependent. Live evals are
|
|
555
|
+
therefore opt-in rather than part of the normal deterministic gate.
|
|
556
|
+
|
|
557
|
+
## Interpreting results
|
|
558
|
+
|
|
559
|
+
Do not rank models by one scalar score alone. Read the report in this order:
|
|
560
|
+
|
|
561
|
+
1. **Hard correctness:** did the required behavioral/post-run checks pass?
|
|
562
|
+
2. **Safety and discipline:** were forbidden tools absent and file/mutation
|
|
563
|
+
budgets respected?
|
|
564
|
+
3. **Workflow:** did the agent reproduce before editing and verify after editing
|
|
565
|
+
where the case requires it?
|
|
566
|
+
4. **Delegation quality:** did orchestration occur only at the intended
|
|
567
|
+
boundary?
|
|
568
|
+
5. **Cost:** how many parent and worker tokens were consumed?
|
|
569
|
+
6. **Latency:** did extra workers or retries materially increase wall-clock
|
|
570
|
+
time?
|
|
571
|
+
|
|
572
|
+
A useful before/after comparison looks like:
|
|
573
|
+
|
|
574
|
+
```text
|
|
575
|
+
success rate 85% -> 95%
|
|
576
|
+
parent tokens -38%
|
|
577
|
+
worker tokens +19%
|
|
578
|
+
total tokens -11%
|
|
579
|
+
tool calls -14%
|
|
580
|
+
wall time +6%
|
|
581
|
+
unwanted escalation 3 -> 0 cases
|
|
582
|
+
```
|
|
583
|
+
|
|
584
|
+
The framework currently reports the raw ingredients rather than inventing a
|
|
585
|
+
single weighted score. If an aggregate score is added later, hard correctness
|
|
586
|
+
and security failures should remain non-negotiable gates outside that score.
|
|
587
|
+
|
|
588
|
+
## Adding a new live case
|
|
589
|
+
|
|
590
|
+
1. Decide whether the behavior actually requires a live model. If a
|
|
591
|
+
deterministic test can prove it, prefer the deterministic test.
|
|
592
|
+
2. Add or reuse a small fixture. A bug-fix fixture should fail before the agent
|
|
593
|
+
acts.
|
|
594
|
+
3. Add a stable case to `test/evals/cases.ts`.
|
|
595
|
+
4. Express as much of the expected behavior as possible using shared assertions.
|
|
596
|
+
5. Use `postRunCommands` for executable correctness and hidden-style regression
|
|
597
|
+
checks.
|
|
598
|
+
6. Add a custom validator only when structured tool arguments need inspection.
|
|
599
|
+
7. Add the live case ID to the relevant extension/tool entries in
|
|
600
|
+
`coverage-manifest.ts`.
|
|
601
|
+
8. Run the deterministic harness tests before spending model tokens.
|
|
602
|
+
9. Run the smallest relevant model/category subset first.
|
|
603
|
+
10. Run the broader matrix before merging a prompt/routing change.
|
|
604
|
+
|
|
605
|
+
Good cases have one clear behavioral question. Avoid giant prompts that test
|
|
606
|
+
five unrelated policies at once because failures become hard to diagnose.
|
|
607
|
+
|
|
608
|
+
## Adding a new extension or tool
|
|
609
|
+
|
|
610
|
+
For a new extension:
|
|
611
|
+
|
|
612
|
+
1. add focused deterministic tests;
|
|
613
|
+
2. add the extension to `EXTENSION_EVAL_COVERAGE`;
|
|
614
|
+
3. add live cases only if model choice/workflow matters.
|
|
615
|
+
|
|
616
|
+
For a new model-facing tool:
|
|
617
|
+
|
|
618
|
+
1. add deterministic registration/execution tests;
|
|
619
|
+
2. add the tool to `TOOL_EVAL_COVERAGE`;
|
|
620
|
+
3. add positive and negative tool-selection cases when selection is non-obvious.
|
|
621
|
+
|
|
622
|
+
The coverage contract test intentionally fails until these steps are complete.
|
|
623
|
+
|
|
624
|
+
## CI strategy
|
|
625
|
+
|
|
626
|
+
A practical split is:
|
|
627
|
+
|
|
628
|
+
### Every change / pull request
|
|
629
|
+
|
|
630
|
+
```bash
|
|
631
|
+
npm run typecheck
|
|
632
|
+
npm test
|
|
633
|
+
npm run test:evals:contracts
|
|
634
|
+
```
|
|
635
|
+
|
|
636
|
+
The last command is redundant with the broad `npm test` traversal but useful as
|
|
637
|
+
a fast explicit gate in targeted jobs.
|
|
638
|
+
|
|
639
|
+
### Changes to tool descriptions, prompts, routing, or discipline
|
|
640
|
+
|
|
641
|
+
Run the relevant live category/model subset in addition to deterministic tests.
|
|
642
|
+
|
|
643
|
+
### Nightly or manual comparison
|
|
644
|
+
|
|
645
|
+
Run the full available GLM/Luna/Terra/Sol matrix and persist the JSON/Markdown
|
|
646
|
+
report as a CI artifact. This is the right place to watch token/cost/latency
|
|
647
|
+
drift because live-model variance and expense make it a poor default PR gate.
|
|
648
|
+
|
|
649
|
+
## Known limitations
|
|
650
|
+
|
|
651
|
+
- Live model behavior is probabilistic; one run is useful for regression
|
|
652
|
+
detection but not a statistically strong benchmark by itself.
|
|
653
|
+
- The current runner records one execution per case/model. Repetition and
|
|
654
|
+
confidence intervals are natural future additions.
|
|
655
|
+
- Verification detection is heuristic and based on observable shell commands.
|
|
656
|
+
- Provider-reported monetary cost may be absent even when tokens are present.
|
|
657
|
+
- Selection cases that block `subagents` measure the delegation decision, not
|
|
658
|
+
nested worker outcome quality.
|
|
659
|
+
- The current orchestration cases validate whether delegation/escalation was
|
|
660
|
+
selected. Deeper end-to-end worker-quality cases should remain separate so
|
|
661
|
+
routing failures and worker failures are diagnosable independently.
|
|
662
|
+
- No LLM judge currently grades prose quality. This is intentional: the initial
|
|
663
|
+
suite prioritizes objective behavior, tool discipline, executable outcomes,
|
|
664
|
+
and cost.
|
|
665
|
+
|
|
666
|
+
## Future extensions
|
|
667
|
+
|
|
668
|
+
Useful next additions include:
|
|
669
|
+
|
|
670
|
+
- repeated runs with pass-rate confidence intervals;
|
|
671
|
+
- baseline report comparison with explicit regression thresholds;
|
|
672
|
+
- richer parent-versus-worker token attribution by role/model;
|
|
673
|
+
- per-case normalized cost budgets;
|
|
674
|
+
- automatic detection of equivalent repeated patch attempts;
|
|
675
|
+
- DCP continuation evals that continue work after forced compaction;
|
|
676
|
+
- comment-checker precision/recall corpora;
|
|
677
|
+
- larger AST transformation fixtures with semantic hidden tests;
|
|
678
|
+
- end-to-end orchestration cases that allow real workers to complete and compare
|
|
679
|
+
`Sol direct` vs `Sol + Terra` vs `Terra + Sol escalation`;
|
|
680
|
+
- machine-readable CI summary output for trend dashboards.
|
|
681
|
+
|
|
682
|
+
The important constraint is to keep correctness evidence objective whenever it
|
|
683
|
+
can be objective, and use live models only for the decisions that genuinely
|
|
684
|
+
depend on model behavior.
|
|
@@ -29,6 +29,10 @@
|
|
|
29
29
|
"test:prompt-evals:async": "PROMPT_EVAL_E2E=1 bun test --concurrent --max-concurrency=5 test/async-subagents/selection-e2e.test.ts test/prompt-evals/async-routing-e2e.test.ts",
|
|
30
30
|
"test:prompt-evals:dcp": "PROMPT_EVAL_E2E=1 bun test --concurrent --max-concurrency=5 test/prompt-evals/dcp-summary-e2e.test.ts",
|
|
31
31
|
"test:prompt-evals": "PROMPT_EVAL_E2E=1 bun test --concurrent --max-concurrency=5 test/tool-selection-e2e.test.ts test/async-subagents/selection-e2e.test.ts test/prompt-evals",
|
|
32
|
+
"test:evals:contracts": "bun test test/evals/extension-contracts.test.ts test/evals/harness.test.ts",
|
|
33
|
+
"test:evals:live": "PI_TOOLS_SUITE_EVALS_LIVE=1 bun test --concurrent --max-concurrency=4 test/evals/live-evals.test.ts",
|
|
34
|
+
"evals": "npm run test:evals:contracts && bun test test/evals/live-evals.test.ts",
|
|
35
|
+
"evals:report": "bun test/evals/run-evals.ts",
|
|
32
36
|
"bench:locate": "PI_LOCATE_BENCH_ITERATIONS=5 PI_LOCATE_BENCH_FAKE_IDX=0 PI_LOCATE_BENCH_MODEL=zai/glm-5-turbo PI_LOCATE_BENCH_MODES=direct-read-grep,ast-structural,repo-search-hybrid,repo-discovery,subagent-search,unrestricted-suite node test/fixtures/hard-to-find-project/benchmark/run-locate-benchmark.mjs",
|
|
33
37
|
"bench:locate:analyze": "node test/fixtures/hard-to-find-project/benchmark/analyze-locate-benchmark.mjs",
|
|
34
38
|
"test:locate-benchmark-e2e": "PI_LOCATE_BENCH_E2E=1 PI_LOCATE_BENCH_MODEL=zai/glm-5-turbo bun test test/locate-benchmark-e2e.test.ts",
|
|
@@ -44,9 +48,9 @@
|
|
|
44
48
|
"vscode-languageserver-protocol": "^3.17.5"
|
|
45
49
|
},
|
|
46
50
|
"peerDependencies": {
|
|
47
|
-
"@earendil-works/pi-ai": "0.85.
|
|
48
|
-
"@earendil-works/pi-coding-agent": "0.85.
|
|
49
|
-
"@earendil-works/pi-tui": "0.85.
|
|
51
|
+
"@earendil-works/pi-ai": "0.85.1",
|
|
52
|
+
"@earendil-works/pi-coding-agent": "0.85.1",
|
|
53
|
+
"@earendil-works/pi-tui": "0.85.1",
|
|
50
54
|
"typebox": "*"
|
|
51
55
|
},
|
|
52
56
|
"devDependencies": {
|