@vibgrate/haile 2026.917.1 → 2026.921.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +216 -0
- package/dist/engine/haile.wasm +0 -0
- package/dist/index.d.ts +132 -1
- package/dist/index.js +11 -11
- package/package.json +17 -4
package/README.md
CHANGED
|
@@ -64,3 +64,219 @@ VG_HAILE_RELEASE=1 pnpm --filter @vibgrate/haile build # refuses without wasm
|
|
|
64
64
|
pnpm --filter @vibgrate/haile test # loader ↔ kernel contract (kernel cases need the wasm)
|
|
65
65
|
(cd crate && cargo test) # kernel unit tests, native
|
|
66
66
|
```
|
|
67
|
+
|
|
68
|
+
## Decision Engine (Stage 0 — not yet wired into any production code path)
|
|
69
|
+
|
|
70
|
+
> **Status: Stage 0.** This is package/ABI/provider scaffolding only. No
|
|
71
|
+
> production code path calls `createDecisionProvider()` yet — see
|
|
72
|
+
> [`docs/DECISION-ENGINE.md`](../../docs/DECISION-ENGINE.md) for the
|
|
73
|
+
> "as built" reference (architecture diagram, worked examples with rendered
|
|
74
|
+
> screenshots, real benchmark numbers) and
|
|
75
|
+
> [`docs/DECISION-ENGINE-WASM-SPEC.md`](../../docs/DECISION-ENGINE-WASM-SPEC.md)
|
|
76
|
+
> for the full original design (that spec originally proposed a standalone
|
|
77
|
+
> `packages/vibgrate-decision/` package; the maintainer decided afterward
|
|
78
|
+
> that this capability should instead live inside the existing
|
|
79
|
+
> `@vibgrate/haile` package, reusing its Rust crate / build / package
|
|
80
|
+
> infrastructure rather than standing up a parallel one — everything else in
|
|
81
|
+
> the spec, including the ABI shape, the six decision families, and every
|
|
82
|
+
> non-negotiable safety rule, applies unchanged). The seed-pack scoring
|
|
83
|
+
> weights in `crate/src/decision/pack_data.rs` are honest, hand-written
|
|
84
|
+
> placeholders — **not trained on a real corpus** — pending a later stacked
|
|
85
|
+
> PR's benchmark harness and a real, labelled `decision-pack.json`.
|
|
86
|
+
|
|
87
|
+
The Vibgrate Decision Engine is a **separate, additive, probabilistic**
|
|
88
|
+
capability living alongside Haile's deterministic architecture classification
|
|
89
|
+
above, in the same compiled WASM artifact. Given evidence Vibgrate already
|
|
90
|
+
knows (impact analysis, test results, verification history, Haile's own
|
|
91
|
+
abstention, …), it answers one of six small bounded questions and returns a
|
|
92
|
+
calibrated probability, never a fact. It never establishes facts, never
|
|
93
|
+
overrides deterministic evidence, and can never be mistaken for Haile's
|
|
94
|
+
`classify()` result — every `DecisionResult` carries its own
|
|
95
|
+
`"schema":"vg.decision.result.v1"` and `"source":"decision"`, fields Haile's
|
|
96
|
+
`HaileClassification` never has.
|
|
97
|
+
|
|
98
|
+
```ts
|
|
99
|
+
import { createDecisionProvider } from "@vibgrate/haile";
|
|
100
|
+
|
|
101
|
+
const decisions = createDecisionProvider();
|
|
102
|
+
// null when dist/engine/haile.wasm lacks the decision/ export family —
|
|
103
|
+
// never a stub provider. decide() itself never throws either: a
|
|
104
|
+
// missing/broken engine or a rejected request is always `null`.
|
|
105
|
+
decisions?.decide({
|
|
106
|
+
schema: "vg.decision.request.v1",
|
|
107
|
+
decision: "failure_related",
|
|
108
|
+
context: {},
|
|
109
|
+
evidence: [
|
|
110
|
+
{ id: "impact:AuthService", kind: "impact", feature: "stack_symbol_in_impact", value: true },
|
|
111
|
+
{ id: "test:AuthServiceTest", kind: "test", feature: "failing_test_covers_changed_symbol", value: true },
|
|
112
|
+
],
|
|
113
|
+
});
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
### The six decision families, one worked example each
|
|
117
|
+
|
|
118
|
+
**`task_complexity`** — recommend an initial Code Mode (advisory only; explicit `--mode`/`--model`/`--provider`, project config, and an active pin all outrank it, and `resolveMode()` still performs the hardware fit).
|
|
119
|
+
|
|
120
|
+
```json
|
|
121
|
+
// in: {"schema":"vg.decision.request.v1","decision":"task_complexity","context":{},
|
|
122
|
+
// "evidence":[{"id":"e1","kind":"change","feature":"migration_indicator","value":true},
|
|
123
|
+
// {"id":"e2","kind":"architecture","feature":"architecture_boundaries_crossed","value":true}]}
|
|
124
|
+
// out: {"schema":"vg.decision.result.v1","decision":"task_complexity","primitive":"choice",
|
|
125
|
+
// "selected":"forge","probabilities":[{"id":"spark","probability":0.0119},
|
|
126
|
+
// {"id":"flow","probability":0.1078},{"id":"forge","probability":0.8803}],
|
|
127
|
+
// "confidence":0.8803,"band":"high","abstain":false,
|
|
128
|
+
// "reasonCodes":["architecture_boundaries_crossed","migration_indicator"],
|
|
129
|
+
// "evidenceRefs":["e2","e1"],"calibrationId":"task-complexity-seed-1","source":"decision"}
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
**`failure_related`** — is a new test/compiler/runtime failure probably caused by the current edit? May route toward the bounded repair loop; may never suppress the failure.
|
|
133
|
+
|
|
134
|
+
```json
|
|
135
|
+
// in: {"schema":"vg.decision.request.v1","decision":"failure_related","context":{},
|
|
136
|
+
// "evidence":[{"id":"impact:AuthService","kind":"impact","feature":"stack_symbol_in_impact","value":true},
|
|
137
|
+
// {"id":"test:AuthServiceTest","kind":"test","feature":"failing_test_covers_changed_symbol","value":true}]}
|
|
138
|
+
// out: {"schema":"vg.decision.result.v1","decision":"failure_related","primitive":"binary",
|
|
139
|
+
// "selected":"related","probabilities":[{"id":"related","probability":0.9837},
|
|
140
|
+
// {"id":"unrelated","probability":0.0163}],"confidence":0.9837,"band":"high","abstain":false,
|
|
141
|
+
// "reasonCodes":["stack_symbol_in_impact","failing_test_covers_changed_symbol"],
|
|
142
|
+
// "evidenceRefs":["impact:AuthService","test:AuthServiceTest"],
|
|
143
|
+
// "calibrationId":"failure-related-seed-1","source":"decision"}
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
**`repair_action`** — after a failed verification attempt, which bounded VG Code mechanism runs next (respecting `maxRepairRounds`)? It never writes code itself.
|
|
147
|
+
|
|
148
|
+
```json
|
|
149
|
+
// in: {"schema":"vg.decision.request.v1","decision":"repair_action","context":{},
|
|
150
|
+
// "evidence":[{"id":"e1","kind":"verification","feature":"capsule_stale","value":true}]}
|
|
151
|
+
// out: {"schema":"vg.decision.result.v1","decision":"repair_action","primitive":"choice",
|
|
152
|
+
// "selected":"refresh_capsule","probabilities":[{"id":"retry_local_repair","probability":0.2123},
|
|
153
|
+
// {"id":"refresh_capsule","probability":0.577},{"id":"expand_context","probability":0.0707},
|
|
154
|
+
// {"id":"run_additional_verification","probability":0.0639},
|
|
155
|
+
// {"id":"escalate_reasoning","probability":0.0474},{"id":"ask_human","probability":0.0287}],
|
|
156
|
+
// "confidence":0.577,"band":"low","abstain":false,"reasonCodes":["capsule_stale"],
|
|
157
|
+
// "evidenceRefs":["e1"],"calibrationId":"repair-action-seed-1","source":"decision"}
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
**`completion_confidence`** — does the implementation appear to satisfy the task? Asymmetric by design: may say "not done yet"; may never bypass a failed test or a protected finding.
|
|
161
|
+
|
|
162
|
+
```json
|
|
163
|
+
// in: {"schema":"vg.decision.request.v1","decision":"completion_confidence","context":{},
|
|
164
|
+
// "evidence":[{"id":"e1","kind":"test","feature":"tests_passed_ratio","value":1},
|
|
165
|
+
// {"id":"e2","kind":"diagnostic","feature":"protected_findings_present","value":true}]}
|
|
166
|
+
// out: {"schema":"vg.decision.result.v1","decision":"completion_confidence","primitive":"binary",
|
|
167
|
+
// "selected":"incomplete","probabilities":[{"id":"complete","probability":0.0037},
|
|
168
|
+
// {"id":"incomplete","probability":0.9963}],"confidence":0.9963,"band":"high","abstain":false,
|
|
169
|
+
// "reasonCodes":["tests_passed_ratio","protected_findings_present"],
|
|
170
|
+
// "evidenceRefs":["e1","e2"],"calibrationId":"completion-confidence-seed-1","source":"decision"}
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
**`finding_attention`** — queue-prioritisation only. Never modifies severity, CVSS, EPSS, KEV status, reachability, RiskScore, DriftScore, `protected_finding`, or the review decision.
|
|
174
|
+
|
|
175
|
+
```json
|
|
176
|
+
// in: {"schema":"vg.decision.request.v1","decision":"finding_attention","context":{},
|
|
177
|
+
// "evidence":[{"id":"e1","kind":"diagnostic","feature":"severity_score","value":1},
|
|
178
|
+
// {"id":"e2","kind":"diagnostic","feature":"exploit_maturity_score","value":1}]}
|
|
179
|
+
// out: {"schema":"vg.decision.result.v1","decision":"finding_attention","primitive":"choice",
|
|
180
|
+
// "selected":"immediate","probabilities":[{"id":"immediate","probability":0.7516},
|
|
181
|
+
// {"id":"review","probability":0.1518},{"id":"normal","probability":0.092},
|
|
182
|
+
// {"id":"defer","probability":0.0046}],"confidence":0.7516,"band":"medium","abstain":false,
|
|
183
|
+
// "reasonCodes":["severity_score","exploit_maturity_score"],"evidenceRefs":["e1","e2"],
|
|
184
|
+
// "calibrationId":"finding-attention-seed-1","source":"decision"}
|
|
185
|
+
```
|
|
186
|
+
|
|
187
|
+
**`architecture_fallback`** — only when Haile's own classifier has explicitly abstained. Uses its own small role vocabulary (never Haile's `taxonomy::ROLES`), reads Haile's abstention only as one input feature, and — like every other decision above — always carries `"source":"decision"` so it can never be mistaken for Haile's deterministic classification.
|
|
188
|
+
|
|
189
|
+
```json
|
|
190
|
+
// in: {"schema":"vg.decision.request.v1","decision":"architecture_fallback","context":{},
|
|
191
|
+
// "evidence":[{"id":"e1","kind":"architecture","feature":"role_hint_authenticate","value":1},
|
|
192
|
+
// {"id":"e2","kind":"architecture","feature":"haile_abstained","value":true}]}
|
|
193
|
+
// out: {"schema":"vg.decision.result.v1","decision":"architecture_fallback","primitive":"choice",
|
|
194
|
+
// "selected":"authenticate","probabilities":[{"id":"authenticate","probability":0.8007},
|
|
195
|
+
// {"id":"persist","probability":0.0399},{"id":"respond","probability":0.0399},
|
|
196
|
+
// {"id":"validate","probability":0.0399},{"id":"orchestrate","probability":0.0399},
|
|
197
|
+
// {"id":"render","probability":0.0399}],"confidence":0.8007,"band":"high","abstain":false,
|
|
198
|
+
// "reasonCodes":["haile_abstained_fallback_engaged","role_hint_authenticate"],
|
|
199
|
+
// "evidenceRefs":["e2","e1"],"calibrationId":"architecture-fallback-seed-1","source":"decision"}
|
|
200
|
+
```
|
|
201
|
+
|
|
202
|
+
### Determinism, calibration, and abstention
|
|
203
|
+
|
|
204
|
+
Same input JSON always yields byte-identical output for a given engine/pack
|
|
205
|
+
version (no randomness, clock, or unordered iteration). Confidence bands
|
|
206
|
+
(`high`/`medium`/`low`/`abstain`) are per-decision, versioned thresholds
|
|
207
|
+
(`crate/src/decision/calibration.rs`), not one global cutoff. An honest
|
|
208
|
+
`abstain` (e.g. no evidence supplied) is a normal, fully-shaped result, not
|
|
209
|
+
an error — the host falls back to existing behaviour exactly as it does when
|
|
210
|
+
the engine itself is absent.
|
|
211
|
+
|
|
212
|
+
### Runnable examples (`examples/decision/` — one worked example each)
|
|
213
|
+
|
|
214
|
+
```bash
|
|
215
|
+
pnpm --filter @vibgrate/haile run examples:decision # all six, in sequence, real WASM output
|
|
216
|
+
pnpm --filter @vibgrate/haile exec tsx examples/decision/failure-related.example.ts # one at a time
|
|
217
|
+
```
|
|
218
|
+
|
|
219
|
+
Six small standalone scripts, one per decision family, each building a
|
|
220
|
+
realistic `DecisionRequest` with an inline comment on every evidence field,
|
|
221
|
+
calling the real `createDecisionProvider()`, and pretty-printing the
|
|
222
|
+
`DecisionResult`. `docs/generate-screenshots.mjs`
|
|
223
|
+
(`pnpm --filter @vibgrate/haile run docs:screenshots`) runs these same six
|
|
224
|
+
scripts to regenerate [`docs/DECISION-ENGINE.md`](../../docs/DECISION-ENGINE.md)'s
|
|
225
|
+
worked-example JSON and screenshots — see that doc for the full architecture
|
|
226
|
+
reference plus rendered examples.
|
|
227
|
+
|
|
228
|
+
### Tests
|
|
229
|
+
|
|
230
|
+
```bash
|
|
231
|
+
pnpm --filter @vibgrate/haile test # includes tests/decision-*.test.ts and tests/bench-harness.test.ts
|
|
232
|
+
(cd crate && cargo test) # includes crate/src/decision/**/*'s own unit tests
|
|
233
|
+
```
|
|
234
|
+
|
|
235
|
+
### Benchmarking (`bench/` — spec §17/§18/§32)
|
|
236
|
+
|
|
237
|
+
```bash
|
|
238
|
+
pnpm --filter @vibgrate/haile run bench:decision # all six decisions
|
|
239
|
+
pnpm --filter @vibgrate/haile run bench:decision:failure-related # one decision
|
|
240
|
+
pnpm --filter @vibgrate/haile run bench:decision:generate-corpus # regenerate bench/corpus/*.jsonl
|
|
241
|
+
```
|
|
242
|
+
|
|
243
|
+
`bench/evaluate.ts` loads each labelled corpus under `bench/corpus/*.jsonl`,
|
|
244
|
+
splits it by `repo_id` (spec §32 — never by random row), runs the real
|
|
245
|
+
`decision-wasm` backend (`bench/backends/decision-wasm.ts`, wrapping
|
|
246
|
+
`createDecisionProvider()` above) over the held-out test split, and prints
|
|
247
|
+
accuracy, precision, recall, F1, Brier score, negative log likelihood,
|
|
248
|
+
expected calibration error, coverage, accuracy-by-confidence-band, and
|
|
249
|
+
abstention rate — plus a naive-heuristic baseline for comparison
|
|
250
|
+
(`bench/backends/full-model.ts`; Jev is stubbed unavailable per spec §24,
|
|
251
|
+
`bench/backends/jev.ts`).
|
|
252
|
+
|
|
253
|
+
> **Every corpus row is SYNTHETIC** (`bench/corpus/README.md`) and the
|
|
254
|
+
> seed-pack weights (`crate/src/decision/pack_data.rs`) are untrained
|
|
255
|
+
> placeholders — so the numbers below are genuinely mediocre. That is the
|
|
256
|
+
> correct, honest Stage 0 result, not a bug in the harness: this corpus is
|
|
257
|
+
> for validating the benchmark machinery itself, not for judging production
|
|
258
|
+
> readiness. It is **not** the real 500–1,000-example corpus spec §31/§32
|
|
259
|
+
> requires before any production hook.
|
|
260
|
+
|
|
261
|
+
Real output from an actual run (`pnpm --filter @vibgrate/haile run bench:decision`), unedited:
|
|
262
|
+
|
|
263
|
+
```
|
|
264
|
+
=== Summary — vibgrate-decision-wasm across all six decisions ===
|
|
265
|
+
decision | n(test) | accuracy | F1(macro) | Brier | NLL | ECE | coverage | abstain%
|
|
266
|
+
task_complexity | 15 | 46.7% | 24.6% | 0.5782 | 0.8517 | 0.3216 | 73.3% | 26.7%
|
|
267
|
+
failure_related | 20 | 35.0% | 25.9% | 1.2023 | 2.6754 | 0.6201 | 100.0% | 0.0%
|
|
268
|
+
repair_action | 18 | 100.0% | 100.0% | 0.2879 | 0.6553 | 0.4757 | 66.7% | 33.3%
|
|
269
|
+
completion_confidence | 20 | 55.0% | 52.0% | 0.7553 | 1.3406 | 0.3805 | 85.0% | 15.0%
|
|
270
|
+
finding_attention | 16 | 37.5% | 34.4% | 0.5932 | 1.0679 | 0.1527 | 31.3% | 68.8%
|
|
271
|
+
architecture_fallback | 21 | 100.0% | 100.0% | 0.2486 | 0.5423 | 0.3416 | 76.2% | 23.8%
|
|
272
|
+
```
|
|
273
|
+
|
|
274
|
+
Per-decision output also prints a naive-heuristic-baseline row for
|
|
275
|
+
comparison and accuracy broken out by confidence band; run the command
|
|
276
|
+
above for the full table. `repair_action` and `architecture_fallback`
|
|
277
|
+
scoring 100% here is a property of their small synthetic test splits (18
|
|
278
|
+
and 21 rows respectively, after the repo-level split), not a general claim —
|
|
279
|
+
`task_complexity` and `failure_related`, evaluated against the same kind of
|
|
280
|
+
corpus, land far lower. Treat every number as "the harness works and these
|
|
281
|
+
are the real, unflattering seed-pack results," not as a production
|
|
282
|
+
readiness signal.
|
package/dist/engine/haile.wasm
CHANGED
|
Binary file
|
package/dist/index.d.ts
CHANGED
|
@@ -150,6 +150,18 @@ interface WasmHaileKernel {
|
|
|
150
150
|
* built the document; it crosses as-is. Empty string = the kernel abstained.
|
|
151
151
|
*/
|
|
152
152
|
evalFactsJson?(document: unknown): string;
|
|
153
|
+
/**
|
|
154
|
+
* Decision Engine bindings (Stage 0, docs/DECISION-ENGINE-WASM-SPEC.md).
|
|
155
|
+
* Structurally separate from every method above — see
|
|
156
|
+
* `../src/decision.ts`'s `createDecisionProvider()`, the only caller of
|
|
157
|
+
* this. Present only on a kernel compiled with the `decision/` module.
|
|
158
|
+
*/
|
|
159
|
+
decision?: {
|
|
160
|
+
version(): number;
|
|
161
|
+
decideJson(requestJson: string): string;
|
|
162
|
+
decideBatchJson(requestsJson: string): string;
|
|
163
|
+
catalogJson(): string;
|
|
164
|
+
};
|
|
153
165
|
}
|
|
154
166
|
/** Load the wasm engine if the artifact exists; null otherwise. Memoized. */
|
|
155
167
|
declare function loadWasmHaileKernel(): WasmHaileKernel | null;
|
|
@@ -228,4 +240,123 @@ interface SidecarDeps {
|
|
|
228
240
|
}
|
|
229
241
|
declare function runSidecarCli(argv?: string[], deps?: SidecarDeps): Promise<number>;
|
|
230
242
|
|
|
231
|
-
|
|
243
|
+
/**
|
|
244
|
+
* Vibgrate Decision Engine — Stage 0 typed entry point.
|
|
245
|
+
* docs/DECISION-ENGINE-WASM-SPEC.md
|
|
246
|
+
*
|
|
247
|
+
* A separate, additive, PROBABILISTIC capability living alongside Haile's
|
|
248
|
+
* deterministic architecture classification in this same package. It
|
|
249
|
+
* answers one of six small bounded questions (`DECISIONS` below) from
|
|
250
|
+
* evidence the host already has; it never establishes facts, never
|
|
251
|
+
* discovers CVEs, never calculates DriftScore/RiskScore, never writes code,
|
|
252
|
+
* and never overrides deterministic Vibgrate evidence (spec §3, §20).
|
|
253
|
+
*
|
|
254
|
+
* # Provenance — the hard safety boundary (non-negotiable)
|
|
255
|
+
*
|
|
256
|
+
* `DecisionResult` is a structurally different TypeScript type from
|
|
257
|
+
* `HaileClassification` (`./types.ts`): different required fields
|
|
258
|
+
* (`schema`, `decision`, `primitive`, `source`), no shared "this is
|
|
259
|
+
* authoritative" field, and every result this module can produce carries
|
|
260
|
+
* `source: "decision"` — spec §10.6's own example, applied here to *all
|
|
261
|
+
* six* decision types, not only `architecture_fallback`, as a uniform
|
|
262
|
+
* structural marker. Host code must branch on `source`/`schema` before
|
|
263
|
+
* treating a result as anything but a probabilistic judgement; nothing in
|
|
264
|
+
* this file, `wasm-kernel.ts`, or the Rust crate (`crate/src/decision/`)
|
|
265
|
+
* ever converts a `DecisionResult` into a `HaileClassification` or vice
|
|
266
|
+
* versa. See `tests/decision-safety-invariant.test.ts`.
|
|
267
|
+
*
|
|
268
|
+
* `createDecisionProvider()` mirrors `createHaileProvider()` (`./provider.ts`)
|
|
269
|
+
* in posture only: it returns `null` when the wasm engine is missing or the
|
|
270
|
+
* Decision Engine's export family (`vg_decide*`) is absent, and
|
|
271
|
+
* `decide()` returns `null` — never throws — whenever the engine or a
|
|
272
|
+
* single request is unavailable/broken/rejected. A `null` result always
|
|
273
|
+
* means "proceed using existing behaviour" (spec §14, §19).
|
|
274
|
+
*
|
|
275
|
+
* Stage 0 status: package, ABI, and this provider only. No production code
|
|
276
|
+
* path calls `createDecisionProvider()` yet — see the README's Stage 0
|
|
277
|
+
* section and spec §31.
|
|
278
|
+
*/
|
|
279
|
+
|
|
280
|
+
/** The six named decision families (spec §6, §10) — no more, no fewer, for Stage 0. */
|
|
281
|
+
type DecisionKind = "task_complexity" | "failure_related" | "repair_action" | "completion_confidence" | "finding_attention" | "architecture_fallback";
|
|
282
|
+
declare const DECISIONS: readonly DecisionKind[];
|
|
283
|
+
/** Free-form host context (spec §6) — never scored directly by the kernel; carried for host-side policy/telemetry only. */
|
|
284
|
+
type DecisionContext = Record<string, unknown>;
|
|
285
|
+
/** Evidence rows the host supplies from facts Vibgrate already established (spec §9). */
|
|
286
|
+
interface DecisionEvidence {
|
|
287
|
+
id: string;
|
|
288
|
+
kind: "graph" | "impact" | "test" | "diagnostic" | "architecture" | "change" | "verification" | "task" | "dependency" | "runtime";
|
|
289
|
+
feature: string;
|
|
290
|
+
value: string | number | boolean;
|
|
291
|
+
confidence?: number;
|
|
292
|
+
provenance?: "observed" | "derived" | "declared" | "classifier";
|
|
293
|
+
}
|
|
294
|
+
/** Free-form request constraints (spec §6) — reserved for later stages; Stage 0 decisions ignore this. */
|
|
295
|
+
type DecisionConstraints = Record<string, unknown>;
|
|
296
|
+
interface DecisionRequest {
|
|
297
|
+
schema: "vg.decision.request.v1";
|
|
298
|
+
decision: DecisionKind;
|
|
299
|
+
context: DecisionContext;
|
|
300
|
+
evidence: DecisionEvidence[];
|
|
301
|
+
constraints?: DecisionConstraints;
|
|
302
|
+
}
|
|
303
|
+
interface DecisionOption {
|
|
304
|
+
id: string;
|
|
305
|
+
probability: number;
|
|
306
|
+
}
|
|
307
|
+
/** Confidence band (spec §18) — per-decision versioned thresholds, never one global cutoff. */
|
|
308
|
+
type DecisionBand = "high" | "medium" | "low" | "abstain";
|
|
309
|
+
/**
|
|
310
|
+
* The engine's result contract (spec §8), plus one addition: `source`. See
|
|
311
|
+
* this file's top-of-file doc comment for why `source` is mandatory and
|
|
312
|
+
* uniform across all six decisions rather than only `architecture_fallback`.
|
|
313
|
+
*/
|
|
314
|
+
interface DecisionResult {
|
|
315
|
+
schema: "vg.decision.result.v1";
|
|
316
|
+
engineVersion: string;
|
|
317
|
+
packVersion: string;
|
|
318
|
+
decision: DecisionKind;
|
|
319
|
+
primitive: "binary" | "choice" | "score";
|
|
320
|
+
selected?: string;
|
|
321
|
+
probabilities?: DecisionOption[];
|
|
322
|
+
score?: number;
|
|
323
|
+
confidence: number;
|
|
324
|
+
band: DecisionBand;
|
|
325
|
+
abstain: boolean;
|
|
326
|
+
reasonCodes: string[];
|
|
327
|
+
evidenceRefs: string[];
|
|
328
|
+
calibrationId: string;
|
|
329
|
+
/**
|
|
330
|
+
* Structural provenance marker (non-negotiable). Always `"decision"`:
|
|
331
|
+
* this result is ALWAYS a probabilistic Decision Engine judgement, NEVER
|
|
332
|
+
* deterministic Haile architecture-classification evidence (spec §2.5,
|
|
333
|
+
* §10.6). Check this (or `schema === "vg.decision.result.v1"`) before
|
|
334
|
+
* treating a result as authoritative for anything.
|
|
335
|
+
*/
|
|
336
|
+
source: "decision";
|
|
337
|
+
}
|
|
338
|
+
interface DecisionCatalog {
|
|
339
|
+
schema: "vg.decision.catalog.v1";
|
|
340
|
+
decisions: DecisionKind[];
|
|
341
|
+
}
|
|
342
|
+
/** Host-facing API (spec §14). Every method is optional-safe: a missing/broken engine yields `null` / `[]`, never a throw. */
|
|
343
|
+
interface DecisionProvider {
|
|
344
|
+
version(): string;
|
|
345
|
+
decide(request: DecisionRequest): DecisionResult | null;
|
|
346
|
+
decideBatch(requests: DecisionRequest[]): DecisionResult[];
|
|
347
|
+
catalog(): DecisionCatalog;
|
|
348
|
+
}
|
|
349
|
+
/** Wire string version (distinct from Haile's `HAILE_ENGINE_VERSION` in `./types.ts`). */
|
|
350
|
+
declare const DECISION_ENGINE_VERSION = "vg-decision-wasm@0.1.0";
|
|
351
|
+
/** Integer form the kernel's `vg_decision_version` export returns (crate/src/decision/mod.rs `ENGINE_VERSION_CODE`). A mismatch means "engine unavailable", never a version to silently coerce. */
|
|
352
|
+
declare const DECISION_ENGINE_VERSION_CODE = 1;
|
|
353
|
+
/**
|
|
354
|
+
* Build the Decision Engine provider. Returns `null` — never a stub/no-op
|
|
355
|
+
* provider — when the wasm engine is missing, when it was compiled without
|
|
356
|
+
* the `decision/` module (`kernel.decision` absent, e.g. an older Haile
|
|
357
|
+
* kernel), or when its integer version doesn't match
|
|
358
|
+
* `DECISION_ENGINE_VERSION_CODE` (a broken/mismatched build).
|
|
359
|
+
*/
|
|
360
|
+
declare function createDecisionProvider(kernel?: WasmHaileKernel | null): DecisionProvider | null;
|
|
361
|
+
|
|
362
|
+
export { DECISIONS, DECISION_ENGINE_VERSION, DECISION_ENGINE_VERSION_CODE, DEFAULT_POLICY, DEFAULT_PROFILE, DEFAULT_SYMBOL_CAP, type DecisionBand, type DecisionCatalog, type DecisionConstraints, type DecisionContext, type DecisionEvidence, type DecisionKind, type DecisionOption, type DecisionProvider, type DecisionRequest, type DecisionResult, HAILE_ENGINE_VERSION, HAILE_IR, HAILE_MAGIC, HAILE_TAXONOMY, type HaileClassification, type HaileClassifyInput, type HailePolicy, type HaileProfile, type HaileProvider, POLICIES, type SidecarDeps, adaptGraph, createDecisionProvider, createHaileProvider, isCallableKind, loadWasmHaileKernel, resetWasmHaileKernelCache, runSidecarCli, symbolKindOf };
|
package/dist/index.js
CHANGED
|
@@ -1,11 +1,11 @@
|
|
|
1
|
-
import{pathToFileURL as
|
|
2
|
-
`),1;if(!
|
|
3
|
-
`),1;let
|
|
4
|
-
`),1;let
|
|
5
|
-
`),1;let
|
|
6
|
-
`),1;let s=
|
|
7
|
-
`),1;let
|
|
8
|
-
`),1;
|
|
9
|
-
`),1}let
|
|
10
|
-
`),
|
|
11
|
-
`),1}return 0}function
|
|
1
|
+
import{pathToFileURL as ge}from"node:url";import*as v from"node:fs";import*as A from"node:path";var w=["hexagonal-v1","layered-v1","vertical-v1"],H="hexagonal-v1",I="vg.arch.v1",P="vg.arch.taxonomy.v1",E="vg.arch.ir.v1",h="haile-fast/2026.903.5",C="balanced",D=8e3,J=new Set(["function","method","route","test","component","job","class","interface"]);var M=240,V=400,$=16,Y=32,X=new Set(["public","private","protected","internal","static","async","override","virtual","abstract","final","def","fn","function","func","sub","pub","export","const","let","var","new","sealed","extern","unsafe","readonly","partial","default","synchronized","native","inline","constexpr","friend","explicit","operator","suspend","open","lateinit","get","set"]);function K(n){return J.has(n)}function F(n){switch(n){case"method":return"method";case"route":return"route";case"test":return"test";case"job":return"job";case"component":return"handler";case"class":return"type";case"interface":return"interface";default:return"function"}}function z(n,e=""){let o=e?n.indexOf(`${e}(`):-1;if(o<0&&(n.startsWith("@")||n.startsWith("[")))return[];let i=o>=0?o+e.length:n.indexOf("("),c=n.lastIndexOf(")");if(i<0||c<=i)return[];let r=n.slice(i+1,c).trim();return!r||r==="void"?[]:r.split(",").map(s=>s.trim()).filter(Boolean).slice(0,16).map(s=>{let d=s.replace(/\b(const|mut|ref|public|private|protected|readonly|final)\b/g,"").trim();if(d.includes(":")){let[a,u]=d.split(":").map(f=>f.trim());return{name:a||"arg",type_name:u||void 0}}let t=d.split(/\s+/);return t.length>=2?{name:t[t.length-1],type_name:t[0]}:{name:d}})}function Q(n,e){let o=n.lastIndexOf("->");if(o>=0)return n.slice(o+2).replace(/[{;].*$/,"").trim()||void 0;let i=n.lastIndexOf(":"),c=n.lastIndexOf(")");if(i>c&&c>=0)return n.slice(i+1).replace(/[{;].*$/,"").replace(/\s*=>.*$/,"").trim()||void 0;let r=e?n.indexOf(`${e}(`):-1;if(r>0){let s=n.slice(0,r).trim(),d=0,t=0;for(let u=s.length-1;u>=0;u--){let f=s[u];if(f===">"||f==="]"||f===")")d++;else if(f==="<"||f==="["||f==="(")d--;else if(/\s/.test(f)&&d===0){t=u+1;break}}let a=s.slice(t).trim();if(a&&!a.startsWith("@")&&!a.startsWith("[")&&!X.has(a)&&a!=="void"&&!a.endsWith(")")&&!a.endsWith("]"))return a}}function Z(n,e,o,i){let c=[],r=[],s=new Set,d=new Set;for(let t of e){if(t.src!==n||t.kind!=="call"&&t.kind!=="references")continue;let a=o.get(t.dst);if(!a)continue;let u=a.qualifiedName||a.name;if(!(!u||s.has(u))&&(s.add(u),c.push(u),a.file&&a.file!==i&&!d.has(a.file)&&r.length<$&&(d.add(a.file),r.push(a.file)),c.length>=24))break}return{calls:c,calleePaths:r}}function ee(n,e,o){let i=new Map;for(let r of n)(r.kind==="file"||r.kind==="module")&&i.set(r.id,r.file);let c=new Map;for(let r of e){if(r.kind!=="import")continue;let s=i.get(r.src);if(!s)continue;let d=o.get(r.dst),t=d?.qualifiedName||d?.name;if(!t)continue;let a=c.get(s);a||(a=[],c.set(s,a)),a.length<Y&&!a.includes(t)&&a.push(t)}return c}function N(n){let e=n.nodes??[],o=n.edges??[],i=new Map;for(let s of e)i.set(s.id,s);let c=ee(e,o,i),r=[];for(let s of e){if(!K(s.kind))continue;let d=typeof s.decorators=="string"?s.decorators.trim():"",t=`${d?`${d} `:""}${s.signature??""}`.slice(0,V),{calls:a,calleePaths:u}=Z(s.id,o,i,s.file),f=typeof s.doc=="string"?s.doc.trim().slice(0,M):"";r.push({node_id:s.id,file_path:s.file,name:s.name,qualified_name:s.qualifiedName||s.name,symbol_kind:F(s.kind),language:s.lang??"",signature:t,calls:a,parameters:z(t,s.name),return_type:Q(t,s.name),...f?{doc:f}:{},callee_paths:u,imports:c.get(s.file)??[],...s.effects&&typeof s.effects=="object"?{effects:s.effects}:{},...Array.isArray(s.duties)?{duties:s.duties.slice(0,48)}:{}})}return r.sort((s,d)=>s.node_id<d.node_id?-1:s.node_id>d.node_id?1:0),r}import*as q from"node:fs";import*as m from"node:path";import{fileURLToPath as se}from"node:url";import*as x from"node:fs";import{fileURLToPath as ne}from"node:url";var g;function ie(){return[new URL("./engine/haile.wasm",import.meta.url),new URL("../engine/haile.wasm",import.meta.url),new URL("../crate/target/wasm32-unknown-unknown/release/vibgrate_haile_kernel.wasm",import.meta.url)].map(n=>ne(n))}function re(n){try{let e=new WebAssembly.Module(n),i=new WebAssembly.Instance(e,{}).exports;if(typeof i.vg_alloc!="function"||typeof i.vg_free!="function"||typeof i.vg_version!="function"||typeof i.vg_classify!="function"||!i.memory)return null;let c=t=>{let a=new TextEncoder().encode(t),u=i.vg_alloc(a.length);return new Uint8Array(i.memory.buffer,u,a.length).set(a),{ptr:u,len:a.length}},r=t=>{if(!t)return"";let a=new DataView(i.memory.buffer,t,4).getUint32(0,!0),u=new TextDecoder().decode(new Uint8Array(i.memory.buffer,t+4,a));return i.vg_free(t,4+a),u},s=(t,a)=>{let{ptr:u,len:f}=c(a);try{return r(t(u,f))}finally{i.vg_free(u,f)}},d={version:()=>{try{return r(i.vg_version())||h}catch{return h}},classifyJson:t=>{let a=JSON.stringify({name:t.name??"",path:t.path??"",symbolKind:t.symbolKind??"function",fileKind:t.fileKind??"source",calls:t.calls??[],types:t.types??[],idents:t.idents??[],profile:t.profile??"balanced",policy:t.policy??"hexagonal-v1",qualifiedName:t.qualifiedName??"",doc:t.doc??"",signature:t.signature??"",params:t.params??[],returnType:t.returnType??"",calleePaths:t.calleePaths??[],imports:t.imports??[],astRole:t.astRole??"",effects:t.effects??{},duties:t.duties??[]});return s(i.vg_classify,a)}};if(typeof i.vg_overview=="function"&&(d.overviewJson=(t,a)=>s(i.vg_overview,JSON.stringify({graph:t,sidecar:a??null}))),typeof i.vg_slice=="function"&&(d.sliceJson=(t,a,u)=>s(i.vg_slice,JSON.stringify({graph:t,sidecar:a??null,spec:u??{}}))),typeof i.vg_arch_page=="function"&&(d.archPage=t=>s(i.vg_arch_page,JSON.stringify({host:t.host??"browser",nonce:t.nonce??"",theme:t.theme??""}))),typeof i.vg_eval_facts=="function"&&(d.evalFactsJson=t=>s(i.vg_eval_facts,JSON.stringify(t))),typeof i.vg_decision_version=="function"&&typeof i.vg_decide=="function"&&typeof i.vg_decide_batch=="function"&&typeof i.vg_decision_catalog=="function"){let t=i.vg_decide,a=i.vg_decide_batch,u=i.vg_decision_catalog,f=i.vg_decision_version;d.decision={version:()=>f(),decideJson:y=>s(t,y),decideBatchJson:y=>s(a,y),catalogJson:()=>r(u())}}return d}catch{return null}}function _(){if(g!==void 0)return g;g=null;try{let n=ie().find(e=>x.existsSync(e));return n&&(g=re(x.readFileSync(n))),g}catch{return g=null,g}}function te(){g=void 0}function T(n){return Array.isArray(n)?n.filter(e=>typeof e=="string"):[]}function O(n){if(!n)return null;try{return JSON.parse(n)}catch{return null}}function oe(n){if(!n)return null;try{let e=JSON.parse(n);if(!e||typeof e!="object")return null;let o=e.role&&typeof e.role=="object"?e.role:e,i=typeof o.primary=="string"?o.primary:"";if(!i)return null;let c=e.intent&&typeof e.intent=="object"?e.intent:{};return{primary:i,confidence:typeof o.confidence=="number"?o.confidence:0,band:typeof o.band=="string"?o.band:"abstain",alternatives:Array.isArray(o.alternatives)?o.alternatives.filter(r=>typeof r?.role=="string").map(r=>({role:r.role,confidence:typeof r.confidence=="number"?r.confidence:0})):[],purposes:Array.isArray(e.purposes)?e.purposes.filter(r=>typeof r?.purpose=="string").map(r=>({purpose:r.purpose,confidence:typeof r.confidence=="number"?r.confidence:0})):[],intent:{text:typeof c.text=="string"?c.text:"",verbs:T(c.verbs),objects:T(c.objects)},findings:Array.isArray(e.findings)?e.findings.filter(r=>typeof r?.rule=="string"&&typeof r?.message=="string").map(r=>({rule:r.rule,severity:typeof r.severity=="string"?r.severity:"warn",message:r.message,...typeof r.line=="number"&&Number.isFinite(r.line)&&r.line>0?{line:Math.floor(r.line)}:{}})).slice(0,8):[],evidence:Array.isArray(e.evidence)?e.evidence.filter(r=>typeof r?.kind=="string"&&typeof r?.signal=="string").map(r=>({kind:r.kind,signal:r.signal,weight:typeof r.weight=="number"?r.weight:0})):[]}}catch{return null}}function S(n){let e=n===void 0?_():n;if(!e)return null;let o={version:()=>e.version(),classify:i=>{try{return oe(e.classifyJson(i))}catch{return null}}};return e.overviewJson&&(o.projectOverview=(i,c)=>{try{return O(e.overviewJson(i,c))}catch{return null}}),e.sliceJson&&(o.projectSlice=(i,c,r)=>{try{return O(e.sliceJson(i,c,r))}catch{return null}}),e.archPage&&(o.renderArchPage=i=>{try{let c=e.archPage({host:i.host,theme:i.theme,nonce:i.nonce});return typeof c=="string"?c:""}catch{return""}}),e.evalFactsJson&&(o.evalFacts=i=>{try{return O(e.evalFactsJson(i))}catch{return null}}),o.archUiAssets=()=>{try{let i=m.dirname(se(import.meta.url)),c=[m.join(i,"arch-ui"),m.join(i,"..","dist","arch-ui"),m.join(i,"..","arch-ui")];for(let r of c)if(q.existsSync(m.join(r,"map.js")))return r;return null}catch{return null}},o}var ae=new Set(["strict","balanced","exploratory"]),ce=new Set(w);function L(n,e){let o=n.indexOf(e);if(!(o<0||o+1>=n.length))return n[o+1]}function le(){return new Promise((n,e)=>{let o=[];process.stdin.on("data",i=>o.push(Buffer.from(i))),process.stdin.on("end",()=>n(Buffer.concat(o).toString("utf8"))),process.stdin.on("error",e)})}async function R(n=process.argv.slice(2),e={}){if(n[0]!=="sidecar")return process.stderr.write(`usage: node index.js sidecar --stdin --out <file> --profile <strict|balanced|exploratory> [--policy <hexagonal-v1|layered-v1|vertical-v1>]
|
|
2
|
+
`),1;if(!n.includes("--stdin"))return process.stderr.write(`sidecar requires --stdin
|
|
3
|
+
`),1;let o=L(n,"--out");if(!o)return process.stderr.write(`sidecar requires --out <file>
|
|
4
|
+
`),1;let i=L(n,"--profile")??C;if(!ae.has(i))return process.stderr.write(`invalid --profile ${i}
|
|
5
|
+
`),1;let c=i,r=L(n,"--policy")??H;if(!ce.has(r))return process.stderr.write(`invalid --policy ${r} (one of ${w.join(", ")})
|
|
6
|
+
`),1;let s=r,d=e.provider===void 0?S():e.provider;if(!d)return process.stderr.write(`architecture kernel wasm missing \u2014 refusing to classify
|
|
7
|
+
`),1;let t;try{let l=(await(e.readInput??le)()).trim();if(!l)return process.stderr.write(`sidecar stdin was empty
|
|
8
|
+
`),1;t=JSON.parse(l)}catch{return process.stderr.write(`sidecar stdin was not valid JSON
|
|
9
|
+
`),1}let a=N({nodes:Array.isArray(t.nodes)?t.nodes:[],edges:Array.isArray(t.edges)?t.edges:[]}),u=d.version()||h,f=[],y=!1;for(let l of a){if(f.length>=8e3){y=!0;break}let p=d.classify({name:l.name,path:l.file_path,symbolKind:l.symbol_kind,fileKind:"source",calls:l.calls,types:[...l.parameters.map(b=>b.type_name).filter(b=>!!b),...l.return_type?[l.return_type]:[]],idents:[l.name,...l.calls].slice(0,24),profile:c,policy:s,qualifiedName:l.qualified_name,doc:l.doc??"",signature:l.signature,params:l.parameters.map(b=>b.name),returnType:l.return_type??"",calleePaths:l.callee_paths,imports:l.imports,...l.ast_role?{astRole:l.ast_role}:{},...l.effects?{effects:l.effects}:{},...l.duties?{duties:l.duties}:{}});p&&f.push({node_id:l.node_id,file_path:l.file_path,name:l.name,qualified_name:l.qualified_name,symbol_kind:l.symbol_kind,role:{primary:p.primary,alternatives:p.alternatives,confidence:p.confidence,band:p.band},purposes:p.purposes,intent:p.intent,evidence:p.evidence,...p.findings.length?{findings:p.findings}:{}})}let U={magic:I,taxonomy:P,ir:E,corpus_hash:t.provenance?.corpusHash??"",engine_version:u,profile:c,policy:s,symbols:f,...y?{symbols_capped:!0}:{}};try{v.mkdirSync(A.dirname(A.resolve(o)),{recursive:!0});let l=`${o}.${process.pid}.tmp`;v.writeFileSync(l,`${JSON.stringify(U)}
|
|
10
|
+
`),v.renameSync(l,o)}catch{return process.stderr.write(`sidecar failed to write ${o}
|
|
11
|
+
`),1}return 0}var B=["task_complexity","failure_related","repair_action","completion_confidence","finding_attention","architecture_fallback"],de="vg-decision-wasm@0.1.0",ue=1,k={schema:"vg.decision.catalog.v1",decisions:[...B]};function j(n){return typeof n=="string"&&B.includes(n)}function fe(n){return n==="high"||n==="medium"||n==="low"||n==="abstain"}function G(n,e){return Array.isArray(n)?n.filter(o=>typeof o=="string").slice(0,e):[]}function W(n){if(!n)return null;try{let e=JSON.parse(n);if(!e||typeof e!="object"||e.schema!=="vg.decision.result.v1"||e.source!=="decision"||!j(e.decision)||e.primitive!=="binary"&&e.primitive!=="choice"&&e.primitive!=="score"||typeof e.engineVersion!="string"||typeof e.packVersion!="string"||typeof e.confidence!="number"||!Number.isFinite(e.confidence)||!fe(e.band)||typeof e.abstain!="boolean"||typeof e.calibrationId!="string")return null;let o={schema:"vg.decision.result.v1",engineVersion:e.engineVersion,packVersion:e.packVersion,decision:e.decision,primitive:e.primitive,confidence:e.confidence,band:e.band,abstain:e.abstain,reasonCodes:G(e.reasonCodes,16),evidenceRefs:G(e.evidenceRefs,32),calibrationId:e.calibrationId,source:"decision"};return typeof e.selected=="string"&&(o.selected=e.selected),Array.isArray(e.probabilities)&&(o.probabilities=e.probabilities.filter(i=>typeof i?.id=="string"&&typeof i?.probability=="number"&&Number.isFinite(i.probability)).slice(0,32).map(i=>({id:i.id,probability:i.probability}))),typeof e.score=="number"&&Number.isFinite(e.score)&&(o.score=e.score),o}catch{return null}}function pe(n){try{let e=JSON.parse(n);if(!e||e.schema!=="vg.decision.catalog.v1"||!Array.isArray(e.decisions))return k;let o=e.decisions.filter(j);return o.length>0?{schema:"vg.decision.catalog.v1",decisions:o}:k}catch{return k}}function De(n){let e=n===void 0?_():n;if(!e||!e.decision)return null;let o=e.decision,i;try{i=o.version()}catch{return null}return i!==ue?null:{version:()=>de,decide:c=>{try{return W(o.decideJson(JSON.stringify(c)))}catch{return null}},decideBatch:c=>{try{let r=JSON.parse(o.decideBatchJson(JSON.stringify(c)));return Array.isArray(r)?r.map(s=>W(typeof s=="string"?s:JSON.stringify(s))).filter(s=>s!==null):[]}catch{return[]}},catalog:()=>{try{return pe(o.catalogJson())}catch{return k}}}}function me(){let n=process.argv[1];if(!n)return!1;try{return import.meta.url===ge(n).href}catch{return!1}}me()&&R().then(n=>{process.exit(n)});export{B as DECISIONS,de as DECISION_ENGINE_VERSION,ue as DECISION_ENGINE_VERSION_CODE,H as DEFAULT_POLICY,C as DEFAULT_PROFILE,D as DEFAULT_SYMBOL_CAP,h as HAILE_ENGINE_VERSION,E as HAILE_IR,I as HAILE_MAGIC,P as HAILE_TAXONOMY,w as POLICIES,N as adaptGraph,De as createDecisionProvider,S as createHaileProvider,K as isCallableKind,_ as loadWasmHaileKernel,te as resetWasmHaileKernelCache,R as runSidecarCli,F as symbolKindOf};
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@vibgrate/haile",
|
|
3
|
-
"version": "2026.
|
|
3
|
+
"version": "2026.921.1",
|
|
4
4
|
"description": "Vibgrate HAILE architecture kernel: deterministic callable classification (role, purposes, intent, band) over a vg-graph. Proprietary distribution — compiled WASM engine plus a minified loader; the taxonomy is not included.",
|
|
5
5
|
"license": "SEE LICENSE IN LICENSE",
|
|
6
6
|
"type": "module",
|
|
@@ -21,20 +21,33 @@
|
|
|
21
21
|
"build": "node build.mjs",
|
|
22
22
|
"typecheck": "tsc --noEmit -p ./",
|
|
23
23
|
"test": "vitest run",
|
|
24
|
-
"gen:logos": "node scripts/gen-logos.mjs"
|
|
24
|
+
"gen:logos": "node scripts/gen-logos.mjs",
|
|
25
|
+
"bench:decision": "tsx bench/evaluate.ts",
|
|
26
|
+
"bench:decision:failure-related": "tsx bench/evaluate.ts failure_related",
|
|
27
|
+
"bench:decision:real": "tsx bench/evaluate.ts failure_related --real",
|
|
28
|
+
"bench:decision:task-complexity": "tsx bench/evaluate.ts task_complexity",
|
|
29
|
+
"bench:decision:repair-action": "tsx bench/evaluate.ts repair_action",
|
|
30
|
+
"bench:decision:completion-confidence": "tsx bench/evaluate.ts completion_confidence",
|
|
31
|
+
"bench:decision:finding-attention": "tsx bench/evaluate.ts finding_attention",
|
|
32
|
+
"bench:decision:architecture-fallback": "tsx bench/evaluate.ts architecture_fallback",
|
|
33
|
+
"bench:decision:generate-corpus": "tsx bench/scripts/generate-corpus.ts",
|
|
34
|
+
"examples:decision": "tsx examples/decision/task-complexity.example.ts && tsx examples/decision/failure-related.example.ts && tsx examples/decision/repair-action.example.ts && tsx examples/decision/completion-confidence.example.ts && tsx examples/decision/finding-attention.example.ts && tsx examples/decision/architecture-fallback.example.ts",
|
|
35
|
+
"docs:screenshots": "node docs/generate-screenshots.mjs"
|
|
25
36
|
},
|
|
26
37
|
"devDependencies": {
|
|
27
|
-
"@types/node": "^26.
|
|
38
|
+
"@types/node": "^26.6.1",
|
|
28
39
|
"@types/react": "^18.3.0",
|
|
29
40
|
"@types/react-dom": "^18.3.0",
|
|
30
41
|
"@xyflow/react": "^12.11.6",
|
|
31
42
|
"esbuild": "^0.28.2",
|
|
43
|
+
"playwright": "^1.63.0",
|
|
32
44
|
"react": "^19.3.0",
|
|
33
45
|
"react-dom": "^19.3.0",
|
|
34
46
|
"simple-icons": "^16.31.0",
|
|
35
47
|
"tsup": "^8.0.0",
|
|
48
|
+
"tsx": "^4.23.13",
|
|
36
49
|
"typescript": "^5.4.0",
|
|
37
|
-
"vitest": "^5.0.
|
|
50
|
+
"vitest": "^5.0.1"
|
|
38
51
|
},
|
|
39
52
|
"engines": {
|
|
40
53
|
"node": ">=22.0.0"
|