wicked-crew 0.7.25 → 0.7.27

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. package/dist/api/endpoint-manifest-live.d.ts.map +1 -1
  2. package/dist/api/endpoint-manifest-live.js +5 -0
  3. package/dist/api/endpoint-manifest-live.js.map +1 -1
  4. package/dist/api/eval-compare.d.ts +306 -0
  5. package/dist/api/eval-compare.d.ts.map +1 -0
  6. package/dist/api/eval-compare.js +610 -0
  7. package/dist/api/eval-compare.js.map +1 -0
  8. package/dist/api/eval-sample.d.ts +158 -0
  9. package/dist/api/eval-sample.d.ts.map +1 -0
  10. package/dist/api/eval-sample.js +88 -0
  11. package/dist/api/eval-sample.js.map +1 -0
  12. package/dist/api/routes.d.ts +5 -0
  13. package/dist/api/routes.d.ts.map +1 -1
  14. package/dist/api/routes.js +31 -0
  15. package/dist/api/routes.js.map +1 -1
  16. package/dist/api/run-files.d.ts +12 -2
  17. package/dist/api/run-files.d.ts.map +1 -1
  18. package/dist/api/run-files.js +17 -5
  19. package/dist/api/run-files.js.map +1 -1
  20. package/dist/api/server.d.ts +20 -0
  21. package/dist/api/server.d.ts.map +1 -1
  22. package/dist/api/server.js +62 -9
  23. package/dist/api/server.js.map +1 -1
  24. package/dist/api/skills.d.ts +88 -0
  25. package/dist/api/skills.d.ts.map +1 -0
  26. package/dist/api/skills.js +264 -0
  27. package/dist/api/skills.js.map +1 -0
  28. package/dist/api/testing.d.ts +2 -75
  29. package/dist/api/testing.d.ts.map +1 -1
  30. package/dist/api/testing.js +14 -28
  31. package/dist/api/testing.js.map +1 -1
  32. package/dist/core/adapter.d.ts +42 -0
  33. package/dist/core/adapter.d.ts.map +1 -1
  34. package/dist/core/adapter.js +83 -12
  35. package/dist/core/adapter.js.map +1 -1
  36. package/dist/interactive/bridge-root.d.ts +117 -7
  37. package/dist/interactive/bridge-root.d.ts.map +1 -1
  38. package/dist/interactive/bridge-root.js +229 -9
  39. package/dist/interactive/bridge-root.js.map +1 -1
  40. package/dist/interactive/doc-delete-routes.d.ts +2 -0
  41. package/dist/interactive/doc-delete-routes.d.ts.map +1 -1
  42. package/dist/interactive/doc-delete-routes.js +3 -19
  43. package/dist/interactive/doc-delete-routes.js.map +1 -1
  44. package/dist/interactive/doc-list-routes.d.ts +42 -0
  45. package/dist/interactive/doc-list-routes.d.ts.map +1 -0
  46. package/dist/interactive/doc-list-routes.js +165 -0
  47. package/dist/interactive/doc-list-routes.js.map +1 -0
  48. package/dist/interactive/project-root.d.ts +36 -0
  49. package/dist/interactive/project-root.d.ts.map +1 -0
  50. package/dist/interactive/project-root.js +45 -0
  51. package/dist/interactive/project-root.js.map +1 -0
  52. package/dist/interactive/proxy-routes.d.ts +4 -1
  53. package/dist/interactive/proxy-routes.d.ts.map +1 -1
  54. package/dist/interactive/proxy-routes.js +4 -22
  55. package/dist/interactive/proxy-routes.js.map +1 -1
  56. package/dist/projects/default-project.d.ts +10 -0
  57. package/dist/projects/default-project.d.ts.map +1 -0
  58. package/dist/projects/default-project.js +10 -0
  59. package/dist/projects/default-project.js.map +1 -0
  60. package/dist/projects/routes.d.ts +0 -2
  61. package/dist/projects/routes.d.ts.map +1 -1
  62. package/dist/projects/routes.js +1 -2
  63. package/dist/projects/routes.js.map +1 -1
  64. package/dist/skills/bundle.d.ts +94 -0
  65. package/dist/skills/bundle.d.ts.map +1 -0
  66. package/dist/skills/bundle.js +187 -0
  67. package/dist/skills/bundle.js.map +1 -0
  68. package/dist/skills/contain.d.ts +57 -0
  69. package/dist/skills/contain.d.ts.map +1 -0
  70. package/dist/skills/contain.js +90 -0
  71. package/dist/skills/contain.js.map +1 -0
  72. package/dist/skills/core-closure.d.ts +83 -0
  73. package/dist/skills/core-closure.d.ts.map +1 -0
  74. package/dist/skills/core-closure.js +130 -0
  75. package/dist/skills/core-closure.js.map +1 -0
  76. package/dist/skills/engine-env.d.ts +48 -0
  77. package/dist/skills/engine-env.d.ts.map +1 -0
  78. package/dist/skills/engine-env.js +69 -0
  79. package/dist/skills/engine-env.js.map +1 -0
  80. package/dist/skills/frontmatter.d.ts +77 -0
  81. package/dist/skills/frontmatter.d.ts.map +1 -0
  82. package/dist/skills/frontmatter.js +190 -0
  83. package/dist/skills/frontmatter.js.map +1 -0
  84. package/dist/skills/guards.d.ts +72 -0
  85. package/dist/skills/guards.d.ts.map +1 -0
  86. package/dist/skills/guards.js +147 -0
  87. package/dist/skills/guards.js.map +1 -0
  88. package/dist/skills/live-generations.d.ts +80 -0
  89. package/dist/skills/live-generations.d.ts.map +1 -0
  90. package/dist/skills/live-generations.js +163 -0
  91. package/dist/skills/live-generations.js.map +1 -0
  92. package/dist/skills/plugin-source.d.ts +226 -0
  93. package/dist/skills/plugin-source.d.ts.map +1 -0
  94. package/dist/skills/plugin-source.js +467 -0
  95. package/dist/skills/plugin-source.js.map +1 -0
  96. package/dist/skills/refs.d.ts +58 -0
  97. package/dist/skills/refs.d.ts.map +1 -0
  98. package/dist/skills/refs.js +133 -0
  99. package/dist/skills/refs.js.map +1 -0
  100. package/dist/skills/root-fence.d.ts +53 -0
  101. package/dist/skills/root-fence.d.ts.map +1 -0
  102. package/dist/skills/root-fence.js +109 -0
  103. package/dist/skills/root-fence.js.map +1 -0
  104. package/dist/skills/root-names.d.ts +42 -0
  105. package/dist/skills/root-names.d.ts.map +1 -0
  106. package/dist/skills/root-names.js +44 -0
  107. package/dist/skills/root-names.js.map +1 -0
  108. package/dist/skills/runtime.d.ts +147 -0
  109. package/dist/skills/runtime.d.ts.map +1 -0
  110. package/dist/skills/runtime.js +320 -0
  111. package/dist/skills/runtime.js.map +1 -0
  112. package/dist/skills/store.d.ts +895 -0
  113. package/dist/skills/store.d.ts.map +1 -0
  114. package/dist/skills/store.js +3596 -0
  115. package/dist/skills/store.js.map +1 -0
  116. package/dist/skills/tree.d.ts +292 -0
  117. package/dist/skills/tree.d.ts.map +1 -0
  118. package/dist/skills/tree.js +645 -0
  119. package/dist/skills/tree.js.map +1 -0
  120. package/dist/skills/venv.d.ts +52 -0
  121. package/dist/skills/venv.d.ts.map +1 -0
  122. package/dist/skills/venv.js +80 -0
  123. package/dist/skills/venv.js.map +1 -0
  124. package/dist/studio/assets/index-CK1uJWs2.css +32 -0
  125. package/dist/studio/assets/index-FcZg3WDR.js +539 -0
  126. package/dist/studio/index.html +2 -2
  127. package/dist/studio/testid-inventory.json +605 -7
  128. package/endpoint-manifest.json +183 -1
  129. package/package.json +4 -3
  130. package/dist/studio/assets/index-C4iz4uAw.js +0 -537
  131. package/dist/studio/assets/index-DEEQcRQ_.css +0 -32
@@ -0,0 +1,610 @@
1
+ /**
2
+ * Cross-run eval comparison (docs/testing/evals-test-plan.md S17) — the OFFLINE, deterministic diff
3
+ * of two recorded {@link EvalRunDetail}s over the same corpus: which per-sample verdicts flipped,
4
+ * whether each flip is the kind a tightened store is allowed to produce (`permitted`) or the kind a
5
+ * release has to explain (`flagged`), whether the two stored summaries reconcile to the per-sample
6
+ * accounting, and which rules gained or lost exercise (core #394 `rule_coverage`).
7
+ *
8
+ * Pure over the two records: no store, no engine, no clock, no I/O — two persisted drilldowns in,
9
+ * one report out, byte-stable for the same inputs (every list is codepoint-sorted by id).
10
+ *
11
+ * # What "comparable" means here
12
+ *
13
+ * Plan §3 keys a comparison by five identities. The records carry four of them today (`corpus`,
14
+ * `rule_store`, `type_filter`, `degraded`; the rule snapshot and the engine build are §5 follow-ups),
15
+ * so `comparable` is derived from CONTENT, never from a name alone: the same corpus name AND the same
16
+ * type filter AND the same sample ids AND, per id, the same `kind` AND the same sample PAYLOAD
17
+ * identity — the `sample.payload_hash` a producer stamped on each result row (`eval-sample.js`
18
+ * `samplePayloadHash`: sha256 over the canonical JSON of id, description, kind, steering_type,
19
+ * signals). The engine echoes only id/description/kind/steering_type per row, never the input
20
+ * `signals`, so a row WITHOUT a payload hash cannot be proven to be the same action as its
21
+ * namesake in the other run: such a pair is `comparable: false` with `comparable_reason` starting
22
+ * `unverified: no sample identity` — unverified, which is not the same as "different". A sample
23
+ * present on one side only, one whose kind changed, or one whose payload hash changed means the
24
+ * corpus changed under the same name — reported, and the pair is not comparable release over
25
+ * release. Two runs under DIFFERENT type filters judged different slices of the corpus over
26
+ * different coverage denominators — not comparable either (`differing-type-filter`, see below).
27
+ * `comparable_reason` is `null` exactly when `comparable` is true.
28
+ *
29
+ * # Flip classification (the S17 rule table)
30
+ *
31
+ * bad gap → caught permitted a rule tightened: a bad behavior is now caught
32
+ * bad caught → gap flagged regression: a bad behavior is no longer caught
33
+ * good false_positive → caught permitted a good sample is no longer denied
34
+ * good caught → false_positive flagged a good sample is newly denied
35
+ * anything else flagged not a verdict pair that kind can take (evals.rs
36
+ * `evaluate_sample`: a bad sample is caught|gap, a
37
+ * good sample is caught|false_positive)
38
+ *
39
+ * `good` is a sample KIND, not a verdict — the draft plan's `good → false_positive` spelling was
40
+ * corrected in revision 3; this module is the executable form of that correction.
41
+ *
42
+ * # Coverage transitions vs. rule-set changes — what a record CANNOT tell (codex rounds 6 and 8)
43
+ *
44
+ * The wire carries `rule_coverage.exercised` as a COUNT and `unexercised` as a LIST of ids
45
+ * (api-types `GovernanceEvalRuleCoverage`; wicked-core `crates/wicked-governance/src/evals.rs` on
46
+ * branch `feat/evals-effect-and-coverage` (core #394/#395, PR #398) at `a87e461` — `RuleCoverage`
47
+ * lines 323-328: `exercised: usize`, `unexercised: Vec<UnexercisedRule>`, `recall_only: usize`,
48
+ * `per_type: BTreeMap<String, TypeCoverage>`). The partition is over the ELIGIBLE rules — the
49
+ * active, effect-bearing rules of the slice (`decide_lane_rules` lines 816-843: `effect.is_some()
50
+ * && !retired && in_slice`); `recall_only` counts the effect-less active rules OUTSIDE it (docs
51
+ * lines 317-319, computed at 875-878). A rule is EXERCISED when it appeared in ANY evaluated claim's
52
+ * `policy_ids`, whatever its effect (`RuleCoverage` docs lines 312-314; `rule_coverage()` lines
53
+ * 849-880 partitions the eligible rules by `triggered`, the union `run_evals` collects of every
54
+ * claim's `policy_ids`). A result row's `fired`, however, is the BLOCKING subset only:
55
+ * `evaluate_sample` lines 906-916 keep the ids whose effect is `Deny`. Hence, for an UNFILTERED run,
56
+ *
57
+ * fired(run) ⊆ exercised(run), and `exercised − |fired|` rules were exercised by a non-blocking
58
+ * (`warn`) effect ALONE — counted, but NEVER named anywhere in the record.
59
+ *
60
+ * NO field of the wire lists a run's rule set: `exercised` and `recall_only` are counts,
61
+ * `unexercised` names only the rules nothing exercised, `fired` only the blocking firings. The rule
62
+ * identities a record enumerates are therefore exactly `unexercised ∪ fired`, and an id one record
63
+ * enumerates and the other does not is NOT thereby absent from the other's store: it may be one of
64
+ * that side's unnamed warn-exercised rules, an effect-less (recall-only) or retired rule outside the
65
+ * eligible partition, under a type filter a rule of another type outside the denominator — or a rule
66
+ * the store really gained or lost. Round 6 replaced a reconstruction of "the inventory" as
67
+ * `unexercised ∪ fired` with a completeness inference: `exercised === |fired|` ⇒ every exercised rule
68
+ * is named ⇒ the silent side's absence was asserted as `added_rules`/`removed_rules`. Round 8 deleted
69
+ * that inference too — under a type filter a fired id is typed into the slice only by the OTHER run's
70
+ * `unexercised` row, and a rule's type in run A does not establish its type in run B (codex's
71
+ * reproduction: A lists R and Q unexercised under `development`; B moves R to `security`, fires R
72
+ * blocking and exercises Q by `warn`; the cross-run intersection named R, declared B complete and
73
+ * reported Q — still present and exercised — as `removed_rules`). The delta asserts ONLY what the
74
+ * records state:
75
+ *
76
+ * gained unexercised in A (listed) AND blocking-fired in B (listed): a sample now exercises
77
+ * the rule — a statement about the RULE, both ends named, certain
78
+ * lost blocking-fired in A AND unexercised in B (listed): certain
79
+ * unidentified PER RECORD, how many exercised rules of its denominator it does not name:
80
+ * unfiltered `exercised − |fired|`; under a type filter the whole `exercised` (a
81
+ * fired id carries no steering_type, so a record types none of its own firings into
82
+ * the slice) — never reduced by what the OTHER run lists
83
+ * added_rules / asserted only from an explicit rule inventory on BOTH records — the wire carries
84
+ * removed_rules none, so both are EMPTY for every daemon-recorded run and `inventory` is `partial`
85
+ * transitions_withheld
86
+ * every id enumerated by one record and not the other, one reason per id saying what
87
+ * it may be besides a rule-set change — WITHHELD, never guessed
88
+ *
89
+ * `exercised_delta` is always the difference of the two counts. Reconciliation errors on coverage are
90
+ * the record contradicting ITSELF (or being malformed — below); two valid reports always `reconcile`.
91
+ *
92
+ * # Type filters (codex round 7) — the denominator is the SLICE; the firings are not
93
+ *
94
+ * A run recorded under `type_filter: T` was produced by `rules eval --type T`. In the engine that
95
+ * filter slices the SAMPLES (`run_evals` lines 974-977: only samples of type T are judged) and the
96
+ * COVERAGE DENOMINATOR (`decide_lane_rules` lines 816-844, `in_slice` at 820: only type-T rules are
97
+ * eligible; `rule_coverage` 849-880 partitions exactly those; `recall_only` 875-878 is sliced too)
98
+ * but NOT the gate: `evaluate_sample` runs `select_any` over every active rule whatever its type
99
+ * (line 901) and a row's `fired` keeps every blocking id it produced (906-916). So under a filter a
100
+ * row may fire a rule OUTSIDE the denominator — codex's reproduction: a development-filtered run
101
+ * with `fired: ["SECURITY-DENY"]` and `exercised: 0` is a VALID record (the security rule is not a
102
+ * development rule), and the unfiltered invariant `exercised ≥ |fired|` does not hold. Hence, per
103
+ * record ({@link EvalCoverageReconciliation}, reported per side in `coverage_reconciliation`):
104
+ *
105
+ * type_filter null `'rows'` — every active rule is in the denominator, so the rows'
106
+ * blocking-fired ids are exercised rules: `exercised ≥ |fired|`, and
107
+ * the exercised set is complete iff `exercised === |fired|`.
108
+ * type_filter T, per_type `'per_type'` — reconciled against the engine's OWN row for the slice
109
+ * present (`rule_coverage.per_type[T]`, `TypeCoverage` lines 300-303):
110
+ * `per_type[T].exercised === exercised`; the rows' fired ids carry no
111
+ * steering type and are NOT a denominator check.
112
+ * type_filter T, no `'n/a (engine reports no per-type coverage)'` — the record carries
113
+ * per_type nothing to check its slice against; nothing fired-based is asserted.
114
+ *
115
+ * Whatever the filter: `per_type`, when present, must sum to `exercised` and agree per type with the
116
+ * listed `unexercised` rows; under a filter every listed row is of the filter's type; no id is
117
+ * listed twice; and no listed-unexercised id fired — that row TYPES the rule into the slice, and a
118
+ * blocking firing is a firing, which makes an eligible rule exercised (862-873) — so this last check
119
+ * is sound under a filter too (the fix brief's floor was "unfiltered only"; the engine's definition
120
+ * makes it general).
121
+ *
122
+ * Two runs under DIFFERENT filters get NO `rule_coverage_delta` (an inventory diff across
123
+ * denominators would report the other slice's rules as changes) and are not comparable. Under the
124
+ * SAME filter T the delta is computed exactly as unfiltered — `gained`/`lost` from ids listed on both
125
+ * ends (the unexercised row carries `steering_type: T`, the firing is a firing), every one-sided id
126
+ * withheld — except that `unidentified` is the whole `exercised` count (see above) and each withheld
127
+ * reason adds that the id may be a rule of another type outside the denominator.
128
+ *
129
+ * # Malformed persisted coverage (codex round 8)
130
+ *
131
+ * The daemon persists a run's `rule_coverage` verbatim and validates none of it (the script's
132
+ * `verifyEngineReport` gates only its own reports). A recorded `rule_coverage` that is not the wire
133
+ * shape — `null`; `unexercised` not an array; `per_type: null`, a row that is not `{ exercised,
134
+ * unexercised }` of non-negative integers, a key that is no steering type … — cannot be reconciled:
135
+ * that side is `coverage_reconciliation: 'unverified (malformed rule_coverage)'`
136
+ * ({@link MALFORMED_RULE_COVERAGE}), a reconciliation error names the run and the defect, no
137
+ * `rule_coverage_delta` is computed, and the comparison never throws over persisted data.
138
+ *
139
+ * # Malformed persisted result rows (Copilot on #475)
140
+ *
141
+ * The same store validates a detail's `results` only as "an array" (`EvalRunStore.get()`), so a row
142
+ * that is not the wire shape (`GovernanceEvalResult` — no `sample`, `sample.id` not a non-empty
143
+ * string, a `kind` or `verdict` outside its union, `fired` not an array of strings, a row that is no
144
+ * object at all) can reach the comparison from a corrupted or hand-edited file. Every row is checked
145
+ * ({@link resultRowProblem}) BEFORE it is indexed, tallied or iterated: a malformed one is ONE
146
+ * reconciliation error naming the run, the index and the defect, and is EXCLUDED from everything —
147
+ * the id index, the identity counts, the tally, the coverage `fired` set. The pair is then not
148
+ * comparable ({@link MALFORMED_RESULT_ROWS}), and that side's stored summary is NOT checked against a
149
+ * results list the comparison could not fully read (the shortfall is the excluded rows, not a summary
150
+ * defect — reporting it as one would misattribute it). Never a throw.
151
+ */
152
+ import { PAYLOAD_HASH_RE } from './eval-sample.js';
153
+ import { STEERING_TYPE_VALUES, STEERING_TYPES } from './governance-steering.js';
154
+ /** The exact prefix of `comparable_reason` when a side carries result rows without a payload hash. */
155
+ export const UNVERIFIED_NO_SAMPLE_IDENTITY = 'unverified: no sample identity';
156
+ /** The exact prefix of `comparable_reason` when the two runs were produced under different type filters. */
157
+ export const DIFFERING_TYPE_FILTER = 'differing-type-filter';
158
+ /** The exact `coverage_reconciliation` value of a side whose persisted `rule_coverage` is not the wire
159
+ * shape — unverified, not reconciled, no delta computed over it (module doc, "Malformed persisted
160
+ * coverage"). */
161
+ export const MALFORMED_RULE_COVERAGE = 'unverified (malformed rule_coverage)';
162
+ /** The `comparable_reason` prefix of a pair one side of which persisted result rows that are not the
163
+ * wire shape (api-types `GovernanceEvalResult`, module doc "Malformed persisted result rows"): each
164
+ * is named in `reconciliation_errors` and EXCLUDED, never thrown over — and a comparison over a
165
+ * record it could not fully read asserts nothing (Copilot on #475). */
166
+ export const MALFORMED_RESULT_ROWS = 'unverified: malformed result row(s)';
167
+ /** The S17 rule table, as a function. Exported so a caller can classify a single flip. */
168
+ export function classifyFlip(kind, from, to) {
169
+ if (kind === 'bad') {
170
+ if (from === 'gap' && to === 'caught')
171
+ return { classification: 'permitted', reason: 'a rule tightened: a bad behavior is now caught' };
172
+ if (from === 'caught' && to === 'gap')
173
+ return { classification: 'flagged', reason: 'regression: a bad behavior is no longer caught' };
174
+ }
175
+ else {
176
+ if (from === 'false_positive' && to === 'caught')
177
+ return { classification: 'permitted', reason: 'a good sample is no longer denied' };
178
+ if (from === 'caught' && to === 'false_positive')
179
+ return { classification: 'flagged', reason: 'a good sample is newly denied' };
180
+ }
181
+ return {
182
+ classification: 'flagged',
183
+ reason: `inconsistent: ${from} → ${to} is not a verdict pair a ${kind} sample can take (a bad sample is caught|gap, a good sample is caught|false_positive)`,
184
+ };
185
+ }
186
+ /** Compare two recorded runs — A is the baseline, B the candidate. See the module doc. */
187
+ export function compareEvalRuns(a, b) {
188
+ const errors = [];
189
+ // Only WELL-FORMED rows take part (module doc, "Malformed persisted result rows"): the store
190
+ // validated `results` as an array and nothing more, so every row is checked here BEFORE it is
191
+ // indexed, tallied or iterated — a malformed one is named in `errors` and excluded, never thrown over.
192
+ const rowsA = wellFormedRows(a, 'a', errors);
193
+ const rowsB = wellFormedRows(b, 'b', errors);
194
+ const malformed_rows = { a: a.results.length - rowsA.length, b: b.results.length - rowsB.length };
195
+ const mapA = indexById(rowsA, a, 'a', errors);
196
+ const mapB = indexById(rowsB, b, 'b', errors);
197
+ const only_in_a = [...mapA.keys()].filter((id) => !mapB.has(id)).sort(codepoint);
198
+ const only_in_b = [...mapB.keys()].filter((id) => !mapA.has(id)).sort(codepoint);
199
+ const shared = [...mapA.keys()].filter((id) => mapB.has(id)).sort(codepoint);
200
+ const flips = [];
201
+ const kind_changed = [];
202
+ const payload_changed = [];
203
+ let unchanged = 0;
204
+ for (const id of shared) {
205
+ const ra = mapA.get(id);
206
+ const rb = mapB.get(id);
207
+ const kindChanged = ra.sample.kind !== rb.sample.kind;
208
+ const ha = identityOf(ra);
209
+ const hb = identityOf(rb);
210
+ const payloadChanged = ha !== null && hb !== null && ha !== hb;
211
+ if (kindChanged)
212
+ kind_changed.push(id);
213
+ if (payloadChanged)
214
+ payload_changed.push(id);
215
+ if (ra.verdict !== rb.verdict) {
216
+ let verdictClass;
217
+ if (kindChanged) {
218
+ verdictClass = {
219
+ classification: 'flagged',
220
+ reason: `the sample's kind changed between runs (${ra.sample.kind} → ${rb.sample.kind}): the corpus changed under this name`,
221
+ };
222
+ }
223
+ else if (payloadChanged) {
224
+ verdictClass = {
225
+ classification: 'flagged',
226
+ reason: "the sample's payload changed between runs (description, steering_type or signals): the corpus changed under this name",
227
+ };
228
+ }
229
+ else {
230
+ verdictClass = classifyFlip(rb.sample.kind, ra.verdict, rb.verdict);
231
+ }
232
+ flips.push({ sample_id: id, kind: rb.sample.kind, from: ra.verdict, to: rb.verdict, ...verdictClass });
233
+ }
234
+ else if (!kindChanged && !payloadChanged) {
235
+ unchanged += 1;
236
+ }
237
+ }
238
+ const unverified_rows = {
239
+ a: rowsA.filter((r) => identityOf(r) === null).length,
240
+ b: rowsB.filter((r) => identityOf(r) === null).length,
241
+ };
242
+ // Each run's summary must be its own results' tally — a disagreement is a stored defect, and a
243
+ // delta over a defective summary would attribute it to a flip.
244
+ // A side whose results could not be FULLY read is not checked here: its tally over the well-formed
245
+ // rows is short by exactly the excluded (already named) rows, and reporting that shortfall as a
246
+ // summary defect would misattribute it. The record is already `reconciles: false`.
247
+ for (const [run, rows, label] of [
248
+ [a, rowsA, 'a'],
249
+ [b, rowsB, 'b'],
250
+ ]) {
251
+ if (rows.length !== run.results.length)
252
+ continue;
253
+ const own = tally(rows);
254
+ if (!sameSummary(own, run.summary)) {
255
+ errors.push(`run ${label} (${run.id}): stored summary ${fmt(run.summary)} does not match its own results ${fmt(own)}`);
256
+ }
257
+ }
258
+ const summary_delta = {
259
+ total: b.summary.total - a.summary.total,
260
+ caught: b.summary.caught - a.summary.caught,
261
+ gaps: b.summary.gaps - a.summary.gaps,
262
+ false_positives: b.summary.false_positives - a.summary.false_positives,
263
+ };
264
+ // The per-sample accounting the delta must equal: every flip moves one count from `from` to
265
+ // `to`; a sample only in B adds its verdict, one only in A removes it.
266
+ const expected = { total: only_in_b.length - only_in_a.length, caught: 0, gaps: 0, false_positives: 0 };
267
+ for (const f of flips) {
268
+ expected[field(f.to)] += 1;
269
+ expected[field(f.from)] -= 1;
270
+ }
271
+ for (const id of only_in_b)
272
+ expected[field(mapB.get(id).verdict)] += 1;
273
+ for (const id of only_in_a)
274
+ expected[field(mapA.get(id).verdict)] -= 1;
275
+ // Withheld, for the same reason, when either side carries excluded rows: the stored summaries count
276
+ // rows the per-sample accounting could not read.
277
+ if (malformed_rows.a === 0 && malformed_rows.b === 0 && !sameSummary(expected, summary_delta)) {
278
+ errors.push(`summary delta ${fmt(summary_delta)} does not reconcile to the per-sample accounting ${fmt(expected)}`);
279
+ }
280
+ const identity = {
281
+ corpus: pair(a.corpus, b.corpus),
282
+ rule_store: pair(a.rule_store, b.rule_store),
283
+ type_filter: pair(a.type_filter, b.type_filter),
284
+ degraded: pair(a.degraded, b.degraded),
285
+ };
286
+ // Why the pair is not comparable, in verification order: a side with excluded (malformed) rows,
287
+ // then an unverified side, first (nothing below can be asserted about actions whose identity is
288
+ // unknown), then the identity and content differences.
289
+ const reasons = [];
290
+ if (malformed_rows.a > 0 || malformed_rows.b > 0) {
291
+ reasons.push(`${MALFORMED_RESULT_ROWS} (${malformed_rows.a} result row(s) in a and ${malformed_rows.b} in b are not the wire shape — each named in reconciliation_errors and excluded)`);
292
+ }
293
+ if (unverified_rows.a > 0 || unverified_rows.b > 0) {
294
+ reasons.push(`${UNVERIFIED_NO_SAMPLE_IDENTITY} (${unverified_rows.a} result row(s) in a and ${unverified_rows.b} in b carry no well-formed sample.payload_hash)`);
295
+ }
296
+ if (!identity.corpus.same)
297
+ reasons.push(`corpus name differs (${JSON.stringify(a.corpus)} vs ${JSON.stringify(b.corpus)})`);
298
+ if (!identity.type_filter.same) {
299
+ reasons.push(`${DIFFERING_TYPE_FILTER} (a: ${JSON.stringify(a.type_filter)}, b: ${JSON.stringify(b.type_filter)}): the runs judged different slices of the corpus over different coverage denominators`);
300
+ }
301
+ if (only_in_a.length > 0 || only_in_b.length > 0) {
302
+ reasons.push(`sample set differs (${only_in_a.length} id(s) only in a, ${only_in_b.length} only in b)`);
303
+ }
304
+ if (kind_changed.length > 0)
305
+ reasons.push(`kind changed for ${kind_changed.length} shared sample(s)`);
306
+ if (payload_changed.length > 0)
307
+ reasons.push(`payload changed for ${payload_changed.length} shared sample(s) (description, steering_type or signals)`);
308
+ // Each record's coverage is reconciled ON ITS OWN whenever it carries one — a record that
309
+ // contradicts itself is a defect whether or not the other side measured coverage. Coverage is
310
+ // OPTIONAL on a record (a pre-#394 engine emits none: `undefined` here); a PRESENT value that is
311
+ // not the wire shape is named as a defect and never thrown over (`null` here).
312
+ const covA = a.rule_coverage === undefined ? undefined : reconcileCoverage(a, rowsA, 'a', errors);
313
+ const covB = b.rule_coverage === undefined ? undefined : reconcileCoverage(b, rowsB, 'b', errors);
314
+ const modeOf = (cov) => (cov === undefined ? null : cov === null ? MALFORMED_RULE_COVERAGE : cov.mode);
315
+ const comparison = {
316
+ a: a.id,
317
+ b: b.id,
318
+ identity,
319
+ comparable: reasons.length === 0,
320
+ comparable_reason: reasons.length === 0 ? null : reasons.join('; '),
321
+ flips,
322
+ unchanged,
323
+ only_in_a,
324
+ only_in_b,
325
+ kind_changed,
326
+ payload_changed,
327
+ unverified_rows,
328
+ summary_delta,
329
+ reconciles: true,
330
+ reconciliation_errors: errors,
331
+ coverage_reconciliation: { a: modeOf(covA), b: modeOf(covB) },
332
+ };
333
+ // The delta exists only when both sides measured WELL-FORMED coverage over the SAME denominator —
334
+ // never fabricated from one side, never over a malformed record, never across two different slices.
335
+ if (covA && covB && identity.type_filter.same) {
336
+ comparison.rule_coverage_delta = coverageDelta(covA, covB, a.type_filter);
337
+ }
338
+ comparison.reconciles = errors.length === 0;
339
+ return comparison;
340
+ }
341
+ // ── helpers ──────────────────────────────────────────────────────────────────────────────────────
342
+ /** Codepoint order — deterministic on every machine (`localeCompare` is locale-bound). */
343
+ function codepoint(x, y) {
344
+ return x < y ? -1 : x > y ? 1 : 0;
345
+ }
346
+ function pair(a, b) {
347
+ return { a, b, same: a === b };
348
+ }
349
+ /** The row's stamped payload identity — only when it is WELL-FORMED (`PAYLOAD_HASH_RE`: `sha256:`
350
+ * + 64 lowercase hex, the one spelling `samplePayloadHash` produces). Anything else — none, an
351
+ * empty string, another algorithm, uppercase, a truncated digest — is null: the row is UNVERIFIED,
352
+ * never treated as an authoritative identity that could mark a pair comparable or "changed"
353
+ * (the value comes from persisted run data; Copilot on #475). */
354
+ function identityOf(r) {
355
+ const h = r.sample.payload_hash;
356
+ return typeof h === 'string' && PAYLOAD_HASH_RE.test(h) ? h : null;
357
+ }
358
+ const RESULT_KINDS = new Set(['good', 'bad']);
359
+ const RESULT_VERDICTS = new Set(['caught', 'gap', 'false_positive']);
360
+ /**
361
+ * Why a persisted result row is not the wire shape (api-types `GovernanceEvalResult`) in the fields
362
+ * this comparison READS — `sample.id` (a non-empty string), `sample.kind` (`good|bad`), `verdict`
363
+ * (`caught|gap|false_positive`), `fired` (an array of rule ids) — or null when it is. Checked BEFORE
364
+ * any row is indexed, tallied or iterated: `EvalRunStore.get()` validates only that `results` is an
365
+ * array, so a corrupted or hand-edited detail file can carry a row without a `sample`, with
366
+ * `fired: null`, or with a verdict `field()` maps nowhere (Copilot on #475: `identityOf` / `indexById`
367
+ * threw, `tally` would count into an undefined field). `payload_hash` is not checked here —
368
+ * `identityOf` already treats anything malformed as no identity.
369
+ */
370
+ function resultRowProblem(r) {
371
+ if (!isPlainObject(r))
372
+ return `expected an object { sample, verdict, fired }, got ${JSON.stringify(r)}`;
373
+ const sample = r['sample'];
374
+ if (!isPlainObject(sample))
375
+ return `sample ${JSON.stringify(sample)} is not an object { id, kind }`;
376
+ if (typeof sample['id'] !== 'string' || sample['id'] === '')
377
+ return `sample.id ${JSON.stringify(sample['id'])} is not a non-empty string`;
378
+ if (typeof sample['kind'] !== 'string' || !RESULT_KINDS.has(sample['kind']))
379
+ return `sample.kind ${JSON.stringify(sample['kind'])} is not good|bad`;
380
+ if (typeof r['verdict'] !== 'string' || !RESULT_VERDICTS.has(r['verdict']))
381
+ return `verdict ${JSON.stringify(r['verdict'])} is not caught|gap|false_positive`;
382
+ const fired = r['fired'];
383
+ if (!Array.isArray(fired) || !fired.every((id) => typeof id === 'string'))
384
+ return `fired ${JSON.stringify(fired)} is not an array of rule ids`;
385
+ return null;
386
+ }
387
+ /** The run's WELL-FORMED result rows ({@link resultRowProblem}). Every other row is ONE reconciliation
388
+ * error naming the run, the row's index and the defect, and takes no part in the comparison. */
389
+ function wellFormedRows(run, label, errors) {
390
+ const rows = [];
391
+ for (const [i, r] of run.results.entries()) {
392
+ const problem = resultRowProblem(r);
393
+ if (problem === null)
394
+ rows.push(r);
395
+ else {
396
+ errors.push(`run ${label} (${run.id}): results[${i}] is malformed — ${problem} — not the wire shape (api-types GovernanceEvalResult), so it is excluded from the per-sample comparison, the identity counts, the tally and the coverage reconciliation`);
397
+ }
398
+ }
399
+ return rows;
400
+ }
401
+ /** The run's well-formed results by sample id. A duplicate id inside ONE run (the engine rejects them
402
+ * at import — an edited record could still carry one) is a reconciliation error, and the last row wins. */
403
+ function indexById(rows, run, label, errors) {
404
+ const map = new Map();
405
+ for (const r of rows) {
406
+ if (map.has(r.sample.id))
407
+ errors.push(`run ${label} (${run.id}): duplicate sample id ${r.sample.id} in results`);
408
+ map.set(r.sample.id, r);
409
+ }
410
+ return map;
411
+ }
412
+ function isPlainObject(x) {
413
+ return x !== null && typeof x === 'object' && !Array.isArray(x);
414
+ }
415
+ function isCount(x) {
416
+ return Number.isInteger(x) && x >= 0;
417
+ }
418
+ /**
419
+ * Why a persisted `rule_coverage` is not the wire shape (api-types `GovernanceEvalRuleCoverage`
420
+ * 0.27.0), or null when it is — checked BEFORE any field is read, because the store writes the
421
+ * engine's value verbatim and the comparison must name a malformed record, never throw over it
422
+ * (codex round 8: `per_type: null` made `Object.values` throw). The shape: `exercised` a non-negative
423
+ * integer; `unexercised` an array of `{ rule_id: non-empty string, steering_type: one of the seven }`;
424
+ * `recall_only`, when present, a non-negative integer; `per_type`, when present, a plain object whose
425
+ * keys are steering types and whose values are `{ exercised, unexercised }` of non-negative integers.
426
+ */
427
+ function coverageShapeProblem(rc) {
428
+ if (!isPlainObject(rc))
429
+ return `expected an object { exercised, unexercised[] }, got ${JSON.stringify(rc)}`;
430
+ if (!isCount(rc['exercised']))
431
+ return `exercised ${JSON.stringify(rc['exercised'])} is not a non-negative integer`;
432
+ const unexercised = rc['unexercised'];
433
+ if (!Array.isArray(unexercised))
434
+ return `unexercised ${JSON.stringify(unexercised)} is not an array`;
435
+ for (const [i, u] of unexercised.entries()) {
436
+ if (!isPlainObject(u) || typeof u['rule_id'] !== 'string' || u['rule_id'] === '')
437
+ return `unexercised[${i}] carries no string rule_id (got ${JSON.stringify(u)})`;
438
+ if (typeof u['steering_type'] !== 'string' || !STEERING_TYPES.has(u['steering_type'])) {
439
+ return `unexercised[${i}] (${u['rule_id']}) steering_type ${JSON.stringify(u['steering_type'])} is not one of ${STEERING_TYPE_VALUES.join('|')}`;
440
+ }
441
+ }
442
+ if (rc['recall_only'] !== undefined && !isCount(rc['recall_only']))
443
+ return `recall_only ${JSON.stringify(rc['recall_only'])} is not a non-negative integer`;
444
+ const perType = rc['per_type'];
445
+ if (perType !== undefined) {
446
+ if (!isPlainObject(perType))
447
+ return `per_type ${JSON.stringify(perType)} is not an object keyed by steering type`;
448
+ for (const [t, row] of Object.entries(perType)) {
449
+ if (!STEERING_TYPES.has(t))
450
+ return `per_type key ${JSON.stringify(t)} is not one of ${STEERING_TYPE_VALUES.join('|')}`;
451
+ if (!isPlainObject(row))
452
+ return `per_type.${t} ${JSON.stringify(row)} is not an object { exercised, unexercised }`;
453
+ for (const k of ['exercised', 'unexercised']) {
454
+ if (!isCount(row[k]))
455
+ return `per_type.${t}.${k} ${JSON.stringify(row[k])} is not a non-negative integer`;
456
+ }
457
+ }
458
+ }
459
+ return null;
460
+ }
461
+ /**
462
+ * Reconcile one record's `rule_coverage` with ITSELF and with its rows, pushing every contradiction
463
+ * to `errors` (naming the run), and say what it was checked against (`mode`). A value that is not
464
+ * the wire shape (`coverageShapeProblem`) is ONE error naming the defect and `null` — nothing else is
465
+ * read from it. Whatever the filter: no duplicate unexercised id; `per_type`, when present, sums to
466
+ * `exercised` and agrees per type with the listed rows; no listed-unexercised id fired (the row types
467
+ * the rule into the slice; a firing makes an eligible rule exercised — evals.rs 862-873). Unfiltered
468
+ * (`'rows'`): `exercised` is at least the distinct blocking-fired ids. Filtered: every listed row is
469
+ * of the filter's type, and with `per_type` present (`'per_type'`) the slice's own row equals the
470
+ * total — the rows' fired ids may name rules of other types and are NOT a denominator check; without
471
+ * `per_type` (`'n/a …'`) nothing more can be checked.
472
+ */
473
+ function reconcileCoverage(run, rows, label, errors) {
474
+ const who = `run ${label} (${run.id})`;
475
+ const malformed = coverageShapeProblem(run.rule_coverage);
476
+ if (malformed !== null) {
477
+ errors.push(`${who}: rule_coverage is malformed — ${malformed} — not the wire shape (api-types GovernanceEvalRuleCoverage), so it is not reconciled and no coverage delta is computed over it`);
478
+ return null;
479
+ }
480
+ const rc = run.rule_coverage;
481
+ const filter = run.type_filter;
482
+ const fired = new Set();
483
+ for (const r of rows)
484
+ for (const id of r.fired)
485
+ fired.add(id);
486
+ const unexercised = new Set();
487
+ const listedByType = new Map();
488
+ for (const u of rc.unexercised) {
489
+ if (unexercised.has(u.rule_id))
490
+ errors.push(`${who}: rule_coverage.unexercised lists ${u.rule_id} twice — a rule is unexercised once or not at all`);
491
+ unexercised.add(u.rule_id);
492
+ listedByType.set(u.steering_type, (listedByType.get(u.steering_type) ?? 0) + 1);
493
+ if (filter !== null && u.steering_type !== filter) {
494
+ errors.push(`${who}: rule_coverage.unexercised lists ${u.rule_id} (${u.steering_type}) under type filter ${filter} — the slice's eligible rules are all of the filter's type (evals.rs decide_lane_rules in_slice)`);
495
+ }
496
+ if (fired.has(u.rule_id)) {
497
+ errors.push(`${who}: rule_coverage.unexercised lists ${u.rule_id}, which fired (blocking) in its results — a rule that fired for any sample is exercised, never unexercised`);
498
+ }
499
+ }
500
+ // `per_type` is read as possibly incomplete: a persisted record is verbatim, and a row missing
501
+ // for a type is a contradiction to NAME, not an index to crash on.
502
+ const perType = rc.per_type;
503
+ if (perType !== undefined) {
504
+ let sumExercised = 0;
505
+ for (const row of Object.values(perType))
506
+ if (row !== undefined)
507
+ sumExercised += row.exercised;
508
+ if (sumExercised !== rc.exercised) {
509
+ errors.push(`${who}: rule_coverage.per_type sums to ${sumExercised} exercised but rule_coverage.exercised is ${rc.exercised} — the per-type rows partition the same eligible rules`);
510
+ }
511
+ for (const t of [...new Set([...Object.keys(perType), ...listedByType.keys()])].sort(codepoint)) {
512
+ const counted = perType[t]?.unexercised;
513
+ const listed = listedByType.get(t) ?? 0;
514
+ if (counted === undefined)
515
+ errors.push(`${who}: rule_coverage.unexercised lists ${listed} ${t} rule(s) but rule_coverage.per_type carries no ${t} row`);
516
+ else if (counted !== listed)
517
+ errors.push(`${who}: rule_coverage.per_type.${t}.unexercised is ${counted} but rule_coverage.unexercised lists ${listed} ${t} rule(s)`);
518
+ }
519
+ }
520
+ const facts = { exercised: rc.exercised, unexercised, fired };
521
+ if (filter === null) {
522
+ if (rc.exercised < fired.size) {
523
+ errors.push(`${who}: rule_coverage.exercised ${rc.exercised} is below the ${fired.size} distinct rule id(s) fired (blocking) across its results (${[...fired].sort(codepoint).join(', ')}) — every blocking firing is an exercised rule`);
524
+ }
525
+ return { mode: 'rows', ...facts };
526
+ }
527
+ if (perType === undefined)
528
+ return { mode: 'n/a (engine reports no per-type coverage)', ...facts };
529
+ const slice = perType[filter];
530
+ if (slice === undefined) {
531
+ errors.push(`${who}: rule_coverage.per_type carries no ${filter} row although the run was filtered to ${filter} — the slice's own numbers are missing`);
532
+ }
533
+ else if (slice.exercised !== rc.exercised) {
534
+ errors.push(`${who}: rule_coverage.per_type.${filter}.exercised is ${slice.exercised} but rule_coverage.exercised is ${rc.exercised} — under type filter ${filter} the eligible rules are exactly the ${filter} slice (evals.rs decide_lane_rules), so the slice's row IS the total`);
535
+ }
536
+ return { mode: 'per_type', ...facts };
537
+ }
538
+ /**
539
+ * The delta of two records produced under the SAME `type_filter` (`filter`), over what they ENUMERATE
540
+ * (module doc, "Coverage transitions vs. rule-set changes"): `gained`/`lost` from ids listed on both
541
+ * ends, `unidentified` per record, and EVERY one-sided id withheld with its reason — the wire carries
542
+ * no rule inventory, so `added_rules`/`removed_rules` are never asserted and `inventory` is `partial`.
543
+ */
544
+ function coverageDelta(covA, covB, filter) {
545
+ // Certain, both ends listed: the unexercised row names the id on one side, the blocking firing on
546
+ // the other (under a filter the row carries the filter's steering_type; a firing is a firing).
547
+ const gained = [...covA.unexercised].filter((id) => covB.fired.has(id)).sort(codepoint);
548
+ const lost = [...covA.fired].filter((id) => covB.unexercised.has(id)).sort(codepoint);
549
+ const unidentified = { a: unidentifiedOf(covA, filter), b: unidentifiedOf(covB, filter) };
550
+ // The identities each record ENUMERATES — its unexercised list plus its blocking-fired ids.
551
+ // Nothing else about a run's rule set is on the wire, so an id the other side does not enumerate
552
+ // is a candidate for a rule-set change and NOTHING more: withheld, with what else it may be.
553
+ const knownA = new Set([...covA.unexercised, ...covA.fired]);
554
+ const knownB = new Set([...covB.unexercised, ...covB.fired]);
555
+ const withheld = [
556
+ ...[...knownB].filter((id) => !knownA.has(id)).map((id) => withheldReason('added_rules', id, covB, unidentified.a, filter)),
557
+ ...[...knownA].filter((id) => !knownB.has(id)).map((id) => withheldReason('removed_rules', id, covA, unidentified.b, filter)),
558
+ ].sort(codepoint);
559
+ return {
560
+ exercised_delta: covB.exercised - covA.exercised,
561
+ inventory: 'partial',
562
+ unidentified,
563
+ gained,
564
+ lost,
565
+ added_rules: [],
566
+ removed_rules: [],
567
+ transitions_withheld: withheld,
568
+ };
569
+ }
570
+ /** How many exercised rules of its denominator ONE record does not name. Unfiltered, every
571
+ * blocking-fired id is an exercised rule of the (universal) denominator, so `exercised − |fired|`
572
+ * (floored at 0: an undercut was already reported as a reconciliation error). Under a type filter a
573
+ * fired id carries no steering_type — the record types none of its own firings into the slice — so
574
+ * the whole count; what the OTHER run lists never reduces it (codex round 8). */
575
+ function unidentifiedOf(cov, filter) {
576
+ return filter === null ? Math.max(0, cov.exercised - cov.fired.size) : cov.exercised;
577
+ }
578
+ /** Why an id enumerated by one record (`lister`: b for `added_rules`, a for `removed_rules`) and not
579
+ * by the other is withheld instead of asserted as a rule-set change — one sentence naming what the
580
+ * id may be instead. `silentUnidentified` is the silent side's `unidentified` count. */
581
+ function withheldReason(kind, id, lister, silentUnidentified, filter) {
582
+ const [listerLabel, silent] = kind === 'added_rules' ? ['b', 'a'] : ['a', 'b'];
583
+ const listedAs = lister.unexercised.has(id) ? 'unexercised' : 'blocking-fired';
584
+ const maybe = [];
585
+ if (silentUnidentified > 0) {
586
+ maybe.push(`one of ${silent}'s ${silentUnidentified} exercised rule(s) named nowhere (${filter === null ? 'fired by a non-blocking effect alone' : `under type filter ${filter} no fired id is typed into the slice`})`);
587
+ }
588
+ maybe.push('an effect-less (recall-only) or retired rule outside the eligible partition');
589
+ if (filter !== null)
590
+ maybe.push(`a rule of another type outside the ${filter} denominator (a row's fired ids carry no steering_type)`);
591
+ const change = `a rule the store ${kind === 'added_rules' ? 'gained' : 'lost'}`;
592
+ return (`${kind} ${id}: enumerated by ${listerLabel} (${listedAs}) and not by ${silent} — the wire lists no rule inventory (rule_coverage.exercised and recall_only are counts; only unexercised and blocking-fired ids are named), ` +
593
+ `so ${silent}'s silence is not absence: ${id} may be ${maybe.join(', ')} or ${change}: not asserted`);
594
+ }
595
+ function field(v) {
596
+ return v === 'caught' ? 'caught' : v === 'gap' ? 'gaps' : 'false_positives';
597
+ }
598
+ function tally(results) {
599
+ const t = { total: results.length, caught: 0, gaps: 0, false_positives: 0 };
600
+ for (const r of results)
601
+ t[field(r.verdict)] += 1;
602
+ return t;
603
+ }
604
+ function sameSummary(x, y) {
605
+ return x.total === y.total && x.caught === y.caught && x.gaps === y.gaps && x.false_positives === y.false_positives;
606
+ }
607
+ function fmt(s) {
608
+ return `{total ${s.total}, caught ${s.caught}, gaps ${s.gaps}, false_positives ${s.false_positives}}`;
609
+ }
610
+ //# sourceMappingURL=eval-compare.js.map