wicked-crew 0.7.25 → 0.7.26
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/api/endpoint-manifest-live.d.ts.map +1 -1
- package/dist/api/endpoint-manifest-live.js +5 -0
- package/dist/api/endpoint-manifest-live.js.map +1 -1
- package/dist/api/eval-compare.d.ts +306 -0
- package/dist/api/eval-compare.d.ts.map +1 -0
- package/dist/api/eval-compare.js +610 -0
- package/dist/api/eval-compare.js.map +1 -0
- package/dist/api/eval-sample.d.ts +158 -0
- package/dist/api/eval-sample.d.ts.map +1 -0
- package/dist/api/eval-sample.js +88 -0
- package/dist/api/eval-sample.js.map +1 -0
- package/dist/api/routes.d.ts +5 -0
- package/dist/api/routes.d.ts.map +1 -1
- package/dist/api/routes.js +31 -0
- package/dist/api/routes.js.map +1 -1
- package/dist/api/run-files.d.ts +12 -2
- package/dist/api/run-files.d.ts.map +1 -1
- package/dist/api/run-files.js +17 -5
- package/dist/api/run-files.js.map +1 -1
- package/dist/api/server.d.ts +20 -0
- package/dist/api/server.d.ts.map +1 -1
- package/dist/api/server.js +62 -9
- package/dist/api/server.js.map +1 -1
- package/dist/api/skills.d.ts +88 -0
- package/dist/api/skills.d.ts.map +1 -0
- package/dist/api/skills.js +264 -0
- package/dist/api/skills.js.map +1 -0
- package/dist/api/testing.d.ts +2 -75
- package/dist/api/testing.d.ts.map +1 -1
- package/dist/api/testing.js +14 -28
- package/dist/api/testing.js.map +1 -1
- package/dist/core/adapter.d.ts +42 -0
- package/dist/core/adapter.d.ts.map +1 -1
- package/dist/core/adapter.js +83 -12
- package/dist/core/adapter.js.map +1 -1
- package/dist/interactive/bridge-root.d.ts +117 -7
- package/dist/interactive/bridge-root.d.ts.map +1 -1
- package/dist/interactive/bridge-root.js +229 -9
- package/dist/interactive/bridge-root.js.map +1 -1
- package/dist/interactive/doc-delete-routes.d.ts +2 -0
- package/dist/interactive/doc-delete-routes.d.ts.map +1 -1
- package/dist/interactive/doc-delete-routes.js +3 -19
- package/dist/interactive/doc-delete-routes.js.map +1 -1
- package/dist/interactive/doc-list-routes.d.ts +42 -0
- package/dist/interactive/doc-list-routes.d.ts.map +1 -0
- package/dist/interactive/doc-list-routes.js +165 -0
- package/dist/interactive/doc-list-routes.js.map +1 -0
- package/dist/interactive/project-root.d.ts +36 -0
- package/dist/interactive/project-root.d.ts.map +1 -0
- package/dist/interactive/project-root.js +45 -0
- package/dist/interactive/project-root.js.map +1 -0
- package/dist/interactive/proxy-routes.d.ts +4 -1
- package/dist/interactive/proxy-routes.d.ts.map +1 -1
- package/dist/interactive/proxy-routes.js +4 -22
- package/dist/interactive/proxy-routes.js.map +1 -1
- package/dist/projects/default-project.d.ts +10 -0
- package/dist/projects/default-project.d.ts.map +1 -0
- package/dist/projects/default-project.js +10 -0
- package/dist/projects/default-project.js.map +1 -0
- package/dist/projects/routes.d.ts +0 -2
- package/dist/projects/routes.d.ts.map +1 -1
- package/dist/projects/routes.js +1 -2
- package/dist/projects/routes.js.map +1 -1
- package/dist/skills/bundle.d.ts +94 -0
- package/dist/skills/bundle.d.ts.map +1 -0
- package/dist/skills/bundle.js +187 -0
- package/dist/skills/bundle.js.map +1 -0
- package/dist/skills/contain.d.ts +57 -0
- package/dist/skills/contain.d.ts.map +1 -0
- package/dist/skills/contain.js +90 -0
- package/dist/skills/contain.js.map +1 -0
- package/dist/skills/core-closure.d.ts +83 -0
- package/dist/skills/core-closure.d.ts.map +1 -0
- package/dist/skills/core-closure.js +130 -0
- package/dist/skills/core-closure.js.map +1 -0
- package/dist/skills/engine-env.d.ts +48 -0
- package/dist/skills/engine-env.d.ts.map +1 -0
- package/dist/skills/engine-env.js +69 -0
- package/dist/skills/engine-env.js.map +1 -0
- package/dist/skills/frontmatter.d.ts +77 -0
- package/dist/skills/frontmatter.d.ts.map +1 -0
- package/dist/skills/frontmatter.js +190 -0
- package/dist/skills/frontmatter.js.map +1 -0
- package/dist/skills/guards.d.ts +72 -0
- package/dist/skills/guards.d.ts.map +1 -0
- package/dist/skills/guards.js +147 -0
- package/dist/skills/guards.js.map +1 -0
- package/dist/skills/live-generations.d.ts +80 -0
- package/dist/skills/live-generations.d.ts.map +1 -0
- package/dist/skills/live-generations.js +163 -0
- package/dist/skills/live-generations.js.map +1 -0
- package/dist/skills/plugin-source.d.ts +109 -0
- package/dist/skills/plugin-source.d.ts.map +1 -0
- package/dist/skills/plugin-source.js +272 -0
- package/dist/skills/plugin-source.js.map +1 -0
- package/dist/skills/refs.d.ts +58 -0
- package/dist/skills/refs.d.ts.map +1 -0
- package/dist/skills/refs.js +133 -0
- package/dist/skills/refs.js.map +1 -0
- package/dist/skills/root-fence.d.ts +53 -0
- package/dist/skills/root-fence.d.ts.map +1 -0
- package/dist/skills/root-fence.js +109 -0
- package/dist/skills/root-fence.js.map +1 -0
- package/dist/skills/root-names.d.ts +42 -0
- package/dist/skills/root-names.d.ts.map +1 -0
- package/dist/skills/root-names.js +44 -0
- package/dist/skills/root-names.js.map +1 -0
- package/dist/skills/runtime.d.ts +130 -0
- package/dist/skills/runtime.d.ts.map +1 -0
- package/dist/skills/runtime.js +246 -0
- package/dist/skills/runtime.js.map +1 -0
- package/dist/skills/store.d.ts +895 -0
- package/dist/skills/store.d.ts.map +1 -0
- package/dist/skills/store.js +3575 -0
- package/dist/skills/store.js.map +1 -0
- package/dist/skills/tree.d.ts +292 -0
- package/dist/skills/tree.d.ts.map +1 -0
- package/dist/skills/tree.js +645 -0
- package/dist/skills/tree.js.map +1 -0
- package/dist/skills/venv.d.ts +52 -0
- package/dist/skills/venv.d.ts.map +1 -0
- package/dist/skills/venv.js +80 -0
- package/dist/skills/venv.js.map +1 -0
- package/dist/studio/assets/index-CK1uJWs2.css +32 -0
- package/dist/studio/assets/index-FcZg3WDR.js +539 -0
- package/dist/studio/index.html +2 -2
- package/dist/studio/testid-inventory.json +605 -7
- package/endpoint-manifest.json +183 -1
- package/package.json +4 -3
- package/dist/studio/assets/index-C4iz4uAw.js +0 -537
- package/dist/studio/assets/index-DEEQcRQ_.css +0 -32
|
@@ -0,0 +1,610 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Cross-run eval comparison (docs/testing/evals-test-plan.md S17) — the OFFLINE, deterministic diff
|
|
3
|
+
* of two recorded {@link EvalRunDetail}s over the same corpus: which per-sample verdicts flipped,
|
|
4
|
+
* whether each flip is the kind a tightened store is allowed to produce (`permitted`) or the kind a
|
|
5
|
+
* release has to explain (`flagged`), whether the two stored summaries reconcile to the per-sample
|
|
6
|
+
* accounting, and which rules gained or lost exercise (core #394 `rule_coverage`).
|
|
7
|
+
*
|
|
8
|
+
* Pure over the two records: no store, no engine, no clock, no I/O — two persisted drilldowns in,
|
|
9
|
+
* one report out, byte-stable for the same inputs (every list is codepoint-sorted by id).
|
|
10
|
+
*
|
|
11
|
+
* # What "comparable" means here
|
|
12
|
+
*
|
|
13
|
+
* Plan §3 keys a comparison by five identities. The records carry four of them today (`corpus`,
|
|
14
|
+
* `rule_store`, `type_filter`, `degraded`; the rule snapshot and the engine build are §5 follow-ups),
|
|
15
|
+
* so `comparable` is derived from CONTENT, never from a name alone: the same corpus name AND the same
|
|
16
|
+
* type filter AND the same sample ids AND, per id, the same `kind` AND the same sample PAYLOAD
|
|
17
|
+
* identity — the `sample.payload_hash` a producer stamped on each result row (`eval-sample.js`
|
|
18
|
+
* `samplePayloadHash`: sha256 over the canonical JSON of id, description, kind, steering_type,
|
|
19
|
+
* signals). The engine echoes only id/description/kind/steering_type per row, never the input
|
|
20
|
+
* `signals`, so a row WITHOUT a payload hash cannot be proven to be the same action as its
|
|
21
|
+
* namesake in the other run: such a pair is `comparable: false` with `comparable_reason` starting
|
|
22
|
+
* `unverified: no sample identity` — unverified, which is not the same as "different". A sample
|
|
23
|
+
* present on one side only, one whose kind changed, or one whose payload hash changed means the
|
|
24
|
+
* corpus changed under the same name — reported, and the pair is not comparable release over
|
|
25
|
+
* release. Two runs under DIFFERENT type filters judged different slices of the corpus over
|
|
26
|
+
* different coverage denominators — not comparable either (`differing-type-filter`, see below).
|
|
27
|
+
* `comparable_reason` is `null` exactly when `comparable` is true.
|
|
28
|
+
*
|
|
29
|
+
* # Flip classification (the S17 rule table)
|
|
30
|
+
*
|
|
31
|
+
* bad gap → caught permitted a rule tightened: a bad behavior is now caught
|
|
32
|
+
* bad caught → gap flagged regression: a bad behavior is no longer caught
|
|
33
|
+
* good false_positive → caught permitted a good sample is no longer denied
|
|
34
|
+
* good caught → false_positive flagged a good sample is newly denied
|
|
35
|
+
* anything else flagged not a verdict pair that kind can take (evals.rs
|
|
36
|
+
* `evaluate_sample`: a bad sample is caught|gap, a
|
|
37
|
+
* good sample is caught|false_positive)
|
|
38
|
+
*
|
|
39
|
+
* `good` is a sample KIND, not a verdict — the draft plan's `good → false_positive` spelling was
|
|
40
|
+
* corrected in revision 3; this module is the executable form of that correction.
|
|
41
|
+
*
|
|
42
|
+
* # Coverage transitions vs. rule-set changes — what a record CANNOT tell (codex rounds 6 and 8)
|
|
43
|
+
*
|
|
44
|
+
* The wire carries `rule_coverage.exercised` as a COUNT and `unexercised` as a LIST of ids
|
|
45
|
+
* (api-types `GovernanceEvalRuleCoverage`; wicked-core `crates/wicked-governance/src/evals.rs` on
|
|
46
|
+
* branch `feat/evals-effect-and-coverage` (core #394/#395, PR #398) at `a87e461` — `RuleCoverage`
|
|
47
|
+
* lines 323-328: `exercised: usize`, `unexercised: Vec<UnexercisedRule>`, `recall_only: usize`,
|
|
48
|
+
* `per_type: BTreeMap<String, TypeCoverage>`). The partition is over the ELIGIBLE rules — the
|
|
49
|
+
* active, effect-bearing rules of the slice (`decide_lane_rules` lines 816-843: `effect.is_some()
|
|
50
|
+
* && !retired && in_slice`); `recall_only` counts the effect-less active rules OUTSIDE it (docs
|
|
51
|
+
* lines 317-319, computed at 875-878). A rule is EXERCISED when it appeared in ANY evaluated claim's
|
|
52
|
+
* `policy_ids`, whatever its effect (`RuleCoverage` docs lines 312-314; `rule_coverage()` lines
|
|
53
|
+
* 849-880 partitions the eligible rules by `triggered`, the union `run_evals` collects of every
|
|
54
|
+
* claim's `policy_ids`). A result row's `fired`, however, is the BLOCKING subset only:
|
|
55
|
+
* `evaluate_sample` lines 906-916 keep the ids whose effect is `Deny`. Hence, for an UNFILTERED run,
|
|
56
|
+
*
|
|
57
|
+
* fired(run) ⊆ exercised(run), and `exercised − |fired|` rules were exercised by a non-blocking
|
|
58
|
+
* (`warn`) effect ALONE — counted, but NEVER named anywhere in the record.
|
|
59
|
+
*
|
|
60
|
+
* NO field of the wire lists a run's rule set: `exercised` and `recall_only` are counts,
|
|
61
|
+
* `unexercised` names only the rules nothing exercised, `fired` only the blocking firings. The rule
|
|
62
|
+
* identities a record enumerates are therefore exactly `unexercised ∪ fired`, and an id one record
|
|
63
|
+
* enumerates and the other does not is NOT thereby absent from the other's store: it may be one of
|
|
64
|
+
* that side's unnamed warn-exercised rules, an effect-less (recall-only) or retired rule outside the
|
|
65
|
+
* eligible partition, under a type filter a rule of another type outside the denominator — or a rule
|
|
66
|
+
* the store really gained or lost. Round 6 replaced a reconstruction of "the inventory" as
|
|
67
|
+
* `unexercised ∪ fired` with a completeness inference: `exercised === |fired|` ⇒ every exercised rule
|
|
68
|
+
* is named ⇒ the silent side's absence was asserted as `added_rules`/`removed_rules`. Round 8 deleted
|
|
69
|
+
* that inference too — under a type filter a fired id is typed into the slice only by the OTHER run's
|
|
70
|
+
* `unexercised` row, and a rule's type in run A does not establish its type in run B (codex's
|
|
71
|
+
* reproduction: A lists R and Q unexercised under `development`; B moves R to `security`, fires R
|
|
72
|
+
* blocking and exercises Q by `warn`; the cross-run intersection named R, declared B complete and
|
|
73
|
+
* reported Q — still present and exercised — as `removed_rules`). The delta asserts ONLY what the
|
|
74
|
+
* records state:
|
|
75
|
+
*
|
|
76
|
+
* gained unexercised in A (listed) AND blocking-fired in B (listed): a sample now exercises
|
|
77
|
+
* the rule — a statement about the RULE, both ends named, certain
|
|
78
|
+
* lost blocking-fired in A AND unexercised in B (listed): certain
|
|
79
|
+
* unidentified PER RECORD, how many exercised rules of its denominator it does not name:
|
|
80
|
+
* unfiltered `exercised − |fired|`; under a type filter the whole `exercised` (a
|
|
81
|
+
* fired id carries no steering_type, so a record types none of its own firings into
|
|
82
|
+
* the slice) — never reduced by what the OTHER run lists
|
|
83
|
+
* added_rules / asserted only from an explicit rule inventory on BOTH records — the wire carries
|
|
84
|
+
* removed_rules none, so both are EMPTY for every daemon-recorded run and `inventory` is `partial`
|
|
85
|
+
* transitions_withheld
|
|
86
|
+
* every id enumerated by one record and not the other, one reason per id saying what
|
|
87
|
+
* it may be besides a rule-set change — WITHHELD, never guessed
|
|
88
|
+
*
|
|
89
|
+
* `exercised_delta` is always the difference of the two counts. Reconciliation errors on coverage are
|
|
90
|
+
* the record contradicting ITSELF (or being malformed — below); two valid reports always `reconcile`.
|
|
91
|
+
*
|
|
92
|
+
* # Type filters (codex round 7) — the denominator is the SLICE; the firings are not
|
|
93
|
+
*
|
|
94
|
+
* A run recorded under `type_filter: T` was produced by `rules eval --type T`. In the engine that
|
|
95
|
+
* filter slices the SAMPLES (`run_evals` lines 974-977: only samples of type T are judged) and the
|
|
96
|
+
* COVERAGE DENOMINATOR (`decide_lane_rules` lines 816-844, `in_slice` at 820: only type-T rules are
|
|
97
|
+
* eligible; `rule_coverage` 849-880 partitions exactly those; `recall_only` 875-878 is sliced too)
|
|
98
|
+
* but NOT the gate: `evaluate_sample` runs `select_any` over every active rule whatever its type
|
|
99
|
+
* (line 901) and a row's `fired` keeps every blocking id it produced (906-916). So under a filter a
|
|
100
|
+
* row may fire a rule OUTSIDE the denominator — codex's reproduction: a development-filtered run
|
|
101
|
+
* with `fired: ["SECURITY-DENY"]` and `exercised: 0` is a VALID record (the security rule is not a
|
|
102
|
+
* development rule), and the unfiltered invariant `exercised ≥ |fired|` does not hold. Hence, per
|
|
103
|
+
* record ({@link EvalCoverageReconciliation}, reported per side in `coverage_reconciliation`):
|
|
104
|
+
*
|
|
105
|
+
* type_filter null `'rows'` — every active rule is in the denominator, so the rows'
|
|
106
|
+
* blocking-fired ids are exercised rules: `exercised ≥ |fired|`, and
|
|
107
|
+
* the exercised set is complete iff `exercised === |fired|`.
|
|
108
|
+
* type_filter T, per_type `'per_type'` — reconciled against the engine's OWN row for the slice
|
|
109
|
+
* present (`rule_coverage.per_type[T]`, `TypeCoverage` lines 300-303):
|
|
110
|
+
* `per_type[T].exercised === exercised`; the rows' fired ids carry no
|
|
111
|
+
* steering type and are NOT a denominator check.
|
|
112
|
+
* type_filter T, no `'n/a (engine reports no per-type coverage)'` — the record carries
|
|
113
|
+
* per_type nothing to check its slice against; nothing fired-based is asserted.
|
|
114
|
+
*
|
|
115
|
+
* Whatever the filter: `per_type`, when present, must sum to `exercised` and agree per type with the
|
|
116
|
+
* listed `unexercised` rows; under a filter every listed row is of the filter's type; no id is
|
|
117
|
+
* listed twice; and no listed-unexercised id fired — that row TYPES the rule into the slice, and a
|
|
118
|
+
* blocking firing is a firing, which makes an eligible rule exercised (862-873) — so this last check
|
|
119
|
+
* is sound under a filter too (the fix brief's floor was "unfiltered only"; the engine's definition
|
|
120
|
+
* makes it general).
|
|
121
|
+
*
|
|
122
|
+
* Two runs under DIFFERENT filters get NO `rule_coverage_delta` (an inventory diff across
|
|
123
|
+
* denominators would report the other slice's rules as changes) and are not comparable. Under the
|
|
124
|
+
* SAME filter T the delta is computed exactly as unfiltered — `gained`/`lost` from ids listed on both
|
|
125
|
+
* ends (the unexercised row carries `steering_type: T`, the firing is a firing), every one-sided id
|
|
126
|
+
* withheld — except that `unidentified` is the whole `exercised` count (see above) and each withheld
|
|
127
|
+
* reason adds that the id may be a rule of another type outside the denominator.
|
|
128
|
+
*
|
|
129
|
+
* # Malformed persisted coverage (codex round 8)
|
|
130
|
+
*
|
|
131
|
+
* The daemon persists a run's `rule_coverage` verbatim and validates none of it (the script's
|
|
132
|
+
* `verifyEngineReport` gates only its own reports). A recorded `rule_coverage` that is not the wire
|
|
133
|
+
* shape — `null`; `unexercised` not an array; `per_type: null`, a row that is not `{ exercised,
|
|
134
|
+
* unexercised }` of non-negative integers, a key that is no steering type … — cannot be reconciled:
|
|
135
|
+
* that side is `coverage_reconciliation: 'unverified (malformed rule_coverage)'`
|
|
136
|
+
* ({@link MALFORMED_RULE_COVERAGE}), a reconciliation error names the run and the defect, no
|
|
137
|
+
* `rule_coverage_delta` is computed, and the comparison never throws over persisted data.
|
|
138
|
+
*
|
|
139
|
+
* # Malformed persisted result rows (Copilot on #475)
|
|
140
|
+
*
|
|
141
|
+
* The same store validates a detail's `results` only as "an array" (`EvalRunStore.get()`), so a row
|
|
142
|
+
* that is not the wire shape (`GovernanceEvalResult` — no `sample`, `sample.id` not a non-empty
|
|
143
|
+
* string, a `kind` or `verdict` outside its union, `fired` not an array of strings, a row that is no
|
|
144
|
+
* object at all) can reach the comparison from a corrupted or hand-edited file. Every row is checked
|
|
145
|
+
* ({@link resultRowProblem}) BEFORE it is indexed, tallied or iterated: a malformed one is ONE
|
|
146
|
+
* reconciliation error naming the run, the index and the defect, and is EXCLUDED from everything —
|
|
147
|
+
* the id index, the identity counts, the tally, the coverage `fired` set. The pair is then not
|
|
148
|
+
* comparable ({@link MALFORMED_RESULT_ROWS}), and that side's stored summary is NOT checked against a
|
|
149
|
+
* results list the comparison could not fully read (the shortfall is the excluded rows, not a summary
|
|
150
|
+
* defect — reporting it as one would misattribute it). Never a throw.
|
|
151
|
+
*/
|
|
152
|
+
import { PAYLOAD_HASH_RE } from './eval-sample.js';
|
|
153
|
+
import { STEERING_TYPE_VALUES, STEERING_TYPES } from './governance-steering.js';
|
|
154
|
+
/** The exact prefix of `comparable_reason` when a side carries result rows without a payload hash. */
|
|
155
|
+
export const UNVERIFIED_NO_SAMPLE_IDENTITY = 'unverified: no sample identity';
|
|
156
|
+
/** The exact prefix of `comparable_reason` when the two runs were produced under different type filters. */
|
|
157
|
+
export const DIFFERING_TYPE_FILTER = 'differing-type-filter';
|
|
158
|
+
/** The exact `coverage_reconciliation` value of a side whose persisted `rule_coverage` is not the wire
|
|
159
|
+
* shape — unverified, not reconciled, no delta computed over it (module doc, "Malformed persisted
|
|
160
|
+
* coverage"). */
|
|
161
|
+
export const MALFORMED_RULE_COVERAGE = 'unverified (malformed rule_coverage)';
|
|
162
|
+
/** The `comparable_reason` prefix of a pair one side of which persisted result rows that are not the
|
|
163
|
+
* wire shape (api-types `GovernanceEvalResult`, module doc "Malformed persisted result rows"): each
|
|
164
|
+
* is named in `reconciliation_errors` and EXCLUDED, never thrown over — and a comparison over a
|
|
165
|
+
* record it could not fully read asserts nothing (Copilot on #475). */
|
|
166
|
+
export const MALFORMED_RESULT_ROWS = 'unverified: malformed result row(s)';
|
|
167
|
+
/** The S17 rule table, as a function. Exported so a caller can classify a single flip. */
|
|
168
|
+
export function classifyFlip(kind, from, to) {
|
|
169
|
+
if (kind === 'bad') {
|
|
170
|
+
if (from === 'gap' && to === 'caught')
|
|
171
|
+
return { classification: 'permitted', reason: 'a rule tightened: a bad behavior is now caught' };
|
|
172
|
+
if (from === 'caught' && to === 'gap')
|
|
173
|
+
return { classification: 'flagged', reason: 'regression: a bad behavior is no longer caught' };
|
|
174
|
+
}
|
|
175
|
+
else {
|
|
176
|
+
if (from === 'false_positive' && to === 'caught')
|
|
177
|
+
return { classification: 'permitted', reason: 'a good sample is no longer denied' };
|
|
178
|
+
if (from === 'caught' && to === 'false_positive')
|
|
179
|
+
return { classification: 'flagged', reason: 'a good sample is newly denied' };
|
|
180
|
+
}
|
|
181
|
+
return {
|
|
182
|
+
classification: 'flagged',
|
|
183
|
+
reason: `inconsistent: ${from} → ${to} is not a verdict pair a ${kind} sample can take (a bad sample is caught|gap, a good sample is caught|false_positive)`,
|
|
184
|
+
};
|
|
185
|
+
}
|
|
186
|
+
/** Compare two recorded runs — A is the baseline, B the candidate. See the module doc. */
|
|
187
|
+
export function compareEvalRuns(a, b) {
|
|
188
|
+
const errors = [];
|
|
189
|
+
// Only WELL-FORMED rows take part (module doc, "Malformed persisted result rows"): the store
|
|
190
|
+
// validated `results` as an array and nothing more, so every row is checked here BEFORE it is
|
|
191
|
+
// indexed, tallied or iterated — a malformed one is named in `errors` and excluded, never thrown over.
|
|
192
|
+
const rowsA = wellFormedRows(a, 'a', errors);
|
|
193
|
+
const rowsB = wellFormedRows(b, 'b', errors);
|
|
194
|
+
const malformed_rows = { a: a.results.length - rowsA.length, b: b.results.length - rowsB.length };
|
|
195
|
+
const mapA = indexById(rowsA, a, 'a', errors);
|
|
196
|
+
const mapB = indexById(rowsB, b, 'b', errors);
|
|
197
|
+
const only_in_a = [...mapA.keys()].filter((id) => !mapB.has(id)).sort(codepoint);
|
|
198
|
+
const only_in_b = [...mapB.keys()].filter((id) => !mapA.has(id)).sort(codepoint);
|
|
199
|
+
const shared = [...mapA.keys()].filter((id) => mapB.has(id)).sort(codepoint);
|
|
200
|
+
const flips = [];
|
|
201
|
+
const kind_changed = [];
|
|
202
|
+
const payload_changed = [];
|
|
203
|
+
let unchanged = 0;
|
|
204
|
+
for (const id of shared) {
|
|
205
|
+
const ra = mapA.get(id);
|
|
206
|
+
const rb = mapB.get(id);
|
|
207
|
+
const kindChanged = ra.sample.kind !== rb.sample.kind;
|
|
208
|
+
const ha = identityOf(ra);
|
|
209
|
+
const hb = identityOf(rb);
|
|
210
|
+
const payloadChanged = ha !== null && hb !== null && ha !== hb;
|
|
211
|
+
if (kindChanged)
|
|
212
|
+
kind_changed.push(id);
|
|
213
|
+
if (payloadChanged)
|
|
214
|
+
payload_changed.push(id);
|
|
215
|
+
if (ra.verdict !== rb.verdict) {
|
|
216
|
+
let verdictClass;
|
|
217
|
+
if (kindChanged) {
|
|
218
|
+
verdictClass = {
|
|
219
|
+
classification: 'flagged',
|
|
220
|
+
reason: `the sample's kind changed between runs (${ra.sample.kind} → ${rb.sample.kind}): the corpus changed under this name`,
|
|
221
|
+
};
|
|
222
|
+
}
|
|
223
|
+
else if (payloadChanged) {
|
|
224
|
+
verdictClass = {
|
|
225
|
+
classification: 'flagged',
|
|
226
|
+
reason: "the sample's payload changed between runs (description, steering_type or signals): the corpus changed under this name",
|
|
227
|
+
};
|
|
228
|
+
}
|
|
229
|
+
else {
|
|
230
|
+
verdictClass = classifyFlip(rb.sample.kind, ra.verdict, rb.verdict);
|
|
231
|
+
}
|
|
232
|
+
flips.push({ sample_id: id, kind: rb.sample.kind, from: ra.verdict, to: rb.verdict, ...verdictClass });
|
|
233
|
+
}
|
|
234
|
+
else if (!kindChanged && !payloadChanged) {
|
|
235
|
+
unchanged += 1;
|
|
236
|
+
}
|
|
237
|
+
}
|
|
238
|
+
const unverified_rows = {
|
|
239
|
+
a: rowsA.filter((r) => identityOf(r) === null).length,
|
|
240
|
+
b: rowsB.filter((r) => identityOf(r) === null).length,
|
|
241
|
+
};
|
|
242
|
+
// Each run's summary must be its own results' tally — a disagreement is a stored defect, and a
|
|
243
|
+
// delta over a defective summary would attribute it to a flip.
|
|
244
|
+
// A side whose results could not be FULLY read is not checked here: its tally over the well-formed
|
|
245
|
+
// rows is short by exactly the excluded (already named) rows, and reporting that shortfall as a
|
|
246
|
+
// summary defect would misattribute it. The record is already `reconciles: false`.
|
|
247
|
+
for (const [run, rows, label] of [
|
|
248
|
+
[a, rowsA, 'a'],
|
|
249
|
+
[b, rowsB, 'b'],
|
|
250
|
+
]) {
|
|
251
|
+
if (rows.length !== run.results.length)
|
|
252
|
+
continue;
|
|
253
|
+
const own = tally(rows);
|
|
254
|
+
if (!sameSummary(own, run.summary)) {
|
|
255
|
+
errors.push(`run ${label} (${run.id}): stored summary ${fmt(run.summary)} does not match its own results ${fmt(own)}`);
|
|
256
|
+
}
|
|
257
|
+
}
|
|
258
|
+
const summary_delta = {
|
|
259
|
+
total: b.summary.total - a.summary.total,
|
|
260
|
+
caught: b.summary.caught - a.summary.caught,
|
|
261
|
+
gaps: b.summary.gaps - a.summary.gaps,
|
|
262
|
+
false_positives: b.summary.false_positives - a.summary.false_positives,
|
|
263
|
+
};
|
|
264
|
+
// The per-sample accounting the delta must equal: every flip moves one count from `from` to
|
|
265
|
+
// `to`; a sample only in B adds its verdict, one only in A removes it.
|
|
266
|
+
const expected = { total: only_in_b.length - only_in_a.length, caught: 0, gaps: 0, false_positives: 0 };
|
|
267
|
+
for (const f of flips) {
|
|
268
|
+
expected[field(f.to)] += 1;
|
|
269
|
+
expected[field(f.from)] -= 1;
|
|
270
|
+
}
|
|
271
|
+
for (const id of only_in_b)
|
|
272
|
+
expected[field(mapB.get(id).verdict)] += 1;
|
|
273
|
+
for (const id of only_in_a)
|
|
274
|
+
expected[field(mapA.get(id).verdict)] -= 1;
|
|
275
|
+
// Withheld, for the same reason, when either side carries excluded rows: the stored summaries count
|
|
276
|
+
// rows the per-sample accounting could not read.
|
|
277
|
+
if (malformed_rows.a === 0 && malformed_rows.b === 0 && !sameSummary(expected, summary_delta)) {
|
|
278
|
+
errors.push(`summary delta ${fmt(summary_delta)} does not reconcile to the per-sample accounting ${fmt(expected)}`);
|
|
279
|
+
}
|
|
280
|
+
const identity = {
|
|
281
|
+
corpus: pair(a.corpus, b.corpus),
|
|
282
|
+
rule_store: pair(a.rule_store, b.rule_store),
|
|
283
|
+
type_filter: pair(a.type_filter, b.type_filter),
|
|
284
|
+
degraded: pair(a.degraded, b.degraded),
|
|
285
|
+
};
|
|
286
|
+
// Why the pair is not comparable, in verification order: a side with excluded (malformed) rows,
|
|
287
|
+
// then an unverified side, first (nothing below can be asserted about actions whose identity is
|
|
288
|
+
// unknown), then the identity and content differences.
|
|
289
|
+
const reasons = [];
|
|
290
|
+
if (malformed_rows.a > 0 || malformed_rows.b > 0) {
|
|
291
|
+
reasons.push(`${MALFORMED_RESULT_ROWS} (${malformed_rows.a} result row(s) in a and ${malformed_rows.b} in b are not the wire shape — each named in reconciliation_errors and excluded)`);
|
|
292
|
+
}
|
|
293
|
+
if (unverified_rows.a > 0 || unverified_rows.b > 0) {
|
|
294
|
+
reasons.push(`${UNVERIFIED_NO_SAMPLE_IDENTITY} (${unverified_rows.a} result row(s) in a and ${unverified_rows.b} in b carry no well-formed sample.payload_hash)`);
|
|
295
|
+
}
|
|
296
|
+
if (!identity.corpus.same)
|
|
297
|
+
reasons.push(`corpus name differs (${JSON.stringify(a.corpus)} vs ${JSON.stringify(b.corpus)})`);
|
|
298
|
+
if (!identity.type_filter.same) {
|
|
299
|
+
reasons.push(`${DIFFERING_TYPE_FILTER} (a: ${JSON.stringify(a.type_filter)}, b: ${JSON.stringify(b.type_filter)}): the runs judged different slices of the corpus over different coverage denominators`);
|
|
300
|
+
}
|
|
301
|
+
if (only_in_a.length > 0 || only_in_b.length > 0) {
|
|
302
|
+
reasons.push(`sample set differs (${only_in_a.length} id(s) only in a, ${only_in_b.length} only in b)`);
|
|
303
|
+
}
|
|
304
|
+
if (kind_changed.length > 0)
|
|
305
|
+
reasons.push(`kind changed for ${kind_changed.length} shared sample(s)`);
|
|
306
|
+
if (payload_changed.length > 0)
|
|
307
|
+
reasons.push(`payload changed for ${payload_changed.length} shared sample(s) (description, steering_type or signals)`);
|
|
308
|
+
// Each record's coverage is reconciled ON ITS OWN whenever it carries one — a record that
|
|
309
|
+
// contradicts itself is a defect whether or not the other side measured coverage. Coverage is
|
|
310
|
+
// OPTIONAL on a record (a pre-#394 engine emits none: `undefined` here); a PRESENT value that is
|
|
311
|
+
// not the wire shape is named as a defect and never thrown over (`null` here).
|
|
312
|
+
const covA = a.rule_coverage === undefined ? undefined : reconcileCoverage(a, rowsA, 'a', errors);
|
|
313
|
+
const covB = b.rule_coverage === undefined ? undefined : reconcileCoverage(b, rowsB, 'b', errors);
|
|
314
|
+
const modeOf = (cov) => (cov === undefined ? null : cov === null ? MALFORMED_RULE_COVERAGE : cov.mode);
|
|
315
|
+
const comparison = {
|
|
316
|
+
a: a.id,
|
|
317
|
+
b: b.id,
|
|
318
|
+
identity,
|
|
319
|
+
comparable: reasons.length === 0,
|
|
320
|
+
comparable_reason: reasons.length === 0 ? null : reasons.join('; '),
|
|
321
|
+
flips,
|
|
322
|
+
unchanged,
|
|
323
|
+
only_in_a,
|
|
324
|
+
only_in_b,
|
|
325
|
+
kind_changed,
|
|
326
|
+
payload_changed,
|
|
327
|
+
unverified_rows,
|
|
328
|
+
summary_delta,
|
|
329
|
+
reconciles: true,
|
|
330
|
+
reconciliation_errors: errors,
|
|
331
|
+
coverage_reconciliation: { a: modeOf(covA), b: modeOf(covB) },
|
|
332
|
+
};
|
|
333
|
+
// The delta exists only when both sides measured WELL-FORMED coverage over the SAME denominator —
|
|
334
|
+
// never fabricated from one side, never over a malformed record, never across two different slices.
|
|
335
|
+
if (covA && covB && identity.type_filter.same) {
|
|
336
|
+
comparison.rule_coverage_delta = coverageDelta(covA, covB, a.type_filter);
|
|
337
|
+
}
|
|
338
|
+
comparison.reconciles = errors.length === 0;
|
|
339
|
+
return comparison;
|
|
340
|
+
}
|
|
341
|
+
// ── helpers ──────────────────────────────────────────────────────────────────────────────────────
|
|
342
|
+
/** Codepoint order — deterministic on every machine (`localeCompare` is locale-bound). */
|
|
343
|
+
function codepoint(x, y) {
|
|
344
|
+
return x < y ? -1 : x > y ? 1 : 0;
|
|
345
|
+
}
|
|
346
|
+
function pair(a, b) {
|
|
347
|
+
return { a, b, same: a === b };
|
|
348
|
+
}
|
|
349
|
+
/** The row's stamped payload identity — only when it is WELL-FORMED (`PAYLOAD_HASH_RE`: `sha256:`
|
|
350
|
+
* + 64 lowercase hex, the one spelling `samplePayloadHash` produces). Anything else — none, an
|
|
351
|
+
* empty string, another algorithm, uppercase, a truncated digest — is null: the row is UNVERIFIED,
|
|
352
|
+
* never treated as an authoritative identity that could mark a pair comparable or "changed"
|
|
353
|
+
* (the value comes from persisted run data; Copilot on #475). */
|
|
354
|
+
function identityOf(r) {
|
|
355
|
+
const h = r.sample.payload_hash;
|
|
356
|
+
return typeof h === 'string' && PAYLOAD_HASH_RE.test(h) ? h : null;
|
|
357
|
+
}
|
|
358
|
+
const RESULT_KINDS = new Set(['good', 'bad']);
|
|
359
|
+
const RESULT_VERDICTS = new Set(['caught', 'gap', 'false_positive']);
|
|
360
|
+
/**
|
|
361
|
+
* Why a persisted result row is not the wire shape (api-types `GovernanceEvalResult`) in the fields
|
|
362
|
+
* this comparison READS — `sample.id` (a non-empty string), `sample.kind` (`good|bad`), `verdict`
|
|
363
|
+
* (`caught|gap|false_positive`), `fired` (an array of rule ids) — or null when it is. Checked BEFORE
|
|
364
|
+
* any row is indexed, tallied or iterated: `EvalRunStore.get()` validates only that `results` is an
|
|
365
|
+
* array, so a corrupted or hand-edited detail file can carry a row without a `sample`, with
|
|
366
|
+
* `fired: null`, or with a verdict `field()` maps nowhere (Copilot on #475: `identityOf` / `indexById`
|
|
367
|
+
* threw, `tally` would count into an undefined field). `payload_hash` is not checked here —
|
|
368
|
+
* `identityOf` already treats anything malformed as no identity.
|
|
369
|
+
*/
|
|
370
|
+
function resultRowProblem(r) {
|
|
371
|
+
if (!isPlainObject(r))
|
|
372
|
+
return `expected an object { sample, verdict, fired }, got ${JSON.stringify(r)}`;
|
|
373
|
+
const sample = r['sample'];
|
|
374
|
+
if (!isPlainObject(sample))
|
|
375
|
+
return `sample ${JSON.stringify(sample)} is not an object { id, kind }`;
|
|
376
|
+
if (typeof sample['id'] !== 'string' || sample['id'] === '')
|
|
377
|
+
return `sample.id ${JSON.stringify(sample['id'])} is not a non-empty string`;
|
|
378
|
+
if (typeof sample['kind'] !== 'string' || !RESULT_KINDS.has(sample['kind']))
|
|
379
|
+
return `sample.kind ${JSON.stringify(sample['kind'])} is not good|bad`;
|
|
380
|
+
if (typeof r['verdict'] !== 'string' || !RESULT_VERDICTS.has(r['verdict']))
|
|
381
|
+
return `verdict ${JSON.stringify(r['verdict'])} is not caught|gap|false_positive`;
|
|
382
|
+
const fired = r['fired'];
|
|
383
|
+
if (!Array.isArray(fired) || !fired.every((id) => typeof id === 'string'))
|
|
384
|
+
return `fired ${JSON.stringify(fired)} is not an array of rule ids`;
|
|
385
|
+
return null;
|
|
386
|
+
}
|
|
387
|
+
/** The run's WELL-FORMED result rows ({@link resultRowProblem}). Every other row is ONE reconciliation
|
|
388
|
+
* error naming the run, the row's index and the defect, and takes no part in the comparison. */
|
|
389
|
+
function wellFormedRows(run, label, errors) {
|
|
390
|
+
const rows = [];
|
|
391
|
+
for (const [i, r] of run.results.entries()) {
|
|
392
|
+
const problem = resultRowProblem(r);
|
|
393
|
+
if (problem === null)
|
|
394
|
+
rows.push(r);
|
|
395
|
+
else {
|
|
396
|
+
errors.push(`run ${label} (${run.id}): results[${i}] is malformed — ${problem} — not the wire shape (api-types GovernanceEvalResult), so it is excluded from the per-sample comparison, the identity counts, the tally and the coverage reconciliation`);
|
|
397
|
+
}
|
|
398
|
+
}
|
|
399
|
+
return rows;
|
|
400
|
+
}
|
|
401
|
+
/** The run's well-formed results by sample id. A duplicate id inside ONE run (the engine rejects them
|
|
402
|
+
* at import — an edited record could still carry one) is a reconciliation error, and the last row wins. */
|
|
403
|
+
function indexById(rows, run, label, errors) {
|
|
404
|
+
const map = new Map();
|
|
405
|
+
for (const r of rows) {
|
|
406
|
+
if (map.has(r.sample.id))
|
|
407
|
+
errors.push(`run ${label} (${run.id}): duplicate sample id ${r.sample.id} in results`);
|
|
408
|
+
map.set(r.sample.id, r);
|
|
409
|
+
}
|
|
410
|
+
return map;
|
|
411
|
+
}
|
|
412
|
+
function isPlainObject(x) {
|
|
413
|
+
return x !== null && typeof x === 'object' && !Array.isArray(x);
|
|
414
|
+
}
|
|
415
|
+
function isCount(x) {
|
|
416
|
+
return Number.isInteger(x) && x >= 0;
|
|
417
|
+
}
|
|
418
|
+
/**
|
|
419
|
+
* Why a persisted `rule_coverage` is not the wire shape (api-types `GovernanceEvalRuleCoverage`
|
|
420
|
+
* 0.27.0), or null when it is — checked BEFORE any field is read, because the store writes the
|
|
421
|
+
* engine's value verbatim and the comparison must name a malformed record, never throw over it
|
|
422
|
+
* (codex round 8: `per_type: null` made `Object.values` throw). The shape: `exercised` a non-negative
|
|
423
|
+
* integer; `unexercised` an array of `{ rule_id: non-empty string, steering_type: one of the seven }`;
|
|
424
|
+
* `recall_only`, when present, a non-negative integer; `per_type`, when present, a plain object whose
|
|
425
|
+
* keys are steering types and whose values are `{ exercised, unexercised }` of non-negative integers.
|
|
426
|
+
*/
|
|
427
|
+
function coverageShapeProblem(rc) {
|
|
428
|
+
if (!isPlainObject(rc))
|
|
429
|
+
return `expected an object { exercised, unexercised[] }, got ${JSON.stringify(rc)}`;
|
|
430
|
+
if (!isCount(rc['exercised']))
|
|
431
|
+
return `exercised ${JSON.stringify(rc['exercised'])} is not a non-negative integer`;
|
|
432
|
+
const unexercised = rc['unexercised'];
|
|
433
|
+
if (!Array.isArray(unexercised))
|
|
434
|
+
return `unexercised ${JSON.stringify(unexercised)} is not an array`;
|
|
435
|
+
for (const [i, u] of unexercised.entries()) {
|
|
436
|
+
if (!isPlainObject(u) || typeof u['rule_id'] !== 'string' || u['rule_id'] === '')
|
|
437
|
+
return `unexercised[${i}] carries no string rule_id (got ${JSON.stringify(u)})`;
|
|
438
|
+
if (typeof u['steering_type'] !== 'string' || !STEERING_TYPES.has(u['steering_type'])) {
|
|
439
|
+
return `unexercised[${i}] (${u['rule_id']}) steering_type ${JSON.stringify(u['steering_type'])} is not one of ${STEERING_TYPE_VALUES.join('|')}`;
|
|
440
|
+
}
|
|
441
|
+
}
|
|
442
|
+
if (rc['recall_only'] !== undefined && !isCount(rc['recall_only']))
|
|
443
|
+
return `recall_only ${JSON.stringify(rc['recall_only'])} is not a non-negative integer`;
|
|
444
|
+
const perType = rc['per_type'];
|
|
445
|
+
if (perType !== undefined) {
|
|
446
|
+
if (!isPlainObject(perType))
|
|
447
|
+
return `per_type ${JSON.stringify(perType)} is not an object keyed by steering type`;
|
|
448
|
+
for (const [t, row] of Object.entries(perType)) {
|
|
449
|
+
if (!STEERING_TYPES.has(t))
|
|
450
|
+
return `per_type key ${JSON.stringify(t)} is not one of ${STEERING_TYPE_VALUES.join('|')}`;
|
|
451
|
+
if (!isPlainObject(row))
|
|
452
|
+
return `per_type.${t} ${JSON.stringify(row)} is not an object { exercised, unexercised }`;
|
|
453
|
+
for (const k of ['exercised', 'unexercised']) {
|
|
454
|
+
if (!isCount(row[k]))
|
|
455
|
+
return `per_type.${t}.${k} ${JSON.stringify(row[k])} is not a non-negative integer`;
|
|
456
|
+
}
|
|
457
|
+
}
|
|
458
|
+
}
|
|
459
|
+
return null;
|
|
460
|
+
}
|
|
461
|
+
/**
|
|
462
|
+
* Reconcile one record's `rule_coverage` with ITSELF and with its rows, pushing every contradiction
|
|
463
|
+
* to `errors` (naming the run), and say what it was checked against (`mode`). A value that is not
|
|
464
|
+
* the wire shape (`coverageShapeProblem`) is ONE error naming the defect and `null` — nothing else is
|
|
465
|
+
* read from it. Whatever the filter: no duplicate unexercised id; `per_type`, when present, sums to
|
|
466
|
+
* `exercised` and agrees per type with the listed rows; no listed-unexercised id fired (the row types
|
|
467
|
+
* the rule into the slice; a firing makes an eligible rule exercised — evals.rs 862-873). Unfiltered
|
|
468
|
+
* (`'rows'`): `exercised` is at least the distinct blocking-fired ids. Filtered: every listed row is
|
|
469
|
+
* of the filter's type, and with `per_type` present (`'per_type'`) the slice's own row equals the
|
|
470
|
+
* total — the rows' fired ids may name rules of other types and are NOT a denominator check; without
|
|
471
|
+
* `per_type` (`'n/a …'`) nothing more can be checked.
|
|
472
|
+
*/
|
|
473
|
+
function reconcileCoverage(run, rows, label, errors) {
|
|
474
|
+
const who = `run ${label} (${run.id})`;
|
|
475
|
+
const malformed = coverageShapeProblem(run.rule_coverage);
|
|
476
|
+
if (malformed !== null) {
|
|
477
|
+
errors.push(`${who}: rule_coverage is malformed — ${malformed} — not the wire shape (api-types GovernanceEvalRuleCoverage), so it is not reconciled and no coverage delta is computed over it`);
|
|
478
|
+
return null;
|
|
479
|
+
}
|
|
480
|
+
const rc = run.rule_coverage;
|
|
481
|
+
const filter = run.type_filter;
|
|
482
|
+
const fired = new Set();
|
|
483
|
+
for (const r of rows)
|
|
484
|
+
for (const id of r.fired)
|
|
485
|
+
fired.add(id);
|
|
486
|
+
const unexercised = new Set();
|
|
487
|
+
const listedByType = new Map();
|
|
488
|
+
for (const u of rc.unexercised) {
|
|
489
|
+
if (unexercised.has(u.rule_id))
|
|
490
|
+
errors.push(`${who}: rule_coverage.unexercised lists ${u.rule_id} twice — a rule is unexercised once or not at all`);
|
|
491
|
+
unexercised.add(u.rule_id);
|
|
492
|
+
listedByType.set(u.steering_type, (listedByType.get(u.steering_type) ?? 0) + 1);
|
|
493
|
+
if (filter !== null && u.steering_type !== filter) {
|
|
494
|
+
errors.push(`${who}: rule_coverage.unexercised lists ${u.rule_id} (${u.steering_type}) under type filter ${filter} — the slice's eligible rules are all of the filter's type (evals.rs decide_lane_rules in_slice)`);
|
|
495
|
+
}
|
|
496
|
+
if (fired.has(u.rule_id)) {
|
|
497
|
+
errors.push(`${who}: rule_coverage.unexercised lists ${u.rule_id}, which fired (blocking) in its results — a rule that fired for any sample is exercised, never unexercised`);
|
|
498
|
+
}
|
|
499
|
+
}
|
|
500
|
+
// `per_type` is read as possibly incomplete: a persisted record is verbatim, and a row missing
|
|
501
|
+
// for a type is a contradiction to NAME, not an index to crash on.
|
|
502
|
+
const perType = rc.per_type;
|
|
503
|
+
if (perType !== undefined) {
|
|
504
|
+
let sumExercised = 0;
|
|
505
|
+
for (const row of Object.values(perType))
|
|
506
|
+
if (row !== undefined)
|
|
507
|
+
sumExercised += row.exercised;
|
|
508
|
+
if (sumExercised !== rc.exercised) {
|
|
509
|
+
errors.push(`${who}: rule_coverage.per_type sums to ${sumExercised} exercised but rule_coverage.exercised is ${rc.exercised} — the per-type rows partition the same eligible rules`);
|
|
510
|
+
}
|
|
511
|
+
for (const t of [...new Set([...Object.keys(perType), ...listedByType.keys()])].sort(codepoint)) {
|
|
512
|
+
const counted = perType[t]?.unexercised;
|
|
513
|
+
const listed = listedByType.get(t) ?? 0;
|
|
514
|
+
if (counted === undefined)
|
|
515
|
+
errors.push(`${who}: rule_coverage.unexercised lists ${listed} ${t} rule(s) but rule_coverage.per_type carries no ${t} row`);
|
|
516
|
+
else if (counted !== listed)
|
|
517
|
+
errors.push(`${who}: rule_coverage.per_type.${t}.unexercised is ${counted} but rule_coverage.unexercised lists ${listed} ${t} rule(s)`);
|
|
518
|
+
}
|
|
519
|
+
}
|
|
520
|
+
const facts = { exercised: rc.exercised, unexercised, fired };
|
|
521
|
+
if (filter === null) {
|
|
522
|
+
if (rc.exercised < fired.size) {
|
|
523
|
+
errors.push(`${who}: rule_coverage.exercised ${rc.exercised} is below the ${fired.size} distinct rule id(s) fired (blocking) across its results (${[...fired].sort(codepoint).join(', ')}) — every blocking firing is an exercised rule`);
|
|
524
|
+
}
|
|
525
|
+
return { mode: 'rows', ...facts };
|
|
526
|
+
}
|
|
527
|
+
if (perType === undefined)
|
|
528
|
+
return { mode: 'n/a (engine reports no per-type coverage)', ...facts };
|
|
529
|
+
const slice = perType[filter];
|
|
530
|
+
if (slice === undefined) {
|
|
531
|
+
errors.push(`${who}: rule_coverage.per_type carries no ${filter} row although the run was filtered to ${filter} — the slice's own numbers are missing`);
|
|
532
|
+
}
|
|
533
|
+
else if (slice.exercised !== rc.exercised) {
|
|
534
|
+
errors.push(`${who}: rule_coverage.per_type.${filter}.exercised is ${slice.exercised} but rule_coverage.exercised is ${rc.exercised} — under type filter ${filter} the eligible rules are exactly the ${filter} slice (evals.rs decide_lane_rules), so the slice's row IS the total`);
|
|
535
|
+
}
|
|
536
|
+
return { mode: 'per_type', ...facts };
|
|
537
|
+
}
|
|
538
|
+
/**
|
|
539
|
+
* The delta of two records produced under the SAME `type_filter` (`filter`), over what they ENUMERATE
|
|
540
|
+
* (module doc, "Coverage transitions vs. rule-set changes"): `gained`/`lost` from ids listed on both
|
|
541
|
+
* ends, `unidentified` per record, and EVERY one-sided id withheld with its reason — the wire carries
|
|
542
|
+
* no rule inventory, so `added_rules`/`removed_rules` are never asserted and `inventory` is `partial`.
|
|
543
|
+
*/
|
|
544
|
+
function coverageDelta(covA, covB, filter) {
|
|
545
|
+
// Certain, both ends listed: the unexercised row names the id on one side, the blocking firing on
|
|
546
|
+
// the other (under a filter the row carries the filter's steering_type; a firing is a firing).
|
|
547
|
+
const gained = [...covA.unexercised].filter((id) => covB.fired.has(id)).sort(codepoint);
|
|
548
|
+
const lost = [...covA.fired].filter((id) => covB.unexercised.has(id)).sort(codepoint);
|
|
549
|
+
const unidentified = { a: unidentifiedOf(covA, filter), b: unidentifiedOf(covB, filter) };
|
|
550
|
+
// The identities each record ENUMERATES — its unexercised list plus its blocking-fired ids.
|
|
551
|
+
// Nothing else about a run's rule set is on the wire, so an id the other side does not enumerate
|
|
552
|
+
// is a candidate for a rule-set change and NOTHING more: withheld, with what else it may be.
|
|
553
|
+
const knownA = new Set([...covA.unexercised, ...covA.fired]);
|
|
554
|
+
const knownB = new Set([...covB.unexercised, ...covB.fired]);
|
|
555
|
+
const withheld = [
|
|
556
|
+
...[...knownB].filter((id) => !knownA.has(id)).map((id) => withheldReason('added_rules', id, covB, unidentified.a, filter)),
|
|
557
|
+
...[...knownA].filter((id) => !knownB.has(id)).map((id) => withheldReason('removed_rules', id, covA, unidentified.b, filter)),
|
|
558
|
+
].sort(codepoint);
|
|
559
|
+
return {
|
|
560
|
+
exercised_delta: covB.exercised - covA.exercised,
|
|
561
|
+
inventory: 'partial',
|
|
562
|
+
unidentified,
|
|
563
|
+
gained,
|
|
564
|
+
lost,
|
|
565
|
+
added_rules: [],
|
|
566
|
+
removed_rules: [],
|
|
567
|
+
transitions_withheld: withheld,
|
|
568
|
+
};
|
|
569
|
+
}
|
|
570
|
+
/** How many exercised rules of its denominator ONE record does not name. Unfiltered, every
|
|
571
|
+
* blocking-fired id is an exercised rule of the (universal) denominator, so `exercised − |fired|`
|
|
572
|
+
* (floored at 0: an undercut was already reported as a reconciliation error). Under a type filter a
|
|
573
|
+
* fired id carries no steering_type — the record types none of its own firings into the slice — so
|
|
574
|
+
* the whole count; what the OTHER run lists never reduces it (codex round 8). */
|
|
575
|
+
function unidentifiedOf(cov, filter) {
|
|
576
|
+
return filter === null ? Math.max(0, cov.exercised - cov.fired.size) : cov.exercised;
|
|
577
|
+
}
|
|
578
|
+
/** Why an id enumerated by one record (`lister`: b for `added_rules`, a for `removed_rules`) and not
|
|
579
|
+
* by the other is withheld instead of asserted as a rule-set change — one sentence naming what the
|
|
580
|
+
* id may be instead. `silentUnidentified` is the silent side's `unidentified` count. */
|
|
581
|
+
function withheldReason(kind, id, lister, silentUnidentified, filter) {
|
|
582
|
+
const [listerLabel, silent] = kind === 'added_rules' ? ['b', 'a'] : ['a', 'b'];
|
|
583
|
+
const listedAs = lister.unexercised.has(id) ? 'unexercised' : 'blocking-fired';
|
|
584
|
+
const maybe = [];
|
|
585
|
+
if (silentUnidentified > 0) {
|
|
586
|
+
maybe.push(`one of ${silent}'s ${silentUnidentified} exercised rule(s) named nowhere (${filter === null ? 'fired by a non-blocking effect alone' : `under type filter ${filter} no fired id is typed into the slice`})`);
|
|
587
|
+
}
|
|
588
|
+
maybe.push('an effect-less (recall-only) or retired rule outside the eligible partition');
|
|
589
|
+
if (filter !== null)
|
|
590
|
+
maybe.push(`a rule of another type outside the ${filter} denominator (a row's fired ids carry no steering_type)`);
|
|
591
|
+
const change = `a rule the store ${kind === 'added_rules' ? 'gained' : 'lost'}`;
|
|
592
|
+
return (`${kind} ${id}: enumerated by ${listerLabel} (${listedAs}) and not by ${silent} — the wire lists no rule inventory (rule_coverage.exercised and recall_only are counts; only unexercised and blocking-fired ids are named), ` +
|
|
593
|
+
`so ${silent}'s silence is not absence: ${id} may be ${maybe.join(', ')} or ${change}: not asserted`);
|
|
594
|
+
}
|
|
595
|
+
function field(v) {
|
|
596
|
+
return v === 'caught' ? 'caught' : v === 'gap' ? 'gaps' : 'false_positives';
|
|
597
|
+
}
|
|
598
|
+
function tally(results) {
|
|
599
|
+
const t = { total: results.length, caught: 0, gaps: 0, false_positives: 0 };
|
|
600
|
+
for (const r of results)
|
|
601
|
+
t[field(r.verdict)] += 1;
|
|
602
|
+
return t;
|
|
603
|
+
}
|
|
604
|
+
function sameSummary(x, y) {
|
|
605
|
+
return x.total === y.total && x.caught === y.caught && x.gaps === y.gaps && x.false_positives === y.false_positives;
|
|
606
|
+
}
|
|
607
|
+
function fmt(s) {
|
|
608
|
+
return `{total ${s.total}, caught ${s.caught}, gaps ${s.gaps}, false_positives ${s.false_positives}}`;
|
|
609
|
+
}
|
|
610
|
+
//# sourceMappingURL=eval-compare.js.map
|