@dzhechkov/harness-core 0.3.134 → 0.3.136

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/sbom.json CHANGED
@@ -25,7 +25,7 @@
25
25
  "hashes": [
26
26
  {
27
27
  "alg": "SHA-256",
28
- "content": "372b0c006527c97bdd9be951af4843f3c02206db197bcab9e62c5940e48cf761"
28
+ "content": "e8d5cf556d6c03587d16b792f9f3ca48a8724ba8485ae7a595b4dabf2e891a98"
29
29
  }
30
30
  ]
31
31
  },
@@ -569,6 +569,46 @@
569
569
  }
570
570
  ]
571
571
  },
572
+ {
573
+ "type": "file",
574
+ "name": "dist/compounding.d.ts",
575
+ "hashes": [
576
+ {
577
+ "alg": "SHA-256",
578
+ "content": "ec191d943989b0c7d96acb31ab31eefc16a539751cd1c94e3af20105a152beff"
579
+ }
580
+ ]
581
+ },
582
+ {
583
+ "type": "file",
584
+ "name": "dist/compounding.d.ts.map",
585
+ "hashes": [
586
+ {
587
+ "alg": "SHA-256",
588
+ "content": "7c78039462fbb3b4990907d8bcc7821b8d46b4962bc6b79397605eb195127f6b"
589
+ }
590
+ ]
591
+ },
592
+ {
593
+ "type": "file",
594
+ "name": "dist/compounding.js",
595
+ "hashes": [
596
+ {
597
+ "alg": "SHA-256",
598
+ "content": "c0500edbd72a22e8850a231078849147a445956d3e9c5894c06d257a4dad5197"
599
+ }
600
+ ]
601
+ },
602
+ {
603
+ "type": "file",
604
+ "name": "dist/compounding.js.map",
605
+ "hashes": [
606
+ {
607
+ "alg": "SHA-256",
608
+ "content": "dfbf44a32606c8ad92fbd21a88ca615f9944802ff3d2d12de8fccbb1036dfad0"
609
+ }
610
+ ]
611
+ },
572
612
  {
573
613
  "type": "file",
574
614
  "name": "dist/cost-scoring.d.ts",
@@ -1015,7 +1055,7 @@
1015
1055
  "hashes": [
1016
1056
  {
1017
1057
  "alg": "SHA-256",
1018
- "content": "63b513058606e37cf080176b76be65b33f21ded4f05b25e78ec56e3a978ac16b"
1058
+ "content": "a6cd04473eff318f35da483c58a5b16d158c81ed6a838c4ea0ab5772498f0f61"
1019
1059
  }
1020
1060
  ]
1021
1061
  },
@@ -1025,7 +1065,7 @@
1025
1065
  "hashes": [
1026
1066
  {
1027
1067
  "alg": "SHA-256",
1028
- "content": "92b4313165d607a7acb9321f490801f8b5d24a202050f7363f5df4836e4fc1de"
1068
+ "content": "a9dfb77a1bff8eade9cfa4c0aa7c6af26704fd60b38756dd6c19e7dc935cbe19"
1029
1069
  }
1030
1070
  ]
1031
1071
  },
@@ -1035,7 +1075,7 @@
1035
1075
  "hashes": [
1036
1076
  {
1037
1077
  "alg": "SHA-256",
1038
- "content": "dd5e32274dee6e2bc787977f4dca188101e77ac74bc9599be56286ba12a3c32f"
1078
+ "content": "585e0b9a9a7d010b35877f5e24e5aafe8dd83efe75f271faa13d0ef49458a294"
1039
1079
  }
1040
1080
  ]
1041
1081
  },
@@ -1045,7 +1085,7 @@
1045
1085
  "hashes": [
1046
1086
  {
1047
1087
  "alg": "SHA-256",
1048
- "content": "5e47bad5dd4e137a9ac61272419e6ad2f23166069776d5db364bc0d8f9d8dee3"
1088
+ "content": "67b706874322e3616d90a9679ee719d4718a28f871a83bddd7ad9815b7bd5b9e"
1049
1089
  }
1050
1090
  ]
1051
1091
  },
@@ -1145,7 +1185,7 @@
1145
1185
  "hashes": [
1146
1186
  {
1147
1187
  "alg": "SHA-256",
1148
- "content": "4ba4f80508fd7d959c49a88eb7995effe1807b891518d6089e4dec50ac2fd1f2"
1188
+ "content": "a1316ae9b953735335d02280177b4ae845e0e5c8452ea14113fa86327a786bc1"
1149
1189
  }
1150
1190
  ]
1151
1191
  },
@@ -1155,7 +1195,7 @@
1155
1195
  "hashes": [
1156
1196
  {
1157
1197
  "alg": "SHA-256",
1158
- "content": "a6f6d044e09f7678305296491ab43dabf454eec798d4b3bfcd167220f6fd3d96"
1198
+ "content": "32c16d8d0793279e60f892fdc8daaa70aeda74d9948c54a3fe6fa4cdf21cc2ea"
1159
1199
  }
1160
1200
  ]
1161
1201
  },
@@ -1165,7 +1205,7 @@
1165
1205
  "hashes": [
1166
1206
  {
1167
1207
  "alg": "SHA-256",
1168
- "content": "0a8cd9f2fe6ff4d3a9c5bf7a342e8480eb0a221a0755a118d6788297cbeccded"
1208
+ "content": "674495cbb7a24844e82bbc70c7ac825c84f3572144acf310ab49f2ae791ddf7e"
1169
1209
  }
1170
1210
  ]
1171
1211
  },
@@ -1535,7 +1575,7 @@
1535
1575
  "hashes": [
1536
1576
  {
1537
1577
  "alg": "SHA-256",
1538
- "content": "1524e786d8d1eaf4777fc34b0cb66c5f1e01af78abca29361eaec17fff9a6e5c"
1578
+ "content": "d938a195e56fac7103c659ba939c9838421eede40b5f81ae1928b6cc372429e4"
1539
1579
  }
1540
1580
  ]
1541
1581
  },
@@ -1545,7 +1585,7 @@
1545
1585
  "hashes": [
1546
1586
  {
1547
1587
  "alg": "SHA-256",
1548
- "content": "e4f528ac602e93ff3b4709d7f4b2def3f72dc0bf064b247574b5c9fc3d4a6470"
1588
+ "content": "82a5f94b7a1ee8b70e80e36b85bb3d9c2fee68330dd4ec81aee11c81a14ea372"
1549
1589
  }
1550
1590
  ]
1551
1591
  },
@@ -1555,7 +1595,7 @@
1555
1595
  "hashes": [
1556
1596
  {
1557
1597
  "alg": "SHA-256",
1558
- "content": "9b0121f2f8c345dc634b55e4ce9ad42aa6ab126844b1b374dbe4e7f7fa47bc1d"
1598
+ "content": "4530f6d3bd52aa1caade1ec732dca19aeb8cc2400a2c5281b054b24a32e89a19"
1559
1599
  }
1560
1600
  ]
1561
1601
  },
@@ -1565,7 +1605,7 @@
1565
1605
  "hashes": [
1566
1606
  {
1567
1607
  "alg": "SHA-256",
1568
- "content": "dc8993d52df53cbb95e1f93cd4bf35f592d540dd23a31820dc48014600fba154"
1608
+ "content": "d33bc9b0ba04f134cb8a36dc8f5c801a8c245dd8e55c05785a3fb77116d8085d"
1569
1609
  }
1570
1610
  ]
1571
1611
  },
@@ -2345,7 +2385,7 @@
2345
2385
  "hashes": [
2346
2386
  {
2347
2387
  "alg": "SHA-256",
2348
- "content": "398863280426f933afb7f962cb5ee51f71dc42188df364cb33f4ee07d09f7c68"
2388
+ "content": "c4587f4494a98d98aebf81919a3626742302ef8a56fb698541c860496edd326f"
2349
2389
  }
2350
2390
  ]
2351
2391
  },
@@ -2355,7 +2395,7 @@
2355
2395
  "hashes": [
2356
2396
  {
2357
2397
  "alg": "SHA-256",
2358
- "content": "6e7be0ee532ac474926d8a557c8f160ec1d689262930116ee08a326115f9648d"
2398
+ "content": "f21b532a45cb8902ee73614c321ab843f69d0c7f9bfb1d027f78344c21b93d67"
2359
2399
  }
2360
2400
  ]
2361
2401
  },
@@ -2365,7 +2405,7 @@
2365
2405
  "hashes": [
2366
2406
  {
2367
2407
  "alg": "SHA-256",
2368
- "content": "3c5b965649ba792514532aefee4a48f3b47b37b3bc764ddc294fce0fb9a36d80"
2408
+ "content": "49010990ed212f5660d3ef506416b128ead1ae83797117ec577b0f263c679961"
2369
2409
  }
2370
2410
  ]
2371
2411
  },
@@ -2455,7 +2495,7 @@
2455
2495
  "hashes": [
2456
2496
  {
2457
2497
  "alg": "SHA-256",
2458
- "content": "b55b94412729f754ffb2ab36807d6bcf0eadbd5690fbeb7d143190c45fd36e0e"
2498
+ "content": "87dd771f68992eb5ed6d6cfd4d6e1148f684b574eea49c8c669c2b7573c8d97e"
2459
2499
  }
2460
2500
  ]
2461
2501
  },
@@ -2589,6 +2629,16 @@
2589
2629
  }
2590
2630
  ]
2591
2631
  },
2632
+ {
2633
+ "type": "file",
2634
+ "name": "src/compounding.ts",
2635
+ "hashes": [
2636
+ {
2637
+ "alg": "SHA-256",
2638
+ "content": "2f051ebc8d86b6de4d359a5482c25b8cee59261a790599ced85c20162e0c87b1"
2639
+ }
2640
+ ]
2641
+ },
2592
2642
  {
2593
2643
  "type": "file",
2594
2644
  "name": "src/cost-scoring.ts",
@@ -2705,7 +2755,7 @@
2705
2755
  "hashes": [
2706
2756
  {
2707
2757
  "alg": "SHA-256",
2708
- "content": "c224021d9de3fbfd777ad74dc2d65baefa112b98e6fcee8eeac4bc0c83fa8b60"
2758
+ "content": "90df35fb71d36b4e2b38eaad98acc6437cb26036b7b021e10cae2bc99d5546fd"
2709
2759
  }
2710
2760
  ]
2711
2761
  },
@@ -2735,7 +2785,7 @@
2735
2785
  "hashes": [
2736
2786
  {
2737
2787
  "alg": "SHA-256",
2738
- "content": "78143fffb04e9ed2f428847966084fbec503e2cd0286664767acc34c7cc3af48"
2788
+ "content": "80b292202fad0be0e9939f67641f16987e93f962b547b56e9771d9ae9a7c94b1"
2739
2789
  }
2740
2790
  ]
2741
2791
  },
@@ -2825,7 +2875,7 @@
2825
2875
  "hashes": [
2826
2876
  {
2827
2877
  "alg": "SHA-256",
2828
- "content": "df8499d9a4394b8ec257c2359d358f5113ff9c70c78949806f060c0317692cfb"
2878
+ "content": "4fc2483e6142f5b1f11fd6c2b810be2a25171835551320ef0fde3e296962f693"
2829
2879
  }
2830
2880
  ]
2831
2881
  },
@@ -3025,7 +3075,7 @@
3025
3075
  "hashes": [
3026
3076
  {
3027
3077
  "alg": "SHA-256",
3028
- "content": "e2b01c71101c2cdfa076a26148e1c26b37d28204ac33b561d98e89cdc30d238f"
3078
+ "content": "a74ad3ca87279e0cb808a40dd8f18f52efed5c6f37f45f8186ea573c4a610fdd"
3029
3079
  }
3030
3080
  ]
3031
3081
  },
@@ -3189,6 +3239,16 @@
3189
3239
  }
3190
3240
  ]
3191
3241
  },
3242
+ {
3243
+ "type": "file",
3244
+ "name": "test/compounding.test.ts",
3245
+ "hashes": [
3246
+ {
3247
+ "alg": "SHA-256",
3248
+ "content": "125ae7b36ad75488d4dd35edc585fc1523691c3f156d3ac0a3b891fd804645d1"
3249
+ }
3250
+ ]
3251
+ },
3192
3252
  {
3193
3253
  "type": "file",
3194
3254
  "name": "test/cost-scoring.test.ts",
@@ -3735,7 +3795,7 @@
3735
3795
  "hashes": [
3736
3796
  {
3737
3797
  "alg": "SHA-256",
3738
- "content": "3159f518ee1cad89a8ed4f996361d1584188811232f7aa8f03af4985b072cda7"
3798
+ "content": "09b9c4ed5bf879e71a6940df7255db9d61c160061e34a7ff596733eaf7775de9"
3739
3799
  }
3740
3800
  ]
3741
3801
  },
@@ -0,0 +1,314 @@
1
+ /**
2
+ * `dz compounding` — does the learning loop actually PAY? (feature compounding, scout C2)
3
+ *
4
+ * Ported from rUv's darwin-mode (`security/compounding.ts`, `security/ablation.ts`,
5
+ * `bench/{stats,promotion}.ts`) with an honesty split the port map demanded:
6
+ * - the STATS machinery ports verbatim (seeded mulberry32, bootstrap lower-95, decidePromotion,
7
+ * the min-n >= 5 rule — darwin's own FDR calibration shows n=3 gives a 33% false-discovery rate);
8
+ * - darwin's MEASUREMENT legs do NOT port: its FP-drop leg ignores the passed corpus (a fixture),
9
+ * `withoutMemory` is hard-coded 0, and "warm" is injected state — theatrical, exactly what this
10
+ * repo's claim-check culture forbids. The measurements here are dz-native, over data that exists.
11
+ *
12
+ * The report NEVER fakes a verdict: a gate without enough samples says INSUFFICIENT_DATA — after the
13
+ * 2026-07-28 inventory found the apply-leg log dead for 19 days, "no data" is a finding, not a pass.
14
+ *
15
+ * Everything here is PURE: callers gather facts (files, store rows); this module only computes.
16
+ */
17
+
18
+ // ── Seeded statistics (verbatim-shape port from darwin-mode bench/stats.ts) ──
19
+
20
+ /** Deterministic PRNG — same seed, same stream, byte-identical reports. */
21
+ export function mulberry32(seed: number): () => number {
22
+ let a = seed >>> 0;
23
+ return () => {
24
+ a = (a + 0x6d2b79f5) >>> 0;
25
+ let t = a;
26
+ t = Math.imul(t ^ (t >>> 15), t | 1);
27
+ t ^= t + Math.imul(t ^ (t >>> 7), t | 61);
28
+ return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
29
+ };
30
+ }
31
+
32
+ export const BOOTSTRAP_RESAMPLES = 5000;
33
+ /** Below this many samples PER ARM a comparison is noise: darwin's own FDR calibration measured a
34
+ * 0.332 empirical false-discovery rate at n=3. */
35
+ export const MIN_SAMPLES_PER_ARM = 5;
36
+
37
+ export interface BootstrapDelta {
38
+ readonly meanDelta: number;
39
+ /** 2.5th percentile of the resampled deltas — the promotion decision reads THIS, not the mean. */
40
+ readonly lower95: number;
41
+ readonly samples: number;
42
+ }
43
+
44
+ /** Paired bootstrap over per-item deltas (b[i] - a[i]). */
45
+ export function bootstrapDelta(a: readonly number[], b: readonly number[], seed = 42): BootstrapDelta | null {
46
+ // PAIRED means paired: unequal lengths silently truncated a decisive observation and promoted on
47
+ // the remainder; a sparse/NaN entry is not an observation at all (Codex #9).
48
+ if (a.length !== b.length || a.length === 0) return null;
49
+ if (![...a, ...b].every((x) => typeof x === 'number' && Number.isFinite(x))) return null;
50
+ const n = a.length;
51
+ const deltas: number[] = [];
52
+ for (let i = 0; i < n; i++) deltas.push((b[i] ?? 0) - (a[i] ?? 0));
53
+ const rand = mulberry32(seed);
54
+ const means: number[] = [];
55
+ for (let r = 0; r < BOOTSTRAP_RESAMPLES; r++) {
56
+ let sum = 0;
57
+ for (let i = 0; i < n; i++) sum += deltas[Math.floor(rand() * n)] ?? 0;
58
+ means.push(sum / n);
59
+ }
60
+ means.sort((x, y) => x - y);
61
+ const meanDelta = deltas.reduce((s, d) => s + d, 0) / n;
62
+ // Conservative nearest-rank percentile: ceil(B*p)-1. floor(B*p) sat one slot ABOVE the 2.5th
63
+ // percentile and could flip a reject into a promote at the boundary (Codex #10 — a defect darwin
64
+ // itself inherits; ported faithfully was still ported wrong).
65
+ const lower95 = means[Math.max(0, Math.ceil(BOOTSTRAP_RESAMPLES * 0.025) - 1)] ?? 0;
66
+ return { meanDelta, lower95, samples: n };
67
+ }
68
+
69
+ export type PromotionVerdict = 'promote' | 'reject' | 'insufficient-data';
70
+
71
+ /** Darwin's decision rule: a positive mean is not enough — the LOWER bound must clear zero. */
72
+ export function decidePromotion(delta: BootstrapDelta | null, minDelta = 0): PromotionVerdict {
73
+ // NaN samples compared false against the minimum and PROMOTED (Codex #9) — every field must be a
74
+ // real number and the count a real integer before any decision exists.
75
+ if (
76
+ delta === null ||
77
+ !Number.isInteger(delta.samples) ||
78
+ delta.samples < MIN_SAMPLES_PER_ARM ||
79
+ !Number.isFinite(delta.meanDelta) ||
80
+ !Number.isFinite(delta.lower95) ||
81
+ !Number.isFinite(minDelta)
82
+ ) {
83
+ return 'insufficient-data';
84
+ }
85
+ return delta.meanDelta > minDelta && delta.lower95 > 0 ? 'promote' : 'reject';
86
+ }
87
+
88
+ // ── The dz-native facts the CLI gathers ─────────────────────────────
89
+
90
+ export interface LessonRow {
91
+ readonly dzId: string;
92
+ readonly uses: number;
93
+ readonly quarantined: boolean;
94
+ readonly reward: number | null;
95
+ }
96
+
97
+ export interface UsageEvent {
98
+ readonly dzId: string;
99
+ readonly ts: string;
100
+ readonly query?: string;
101
+ readonly runId?: string;
102
+ /** One id per PROMPT: the hook writes one row per injected hit (up to 3 per prompt), and counting
103
+ * rows as independent replay pairs fabricated readiness (Codex #1). */
104
+ readonly eventId?: string;
105
+ /** A truncated query cannot reproduce the original recall — it must not count (Codex #3). */
106
+ readonly queryTruncated?: boolean;
107
+ }
108
+
109
+ export interface GuardEvent {
110
+ readonly ts: string;
111
+ readonly verdict: string;
112
+ readonly rules: readonly string[]; // violated rule ids
113
+ }
114
+
115
+ export interface CompoundingFacts {
116
+ readonly lessons: readonly LessonRow[];
117
+ readonly usage: readonly UsageEvent[];
118
+ readonly guard: readonly GuardEvent[];
119
+ readonly nowTs: string;
120
+ }
121
+
122
+ // ── The report ──────────────────────────────────────────────────────
123
+
124
+ export interface PoolPayoff {
125
+ readonly total: number;
126
+ /** Ever surfaced by the APPLY leg (hook injection) — the strict payoff bar. */
127
+ readonly injectedEver: number;
128
+ /** Touched by ANY recall path (store `uses` counter). */
129
+ readonly touchedEver: number;
130
+ readonly neverTouched: number;
131
+ readonly quarantined: number;
132
+ /** Fraction of the pool that is write-only under the strict bar. */
133
+ readonly writeOnlyRatio: number;
134
+ }
135
+
136
+ export interface GuardRuleTrajectory {
137
+ readonly rule: string;
138
+ readonly firstHalfViolations: number;
139
+ readonly secondHalfViolations: number;
140
+ readonly firstHalfAudits: number;
141
+ readonly secondHalfAudits: number;
142
+ /** Improvement is judged on the RATE (violations per audit), not raw counts: ten violations in a
143
+ * hundred early audits vs one in one late audit is a WORSENING, not progress (Codex #7). */
144
+ readonly improved: boolean;
145
+ }
146
+
147
+ export type ReadinessVerdict = 'ready' | 'insufficient-data';
148
+
149
+ export interface ReplayReadiness {
150
+ /** UNIQUE, untruncated prompt events — the pairs a cold-vs-warm replay needs. */
151
+ readonly replayablePairs: number;
152
+ readonly minNeeded: number;
153
+ /** READINESS only. `promote`/`reject` exist solely after a real cold/warm A-B has been run and
154
+ * bootstrapped — readiness must never look like a result (Codex #1). */
155
+ readonly verdict: ReadinessVerdict;
156
+ readonly note: string;
157
+ }
158
+
159
+ export interface InstrumentationHealth {
160
+ readonly lastUsageTs: string | null;
161
+ readonly gapDays: number | null;
162
+ /** True when the newest usage record is recent enough to trust the leg is alive. */
163
+ readonly applyLegLive: boolean;
164
+ }
165
+
166
+ export interface CompoundingReport {
167
+ readonly pool: PoolPayoff;
168
+ readonly guardTrajectory: readonly GuardRuleTrajectory[];
169
+ readonly replay: ReplayReadiness;
170
+ readonly instrumentation: InstrumentationHealth;
171
+ /** The one-line honest answer. */
172
+ readonly verdict: string;
173
+ }
174
+
175
+ const APPLY_LEG_STALE_DAYS = 7;
176
+
177
+ export function assembleCompoundingReport(facts: CompoundingFacts): CompoundingReport {
178
+ const { lessons, usage, guard } = facts;
179
+
180
+ // 1. Pool payoff — the "loops need all three legs" question, quantified.
181
+ const injectedIds = new Set(usage.map((u) => u.dzId));
182
+ const injectedEver = lessons.filter((l) => injectedIds.has(l.dzId)).length;
183
+ const touchedEver = lessons.filter((l) => l.uses > 0 || injectedIds.has(l.dzId)).length;
184
+ const total = lessons.length;
185
+ const pool: PoolPayoff = {
186
+ total,
187
+ injectedEver,
188
+ touchedEver,
189
+ neverTouched: total - touchedEver,
190
+ quarantined: lessons.filter((l) => l.quarantined).length,
191
+ writeOnlyRatio: total === 0 ? 0 : (total - injectedEver) / total,
192
+ };
193
+
194
+ // 2. Guard trajectory — do the same mistakes recur less over time? Split the record span in half
195
+ // by TIME (not by count: a busy afternoon must not masquerade as an era).
196
+ // Only events with a PARSEABLE timestamp participate; a span of zero has no halves (Codex #7).
197
+ const guardTimed = guard
198
+ .map((g) => ({ ...g, ms: Date.parse(g.ts) }))
199
+ .filter((g) => Number.isFinite(g.ms))
200
+ .sort((a, b) => a.ms - b.ms);
201
+ const trajectory: GuardRuleTrajectory[] = [];
202
+ const t0 = guardTimed.length > 0 ? guardTimed[0]!.ms : 0;
203
+ const t1 = guardTimed.length > 0 ? guardTimed[guardTimed.length - 1]!.ms : 0;
204
+ if (guardTimed.length >= 2 && t1 > t0) {
205
+ const mid = t0 + (t1 - t0) / 2;
206
+ let firstAudits = 0;
207
+ let secondAudits = 0;
208
+ for (const g of guardTimed) {
209
+ if (g.ms <= mid) firstAudits += 1;
210
+ else secondAudits += 1;
211
+ }
212
+ const perRule = new Map<string, { first: number; second: number }>();
213
+ for (const g of guardTimed) {
214
+ const inFirst = g.ms <= mid;
215
+ for (const rule of g.rules) {
216
+ const e = perRule.get(rule) ?? { first: 0, second: 0 };
217
+ if (inFirst) e.first += 1;
218
+ else e.second += 1;
219
+ perRule.set(rule, e);
220
+ }
221
+ }
222
+ // Both halves must contain OBSERVATIONS for a rate comparison to mean anything.
223
+ if (firstAudits > 0 && secondAudits > 0) {
224
+ for (const [rule, e] of [...perRule.entries()].sort()) {
225
+ const firstRate = e.first / firstAudits;
226
+ const secondRate = e.second / secondAudits;
227
+ trajectory.push({
228
+ rule,
229
+ firstHalfViolations: e.first,
230
+ secondHalfViolations: e.second,
231
+ firstHalfAudits: firstAudits,
232
+ secondHalfAudits: secondAudits,
233
+ improved: secondRate < firstRate,
234
+ });
235
+ }
236
+ }
237
+ }
238
+
239
+ // 3. Replay readiness — cold-vs-warm needs (query -> injected lesson) pairs. They were never
240
+ // recorded before 2026-07-28, so this gate REPORTS accrual instead of faking a verdict.
241
+ const replayKeys = new Set<string>();
242
+ for (const u of usage) {
243
+ if (typeof u.query !== 'string' || u.query.trim() === '') continue;
244
+ if (u.queryTruncated === true) continue; // a prefix is not the prompt
245
+ // one prompt = one pair, however many hits it injected
246
+ replayKeys.add(u.eventId ?? `${u.runId ?? ''}|${u.ts}|${u.query}`);
247
+ }
248
+ const replayablePairs = replayKeys.size;
249
+ const replay: ReplayReadiness = {
250
+ replayablePairs,
251
+ minNeeded: MIN_SAMPLES_PER_ARM,
252
+ verdict: replayablePairs >= MIN_SAMPLES_PER_ARM ? 'ready' : 'insufficient-data',
253
+ note:
254
+ replayablePairs >= MIN_SAMPLES_PER_ARM
255
+ ? `${replayablePairs} unique prompt event(s) recorded — a cold-vs-warm replay can now be RUN (readiness, not a result)`
256
+ : `${replayablePairs} unique prompt event(s); ${MIN_SAMPLES_PER_ARM} needed — queries are recorded as of 2026-07-28, data is accruing`,
257
+ };
258
+
259
+ // 4. Instrumentation health — "no data" must be a finding, never a silent pass.
260
+ // Liveness compares RAW milliseconds (a 7d23h gap floored to "7 days" read as live), rejects
261
+ // garbage timestamps, and treats a FUTURE timestamp beyond small clock skew as evidence of a
262
+ // broken clock, not of liveness (Codex #8).
263
+ const usageTimed = usage
264
+ .map((u) => ({ ts: u.ts, ms: Date.parse(u.ts) }))
265
+ .filter((u) => Number.isFinite(u.ms))
266
+ .sort((a, b) => a.ms - b.ms);
267
+ const lastUsage = usageTimed.length > 0 ? usageTimed[usageTimed.length - 1]! : null;
268
+ const nowMs = Date.parse(facts.nowTs);
269
+ const CLOCK_SKEW_MS = 60_000;
270
+ const gapMs = lastUsage && Number.isFinite(nowMs) ? nowMs - lastUsage.ms : null;
271
+ const gapValid = gapMs !== null && gapMs >= -CLOCK_SKEW_MS;
272
+ const instrumentation: InstrumentationHealth = {
273
+ lastUsageTs: lastUsage?.ts ?? null,
274
+ gapDays: gapValid ? Math.max(0, Math.floor(gapMs / 86_400_000)) : null,
275
+ applyLegLive: gapValid && gapMs <= APPLY_LEG_STALE_DAYS * 86_400_000,
276
+ };
277
+
278
+ const improvedRules = trajectory.filter((t) => t.improved).length;
279
+ const verdict = [
280
+ `pool: ${injectedEver}/${total} lessons ever injected (${Math.round(pool.writeOnlyRatio * 100)}% write-only under the strict bar)`,
281
+ trajectory.length > 0 ? `guard: ${improvedRules}/${trajectory.length} rules recur less in the later half` : 'guard: not enough history',
282
+ `cold-vs-warm: ${replay.verdict === 'insufficient-data' ? 'INSUFFICIENT DATA (accruing)' : 'READY to measure'}`,
283
+ instrumentation.applyLegLive ? 'apply leg: live' : 'apply leg: STALE — fix the instrumentation before trusting anything above',
284
+ ].join(' · ');
285
+
286
+ return { pool, guardTrajectory: trajectory, replay, instrumentation, verdict };
287
+ }
288
+
289
+ export function renderCompoundingReport(r: CompoundingReport): string {
290
+ const out: string[] = [];
291
+ out.push('dz compounding — does the learning loop pay? (honest report: gates without data say so)');
292
+ out.push('');
293
+ out.push(` POOL PAYOFF: ${r.pool.total} lessons · ${r.pool.injectedEver} ever injected by the apply leg · ${r.pool.touchedEver} touched by any recall · ${r.pool.neverTouched} never touched · ${r.pool.quarantined} quarantined`);
294
+ out.push(` write-only ratio (strict bar): ${(r.pool.writeOnlyRatio * 100).toFixed(0)}%`);
295
+ out.push('');
296
+ if (r.guardTrajectory.length > 0) {
297
+ out.push(' GUARD TRAJECTORY (violations, first half vs second half of the audit span):');
298
+ for (const t of r.guardTrajectory) {
299
+ out.push(` ${t.improved ? '↓' : '·'} ${t.rule}: ${t.firstHalfViolations} → ${t.secondHalfViolations}`);
300
+ }
301
+ } else {
302
+ out.push(' GUARD TRAJECTORY: not enough audit history to split');
303
+ }
304
+ out.push('');
305
+ out.push(` COLD-VS-WARM REPLAY: ${r.replay.note}`);
306
+ out.push(
307
+ ` INSTRUMENTATION: last apply-leg record ${r.instrumentation.lastUsageTs ?? 'never'}` +
308
+ (r.instrumentation.gapDays !== null ? ` (${r.instrumentation.gapDays}d ago)` : '') +
309
+ ` — ${r.instrumentation.applyLegLive ? 'live' : 'STALE'}`,
310
+ );
311
+ out.push('');
312
+ out.push(` VERDICT: ${r.verdict}`);
313
+ return out.join('\n');
314
+ }
package/src/index.ts CHANGED
@@ -321,3 +321,8 @@ export * from './delivery-check.js';
321
321
  // Skill-registration gate (feature skills-verify, ADR-001) — static layout scan + the deterministic
322
322
  // `system/init` listing. Fail-closed: an unobservable registration is `inconclusive`, never `pass`.
323
323
  export * from './skills-verify.js';
324
+
325
+ // Learning-loop payoff measurement (feature compounding, scout C2) — seeded stats ported verbatim
326
+ // from darwin-mode; measurements are dz-native and NEVER fake a verdict (INSUFFICIENT_DATA is a
327
+ // finding, not a pass).
328
+ export * from './compounding.js';
package/src/operations.ts CHANGED
@@ -749,6 +749,30 @@ export async function runDoctor(options: { projectRoot: string }): Promise<Docto
749
749
  detail: `deployed writer v${deployed} < current v${AGENTDB_WRITER_VERSION} — re-run dz setup to upgrade (no --force needed)`,
750
750
  });
751
751
  }
752
+ // APPLY-LEG LIVENESS (2026-07-28): the recall hook injects lessons only while the embed daemon's
753
+ // socket is alive, and the daemon is started at SessionStart only — when it died mid-way through a
754
+ // long-lived session the whole apply leg went silently dark for 19 days (MEASURED:
755
+ // recall-usage.jsonl last record 2026-07-09 with the socket absent). A dead leg must be VISIBLE.
756
+ try {
757
+ const settingsPath = join(root, '.claude', 'settings.json');
758
+ if (existsSync(settingsPath)) {
759
+ const settingsText = readFileSync(settingsPath, 'utf-8');
760
+ const applyLegWired = settingsText.includes('recall-hook.cjs') && settingsText.includes('dz-embed-daemon.mjs');
761
+ if (applyLegWired) {
762
+ const sockAlive = existsSync(join(root, '.dz', 'embed.sock'));
763
+ checks.push({
764
+ name: 'apply-leg alive (embed daemon)',
765
+ ok: sockAlive,
766
+ detail: sockAlive
767
+ ? 'embed.sock present — recall injection can run'
768
+ : 'embed.sock ABSENT: the recall hook is wired but cannot inject (the hook self-heals on the next prompt; a persistent absence means the daemon cannot start)',
769
+ });
770
+ }
771
+ }
772
+ } catch {
773
+ /* doctor never throws on a diagnostic */
774
+ }
775
+
752
776
  // Version drift between the local agentdb copy and the .mcp.json pin (gap G7)
753
777
  try {
754
778
  const localVer = (JSON.parse(readFileSync(join(root, 'node_modules', 'agentdb', 'package.json'), 'utf-8')) as { version?: string }).version;