@dzhechkov/harness-core 0.8.35 → 0.8.37

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (135) hide show
  1. package/.dz-manifest.json +224 -104
  2. package/README.md +335 -10
  3. package/dist/agentdb-index.d.ts +87 -7
  4. package/dist/agentdb-index.d.ts.map +1 -1
  5. package/dist/agentdb-index.js +416 -57
  6. package/dist/agentdb-index.js.map +1 -1
  7. package/dist/apply-leg.d.ts +57 -1
  8. package/dist/apply-leg.d.ts.map +1 -1
  9. package/dist/apply-leg.js +450 -52
  10. package/dist/apply-leg.js.map +1 -1
  11. package/dist/codex-hooks-assets.d.ts.map +1 -1
  12. package/dist/codex-hooks-assets.js +67 -5
  13. package/dist/codex-hooks-assets.js.map +1 -1
  14. package/dist/codex-hooks.d.ts +13 -1
  15. package/dist/codex-hooks.d.ts.map +1 -1
  16. package/dist/codex-hooks.js +13 -1
  17. package/dist/codex-hooks.js.map +1 -1
  18. package/dist/codex-rollouts.d.ts +118 -0
  19. package/dist/codex-rollouts.d.ts.map +1 -0
  20. package/dist/codex-rollouts.js +297 -0
  21. package/dist/codex-rollouts.js.map +1 -0
  22. package/dist/cost-ledger.d.ts +56 -4
  23. package/dist/cost-ledger.d.ts.map +1 -1
  24. package/dist/cost-ledger.js +176 -20
  25. package/dist/cost-ledger.js.map +1 -1
  26. package/dist/cross-family-control.d.ts +345 -0
  27. package/dist/cross-family-control.d.ts.map +1 -0
  28. package/dist/cross-family-control.js +802 -0
  29. package/dist/cross-family-control.js.map +1 -0
  30. package/dist/debt-ratchet.d.ts +53 -0
  31. package/dist/debt-ratchet.d.ts.map +1 -0
  32. package/dist/debt-ratchet.js +107 -0
  33. package/dist/debt-ratchet.js.map +1 -0
  34. package/dist/embedding-config.d.ts +42 -0
  35. package/dist/embedding-config.d.ts.map +1 -1
  36. package/dist/embedding-config.js +106 -10
  37. package/dist/embedding-config.js.map +1 -1
  38. package/dist/feature-adr-checkpoints.d.ts +6 -0
  39. package/dist/feature-adr-checkpoints.d.ts.map +1 -1
  40. package/dist/feature-adr-checkpoints.js +29 -0
  41. package/dist/feature-adr-checkpoints.js.map +1 -1
  42. package/dist/feature-adr-decision-recall.d.ts +2 -2
  43. package/dist/feature-adr-decision-recall.d.ts.map +1 -1
  44. package/dist/feature-adr-decision-recall.js +5 -3
  45. package/dist/feature-adr-decision-recall.js.map +1 -1
  46. package/dist/feature-adr-envelope.d.ts +96 -0
  47. package/dist/feature-adr-envelope.d.ts.map +1 -0
  48. package/dist/feature-adr-envelope.js +183 -0
  49. package/dist/feature-adr-envelope.js.map +1 -0
  50. package/dist/feature-adr-routing.d.ts +64 -0
  51. package/dist/feature-adr-routing.d.ts.map +1 -1
  52. package/dist/feature-adr-routing.js +122 -2
  53. package/dist/feature-adr-routing.js.map +1 -1
  54. package/dist/feature-adr-stage-canon.d.ts +79 -0
  55. package/dist/feature-adr-stage-canon.d.ts.map +1 -0
  56. package/dist/feature-adr-stage-canon.js +117 -0
  57. package/dist/feature-adr-stage-canon.js.map +1 -0
  58. package/dist/index.d.ts +23 -12
  59. package/dist/index.d.ts.map +1 -1
  60. package/dist/index.js +15 -7
  61. package/dist/index.js.map +1 -1
  62. package/dist/loop-blobs.generated.js +4 -4
  63. package/dist/loop-blobs.generated.js.map +1 -1
  64. package/dist/mutation-gate.d.ts +51 -0
  65. package/dist/mutation-gate.d.ts.map +1 -1
  66. package/dist/mutation-gate.js +295 -0
  67. package/dist/mutation-gate.js.map +1 -1
  68. package/dist/operations.d.ts +1 -0
  69. package/dist/operations.d.ts.map +1 -1
  70. package/dist/operations.js +18 -2
  71. package/dist/operations.js.map +1 -1
  72. package/dist/publish.d.ts +59 -7
  73. package/dist/publish.d.ts.map +1 -1
  74. package/dist/publish.js +205 -32
  75. package/dist/publish.js.map +1 -1
  76. package/dist/qe-bridge.d.ts.map +1 -1
  77. package/dist/qe-bridge.js +4 -2
  78. package/dist/qe-bridge.js.map +1 -1
  79. package/dist/qe-findings.d.ts +107 -0
  80. package/dist/qe-findings.d.ts.map +1 -0
  81. package/dist/qe-findings.js +417 -0
  82. package/dist/qe-findings.js.map +1 -0
  83. package/dist/recap.d.ts +1 -1
  84. package/dist/recap.d.ts.map +1 -1
  85. package/dist/recap.js +4 -2
  86. package/dist/recap.js.map +1 -1
  87. package/dist/release-line.d.ts +16 -0
  88. package/dist/release-line.d.ts.map +1 -1
  89. package/dist/release-line.js +31 -0
  90. package/dist/release-line.js.map +1 -1
  91. package/dist/round.d.ts +74 -1
  92. package/dist/round.d.ts.map +1 -1
  93. package/dist/round.js +112 -4
  94. package/dist/round.js.map +1 -1
  95. package/dist/run-records.d.ts +60 -0
  96. package/dist/run-records.d.ts.map +1 -1
  97. package/dist/run-records.js +244 -2
  98. package/dist/run-records.js.map +1 -1
  99. package/dist/score.d.ts +44 -1
  100. package/dist/score.d.ts.map +1 -1
  101. package/dist/score.js +78 -5
  102. package/dist/score.js.map +1 -1
  103. package/dist/vector-tier.d.ts +34 -3
  104. package/dist/vector-tier.d.ts.map +1 -1
  105. package/dist/vector-tier.js +105 -14
  106. package/dist/vector-tier.js.map +1 -1
  107. package/package.json +2 -2
  108. package/sbom.json +403 -103
  109. package/src/agentdb-index.ts +423 -60
  110. package/src/apply-leg.ts +469 -50
  111. package/src/codex-hooks-assets.ts +67 -5
  112. package/src/codex-hooks.ts +13 -1
  113. package/src/codex-rollouts.ts +374 -0
  114. package/src/cost-ledger.ts +232 -24
  115. package/src/cross-family-control.ts +960 -0
  116. package/src/debt-ratchet.ts +143 -0
  117. package/src/embedding-config.ts +131 -10
  118. package/src/feature-adr-checkpoints.ts +29 -0
  119. package/src/feature-adr-decision-recall.ts +6 -4
  120. package/src/feature-adr-envelope.ts +242 -0
  121. package/src/feature-adr-routing.ts +139 -2
  122. package/src/feature-adr-stage-canon.ts +141 -0
  123. package/src/index.ts +66 -7
  124. package/src/loop-blobs.generated.ts +4 -4
  125. package/src/mutation-gate.ts +316 -0
  126. package/src/operations.ts +18 -3
  127. package/src/publish.ts +247 -30
  128. package/src/qe-bridge.ts +4 -2
  129. package/src/qe-findings.ts +463 -0
  130. package/src/recap.ts +10 -3
  131. package/src/release-line.ts +32 -0
  132. package/src/round.ts +165 -6
  133. package/src/run-records.ts +282 -2
  134. package/src/score.ts +115 -6
  135. package/src/vector-tier.ts +127 -14
package/README.md CHANGED
@@ -11,6 +11,52 @@ The serial paths in `test/serial-suites.txt` are regenerated from
11
11
  `test/serial-suites-census.test.ts`, which scans test sources for process and timing markers,
12
12
  including `execSync(` and `execFile(`, and fails when the list and census differ.
13
13
 
14
+ ### Full-suite worker ceiling (`CORE_MAX_WORKERS`, `vitest.config.ts`)
15
+
16
+ The root `test` block caps `maxWorkers` at `CORE_MAX_WORKERS` (2, `minWorkers: 1`), so
17
+ `npx vitest run` with no flags is safe by default. **The ceiling lives on the root `test` block,
18
+ not on the `parallel` project's `poolOptions`** — an earlier version of this config set
19
+ `poolOptions.forks.maxForks` on the `parallel` project instead, and that was a false guarantee: a
20
+ fix-round measurement (2026-09-16) compared process names (`node (vitest N)`, polled from
21
+ `/proc/<pid>/cmdline`) over the same 30-file parallel set and saw names `vitest 1`..`vitest 7`
22
+ (14 workers observed) under the project-level `poolOptions`, against never more than
23
+ `vitest 1`/`vitest 2` under either a `--maxWorkers=2` CLI flag or `maxWorkers` on the root `test`
24
+ block. Vitest 3.2.4 simply does not honour `poolOptions.forks.maxForks` set on a project the way it
25
+ honours the root-level knob (or the equivalent CLI flag) — full proof in
26
+ `features/core-suite-memory-ceiling/07_code_changes/change_manifest.md`, section "Фикс-раунд 1".
27
+
28
+ The number itself is not "8 minus a guess" — it is arithmetic from a MEASURED trace (see the
29
+ comment above the constant in `vitest.config.ts`): the memory pressure is NOT the embedding model
30
+ loaded inside the vitest worker (that guess is REFUTED), it is a CHILD process that some tests
31
+ spawn per file (an embedding daemon, or a `dz teach` invocation). On 2026-09-16 three full-suite
32
+ runs were traced with the same instruments, and each number below
33
+ says WHICH run produced it, because two of the three runs did not have a ceiling that actually
34
+ bound:
35
+
36
+ | run | ceiling | tree peak | minimum free | result |
37
+ |---|---|---|---|---|
38
+ | 06:24–06:29 | `--maxWorkers=2` CLI flag (binds) | 5721 MB | 2995 MB | green, 7018 passed, 281 s |
39
+ | 06:46–06:49 | `poolOptions` on the `parallel` project (does NOT bind — effectively unbounded) | 8207 MB | 804 MB | green, but see below |
40
+ | 07:01–07:06 | root `test.maxWorkers` (binds), **no flags** | 5383 MB | 3804 MB | green, 333 files, 7024 passed, 275 s |
41
+
42
+ The last row is the profile of the shipped configuration — the command a person actually types.
43
+ The middle row is the DEFECT being measured, not this configuration, and it is where the
44
+ per-process-class peaks come from: a `vitest` process 2324 MB, an embedding-daemon child spawned by
45
+ a test 2321 MB, a child `dz teach --from-json` 1786 MB, with up to 4 daemon children alive at once
46
+ (3917 MB combined). Those per-class numbers are real, but quoting them as the memory profile of the
47
+ 2-worker run would be a misattribution — a Codex round-2 finding, fixed here.
48
+
49
+ The load-bearing fact stays: the memory is NOT the embedding model inside the vitest worker, it is
50
+ in the CHILD processes the tests spawn, and the ceiling bounds how many worker-plus-child pairs are
51
+ alive together. `CORE_MAX_WORKERS` must not be raised without a fresh trace of the last shape.
52
+
53
+ `vitest.config.ts` also fails LOUD at config-load time if `CORE_MAX_WORKERS` is ever set to
54
+ something other than a positive integer (0, negative, or fractional) — a Codex fix-round finding
55
+ that a ceiling accepting those values is not a ceiling at all.
56
+
57
+ `test/mutation-registry.json`'s `maxWorkers` is kept equal to this same constant
58
+ (`test/suite-worker-ceiling.test.ts` reddens if either drifts from the other).
59
+
14
60
  `findExactLesson(records, text, domain?)` finds the earliest lesson whose trimmed,
15
61
  whitespace-collapsed text matches exactly (case-sensitive), optionally within one metadata domain,
16
62
  and reports whether that existing lesson is quarantined.
@@ -192,6 +238,49 @@ evidence; **high-volume** (`gpt-5.6-luna`) for mechanics. Eco lowers each select
192
238
  Codex tier by one level; these are capability assignments, not measured prices.
193
239
  Claude-primary cells and the cross-family QE rule retain their existing behavior.
194
240
 
241
+ ## Experiment envelope (`feature-adr-envelope.ts`, ADR-001 envelope-before-dispatch)
242
+
243
+ The feature-adr conveyor writes an OUTCOME per run (grade, some tokens) but never used to write the
244
+ DECISION behind it — what kind of task this was, how big, at what priority, which model arms the
245
+ router considered, which one it picked, and who judged it. `buildExperimentEnvelope` assembles that
246
+ as one plain-data object, built exactly ONCE per run (right after the Step-0 router, once `tier` and
247
+ `taskKind` are known, and before Step 1 dispatches anything), then threaded byte-for-byte into every
248
+ place the run reports itself:
249
+
250
+ - every autowritten run-cost ledger row (`.dz/feature-adr/run-cost-ledger.jsonl`, field `envelope`);
251
+ - every captured training pair (`.dz/fa-training/<slug>/<stage>.jsonl`, field `envelope`, alongside
252
+ the narrower legacy `budgetMode` — not instead of it);
253
+ - the round state opened via `dz round open --envelope <json>`, copied into the round's ledger row
254
+ by `closeRound` on `dz round close`.
255
+
256
+ Shape (`ExperimentEnvelope`): `schema:1`, `runId`, `attempt` (integer ≥ 1), `taskKind` (one of
257
+ `feature|bugfix|refactor|tooling|docs|research`), `tier` (`S|M|L|XL`), `priority`
258
+ (`speed|balance|quality|unset`), `treeSha` (40-hex or `null` + `treeShaReason`), `arms` (`{mode:
259
+ string[], stages: {stage: string[]}}` — what the routing tables OFFERED), `chosen` (`{mode, stages:
260
+ {stage: spec}}` — what was actually resolved), `policy` (`{name, version, propensity}`), `evaluator`
261
+ (`{family, model, source: 'planned'|'actual'}`). `validateExperimentEnvelope(value)` returns
262
+ `{ok:true}` or `{ok:false, reason}` naming the FIRST invalid field.
263
+
264
+ **The writer refuses an automated row without one (FR-5, D2).** In `run-records.ts`,
265
+ `decideRecordWrite` for `kind:'ledger'` refuses (`exit 2`) an `auto:true` row that carries no
266
+ `envelope`, and refuses ANY row (auto or manual) whose present `envelope` fails validation. A manual
267
+ row without `auto`/`envelope` is unaffected — the old shape still writes exactly as before (C-3).
268
+ Read it back with `jq '.envelope' .dz/feature-adr/run-cost-ledger.jsonl`.
269
+
270
+ **`args.priority` (FR-4).** A learning-stratum LABEL, one level above `budget`/`deliveryGate` — an
271
+ explicit knob always wins over the preset:
272
+
273
+ | `priority` | `budget` preset | `deliveryGate` |
274
+ |---|---|---|
275
+ | `speed` | `eco` | `false` |
276
+ | `balance` | `normal` | `false` |
277
+ | `quality` | `normal` | `true` |
278
+ | `unset` (default) | whatever `args.budget` says | whatever `args.deliveryGate` says |
279
+
280
+ `PRIORITY_PRESETS` + `resolvePriority(raw)` + `applyPriorityPreset(priority, explicit)` live next to
281
+ `BUDGET_PRESETS` in `feature-adr-routing.ts`; an unknown priority is a startup error naming the valid
282
+ list, never a silent `unset`. Setting `priority` alone (no other routing knob) turns routing on.
283
+
195
284
  ## What it provides
196
285
 
197
286
  ### Evidence-gated companion integrations
@@ -224,7 +313,9 @@ explicit skills-only short circuit. `--no-verify` cannot authorize emission. A C
224
313
  | `no-stubs` | `scanStubs`, `checkNoStubs`, `scannableStubPath`, `STUB_MARKERS`, `STUB_PHRASES`, `STUB_SCAN_EXTENSIONS` | Pure unfinished-stub scanner behind the SOFT `no-stubs` publish rule (backlog 0b403a0106103901, Karpathy-Michaels rule XI): bare markers (`TODO`/`FIXME`/`HACK`/`XXX`/`PLACEHOLDER`) case-SENSITIVE with hard word boundaries (`hackathon`/`todos`/a marker inside a hash never fire; MEASURED: relaxing case doubles this repo's hits and adds only prose) + the `implement later` phrase case-insensitive. SCOPE = the CHANGE-SET (the working-tree `git status --porcelain -uall` diff — `-uall` so a brand-new untracked DIRECTORY is scanned file-by-file instead of collapsing to one invisible `?? newdir/` line; `.gitignore` semantics unchanged), never the whole tree — MEASURED: a tree-wide scan is 32+25 hits of mostly ancient legitimate markers, i.e. noise that gets a gate switched off. Markdown gets PROSE scoping (fenced blocks + backticked spans are QUOTES, not stubs). Waiver-with-REASON only, per line (`no-stubs: <reason>`) or per path (`.dz/guard.json` `stubWaivers`, the feature-adr-setup --guards shape); a reasonless waiver is REFUSED as its own finding and exempts nothing. Self-exemption is STRUCTURAL: every marker in the module and its tests is assembled from string fragments, so the gate's own source scans clean — a tested property, not a path skip. Fail-open on missing evidence (no change fact / ungathered contents ⇒ nothing reported) but never fail-SILENT: skipped scannable files (deleted/oversize/unreadable/beyond the file cap) surface as ONE aggregate `notes` entry in the `GuardResult` + audit record — information that can never move the verdict. KNOWN LIMITS are documented at the top of `no-stubs.ts` instead of implied away (whole-line inline waiver token = layer-4 auditability defence; reason QUALITY not judged; boolean fence model, not CommonMark; git-quoted paths undecoded; TS-monorepo extension allowlist; worktree-not-index reads; exact-string config-waiver paths). Mutation-defended (`no-stubs-bare-marker-fires` observed 10 red, `no-stubs-skipped-note-emitted` observed 2 red) |
225
314
  | `feature-adr-setup` (P3) | `renderGuardsConfig`, `renderGuardsRunner` | Scaffolds deterministic guard tests into a TARGET project: `guards.config.json` + a zero-dependency `check.mjs` runner (loc-cap, secret-scan, frozen-file sha256 pins, waivers-with-reasons) — `dz feature-adr-setup --guards` |
226
315
  | `usage` | `computeUsage`, `TOKEN_WEIGHTS`, `readUsageLimits`, `deriveUsageCalibration` | Read-only Claude usage ESTIMATE behind `dz usage`. Tokens are COST-WEIGHTED input-equivalents (input 1x, cache-write 1.25x / 1h 2x, cache-read 0.1x, output 5x) — a flat sum is 89-99.7% cache-read (MEASURED) and tracks conversation length, not work. Scans subagent transcripts too (`<session>/subagents/*.jsonl`), follows no symlinks, reads only regular files (symlinked FILES and DIRECTORY components alike are skipped), and caps the walk BY RECENCY so a huge history cannot discard current usage. `pct` stays `null` while limits are unconfigured — an unconfigured estimate is never dressed up as a number |
227
- | `cost-ledger` | `deriveCostLedger`, `buildCostLedger`, `verifyCostLedgerReport`, `stageCostAggregates`, `renderCostLedger`, `writeCostLedgerJsonl`, `COST_LEDGER_SCOPE` | Per-stage cost ledger behind `dz usage --by-stage`. A feature-adr run reports ONE number; this joins the workflow's own `stageLabel()` strings to the per-agent transcripts the harness already writes, so a run becomes an itemized receipt. POST-HOC DERIVER, not a writer — no workflow edit, and a KILLED run is still derivable. The invariant: `accounted + unaccounted === runTotal` and `accounted + doubleAttributed === Σ stages`, RAW integer equality (rounding happens exactly once, per sample, at extraction), re-derived from the emitted report by `verifyCostLedgerReport` — the writer clamps, the verifier enforces. A mismatch is a NAMED defect (`Unaccounted`, `DoubleAttributed`, `ForeignSample`, `MissingStageTranscript`, `MalformedRecord`), never a rounding remainder, so `epsilon` defaults to 0. The run total comes from the run's transcript DIRECTORY LISTING, NOT the record's own `totalTokens` — that field is exactly `Σ workflowProgress[].tokens` in 29 of 29 recorded runs (MEASURED), so an invariant against it can never fail. Both sides share ONE estimator with `dz usage` (`weightedTokensOf`). `stageCostAggregates` is a pure feed-forward reader for auto-cost routing that EXCLUDES non-reconciling runs; wiring it into routing is deliberately out of scope. HONEST SCOPE, printed by every surface: local transcript ESTIMATES, not billed amounts — it catches ATTRIBUTION errors, NOT pricing errors; `hasKnownPricing` marks rows priced by the sonnet-class fallback. `INSUFFICIENT_DATA` is a distinct verdict, never collapsed into `BALANCED` |
316
+ | `cost-ledger` | `deriveCostLedger`, `buildCostLedger`, `verifyCostLedgerReport`, `stageCostAggregates`, `renderCostLedger`, `writeCostLedgerJsonl`, `COST_LEDGER_SCOPE` | Per-stage cost ledger behind `dz usage --by-stage`. A feature-adr run reports ONE number; this joins the workflow's own `stageLabel()` strings to the per-agent transcripts the harness already writes, so a run becomes an itemized receipt. POST-HOC DERIVER, not a writer — no workflow edit, and a KILLED run is still derivable. The invariant: `accounted + unaccounted === runTotal` and `accounted + doubleAttributed === Σ stages`, RAW integer equality (rounding happens exactly once, per sample, at extraction), re-derived from the emitted report by `verifyCostLedgerReport` — the writer clamps, the verifier enforces. A mismatch is a NAMED defect (`Unaccounted`, `DoubleAttributed`, `ForeignSample`, `MissingStageTranscript`, `MalformedRecord`), never a rounding remainder, so `epsilon` defaults to 0. The run total comes from the run's transcript DIRECTORY LISTING, NOT the record's own `totalTokens` — that field is exactly `Σ workflowProgress[].tokens` in 29 of 29 recorded runs (MEASURED), so an invariant against it can never fail. Both sides share ONE estimator with `dz usage` (`weightedTokensOf`). `stageCostAggregates` is a pure feed-forward reader for auto-cost routing that EXCLUDES non-reconciling runs (now also `INCOMPLETE_INVENTORY` runs — the `!== 'BALANCED'` gate already excludes it, no second branch to forget); wiring it into routing is deliberately out of scope. HONEST SCOPE, printed by every surface: local transcript ESTIMATES, not billed amounts — it catches ATTRIBUTION errors, NOT pricing errors; `hasKnownPricing` marks rows priced by the sonnet-class fallback. `INSUFFICIENT_DATA` is a distinct verdict, never collapsed into `BALANCED`. **measurement-integrity (ADR-001 D1/D2):** every row also carries `stageCanonical` (the verbatim `stage` classified against `feature-adr-stage-canon.ts`'s one ordered table — see that module below — `'unknown'` when no rule matches, never silently folded into `infra`) plus `attempt`/`attempts` (a label repeated N times in one run is N separate rows, each tagged `attempt: i` of `attempts: N`, instead of one row silently summing them). The report gains `byCanonicalStage` (every canonical stage + `unknown` + `unattributed`, `{tokens, agents, attempts}`) and `reconciliation.orphanTranscripts` (`{count, tokens, ids, method}` — transcripts present in the run directory with NO `workflowProgress[]` entry; `method: 'per-transcript'` is an exact sum over each orphan's own samples, `'count-fallback'` is the best estimate when only ids are known). A run whose ONLY problem is a named orphan (nothing else defective) reports verdict `INCOMPLETE_INVENTORY` — outranks `BALANCED`, outranked by `DEFECT` — never the old `Unaccounted`/`DEFECT` pair that used to swallow the orphan into the generic bucket |
317
+ | `feature-adr-stage-canon` | `CANONICAL_STAGES`, `STAGE_LABEL_RULES`, `canonicalStage` | measurement-integrity ADR-001 D1: the canonical stage taxonomy — 11 pipeline stages (`router`, `requirements`, `research`, `adr`, `ideation`, `ddd`, `architecture`, `plan`, `code`, `qe`, `fleet`) + `infra` for the bookkeeping/plumbing labels around them. `canonicalStage(label)` classifies ONE verbatim `stageLabel()` string against ONE ordered prefix table (first match wins; a `label · model` suffix is matched on the part before ` · `) and returns `{stage, label, known}` — the input label is NEVER rewritten, only classified next to it. An unrecognised label is `{stage:'unknown', known:false}`, never silently `infra`. The completeness fixture (`test/feature-adr-stage-canon.test.ts`) is 47 labels copied verbatim from a live recorded run (`wf_5a7755c7-f92`) — Step 0's assessment counted 48 on the same record; a live reproducer counted 47, and the one-label gap does not change which prefixes are needed. Pure — no filesystem, no clock; the `core-boundary` ratchet pins it at zero `node:fs` imports |
318
+ | `codex-rollouts` | `parseCodexRollout`, `matchCodexRollouts` | measurement-integrity ADR-001 D3: a pure reader for Codex CLI rollout logs (`~/.codex/sessions/YYYY/MM/DD/rollout-<ts>-<uuid>.jsonl`) — 130 of 156 recorded Codex ledger rows carry `tokens: null` even though the spend is sitting on disk, because the pipeline dispatches `codex exec` without an explicit session id. `parseCodexRollout(text, fileName?)` extracts `{id, cwd, model, startedAt, endedAt, totals}` from one file's TEXT (never opens a file itself — the CLI does that); it accepts BOTH the schema Step 0 documented (`type:"token_count"`, `payload.info.total_token_usage`) AND the schema actually observed live on this machine 2026-09-16, `cli_version 0.154.0` (`type:"token_usage_record"`, `payload.usage`; `model` on `turn_context`, not `session_meta`) — a reader that understood only a shape nothing on disk still emits would fail at the exact thing it exists to fix. `matchCodexRollouts(rollouts, {from, to, cwd?, model?})` joins a stage's time window to the rollout that produced its spend by INTERVAL OVERLAP, never "nearest in time" (two reviews back to back would misattribute) — `0` matches is `{status:'none'}`, `1` is `{status:'one', rollout}`, `>1` is `{status:'ambiguous', candidates}`, never a first-pick. Pure — the `core-boundary` ratchet pins it at zero `node:fs` imports |
228
319
  | `compounding` | `mulberry32`, `bootstrapDelta`, `decidePromotion`, `assembleCompoundingReport`, `assembleLessonToRuleFunnel` | Pure learning-loop payoff engine behind `dz compounding`: seeded deterministic bootstrap (conservative nearest-rank lower-95), promotion that refuses non-finite/malformed input and anything under 5 samples per arm, and dz-native measurements (pool write-only ratio, guard trajectory by RATE, replay readiness over unique untruncated prompt events). Its lesson-to-rule funnel reports UTC calendar-month `eligible → attempted → accepted → executions` counts from prospective promotion-run and anchored guard-audit evidence. Zero alone is not a finding: only a non-empty predecessor followed by an empty named successor in three consecutive measured months produces one; unavailable evidence remains `NOT MEASURED` with its reason. Compaction keeps the newest query-bearing rows verbatim and aggregates ONLY the rest (read totals are invariant across compactions). Also reports EVENT-CHAIN health of the evidence logs it computed from (`evidenceLogs` in, `instrumentation.chains` out) — verified / defect kinds / uncovered pre-chain prefix, with no logs handed in producing no line at all rather than a vacuous "clean" |
229
320
  | `event-chain` | `fnv1a32`, `nextChainFields`, `appendChainedLines`, `chainRewrite`, `guardedRewrite`, `verifyEventChain`, `EVENT_CHAIN_SCOPE` | Pure hash-chain over the two learning-evidence logs (`.dz/recall-usage.jsonl`, `.dz/guard-audit.jsonl`): each appended record carries `seq` + `prevHash` (FNV-1a over the previous line AS WRITTEN, so key order cannot make writer and verifier disagree), derived from the LAST LINE ONLY so a per-prompt hook stays O(1). `verifyEventChain` names eight classes — `BrokenLink`, `DuplicateSeq`, `NonMonotonicSeq`, `TornTail`, `DoubleCounted`, `LedgerImbalance`, `MalformedLedger`, `ClaimInterrupted`. The last four exist because a rewriter must not be able to certify itself: the compaction ledger's arithmetic (`Σ weight + dropped === source`, `dropped ∈ [0, source]`) is enforced with NO clamps in the verifier (the clamp belongs to the writer), a damaged ledger line is a defect rather than a silently-disabled check, and a claim that never reached its `throughSeq` — because the segment restarted or the file ended — is reported instead of escaping through the discontinuity. `guardedRewrite` is the concurrency guard for any whole-file rewrite: exclusive lock, plus a re-read of the live file after computing the new text and BEFORE the rename, so a concurrent append aborts the attempt and is folded into a bounded retry rather than overwritten (it narrows the read→rename window; it cannot close it, and says so). Records written before chaining existed stay LEGAL and are counted as an uncovered `preChainPrefix`; an unreadable tail never blocks a write (fresh MARKED segment — an unreadable tail WINDOW is distinguished from an empty file — and the appender starts on a new line so one torn write cannot eat the next record); an unmarked restart is reported once and then re-anchored, so one incident is one defect instead of a cascade. HONEST SCOPE, carried in every result and printed by every surface: corruption detection for our own bugs — FNV-1a is not cryptography, its collisions are constructible, the threat model has no adversary, and a regression test fails if the module regrows tamper-proofing vocabulary |
230
321
  | `feature-adr-checkpoints` | `checkpointInputHash`, `decideCheckpointResume`, `parseCheckpointRead`, `serializeCheckpoint`, `fnv1a64`, `CKPT_SCHEMA_VERSION`, `STAGE_ARTIFACTS`, `DESIGN_SUBSTAGES`, `designStageKey`, `decideDesignFanResume`, `parseArtifactProbe` | The PURE half of feature-adr's durable per-stage checkpoints (`features/<slug>/.fa-state/checkpoints.jsonl`): a dead L/XL run — or the standard stop-after-plan re-invoke — resumes completed stages instead of re-spending them. Resume = INPUT-identity (64-bit salted FNV over a schema-versioned JSON tuple incl. upstream stage results) + presence of EVERY tier-required artifact; a stale-input hash never resumes in ANY mode (`force` relaxes only the artifact probe — the tested load-bearing property). HONEST SCOPE: it does NOT fingerprint the working tree (a crash-resume legitimately sees the dead run's uncommitted writes) — after manual edits use `resume:'never'` and re-QE. Null results are never persisted or resumable; a stage-identifiable corrupt record ERASES its older entry (last-wins holds for corruption too); the code stage's persist predicate is now an ALLOWLIST (`codeCheckpointPersistAllowed` + `codeStageResultShapeValid`, ADR-003 Condition 3): ONLY `landed` on a barrier-required run and `synchronous` on a non-barrier run may be checkpointed — inconclusive, not-landed, garbage and a mislabeled `synchronous` are all refused, and the `landing-v2` hash token makes every pre-protocol code checkpoint stale. **Since 0.5.3 the design fan is checkpointed PER SIBLING** (`design:requirements` / `adr` / `qcsd` / `architecture` via `designStageKey`), so one dead agent no longer discards three finished siblings, and a fix to one step's instructions invalidates that step alone. What may be CONSUMED is judged separately from what may be WRITTEN: `decideDesignFanResume` returns a named reason (`substage-missing` / `artifact-missing` / `probe-not-established` / `ok`) and the workflow REFUSES at the Step-5/6 boundary rather than planning off a partial design. The artifact half is judged against a POST-RUN probe that never prints filenames (`[ -f <exact rel> ]` per required artifact) — a listing is a list of filenames, and a file whose NAME ends in a newline was measured satisfying the requirement for the real file. `parseArtifactProbe` then validates the WHOLE transcript, because the probe is relayed by a model: an agent that merely NARRATES the expected output emits the token byte-identically. Inconclusive is never a pass. The workflow mirrors this inline (wiring-guarded); RU: чекпоинт после каждой дорогой стадии — упавший ран возобновляется, а не пере-тратит завершённое; веер проектирования — по каждому участнику отдельно, а неполный веер получает отказ, а не запись в лог |
@@ -607,6 +698,13 @@ instead of only ever checking the plain project path. The daemon also now prints
607
698
  (path N bytes)` and exits non-zero, never a silent "ready" for a socket that was never created
608
699
  (`APPLY_LEG_VERSION` bumped 3→4 for this and the resolver change).
609
700
 
701
+ `probeApplyLeg` (and `probeHookLiveness` beneath it) now report **`groupKillAttempted`** — whether the
702
+ liveness probe actually issued its process-group `SIGKILL` (attempted after EVERY probe outcome — timeout, exit, spawn error; ESRCH still counts as attempted) — as an OBSERVED field,
703
+ set in the `finally` right after the kill attempt and present only on branches that reached a spawn.
704
+ A test can therefore assert "the kill was attempted" separately from "the child is dead" instead of
705
+ inferring the first from the source (feature `full-suite-flake-fixes-3`, Codex HIGH-1; the mutant
706
+ that never reports the attempt is registry entry `probe-group-kill-attempted`).
707
+
610
708
  ### One recall engine for hook and CLI (`hook-recall-hybrid-parity`, ADR-001, `APPLY_LEG_VERSION` 4→5)
611
709
 
612
710
  The daemon's `op: recall` handler used to run its own brute-force cosine loop over the in-memory
@@ -677,9 +775,30 @@ find module` from a foreign session, swallowed by `2>/dev/null || true`.
677
775
  - **Limits, named plainly.** One store per install: from a foreign project's session the hook injects
678
776
  the INSTALL ROOT's lessons, not the session project's — that is the requested behavior (a
679
777
  per-project store alongside a user-level one is a separate feature). The Codex host's own hook
680
- (`codex-hooks-assets.ts:232`/`:373`) resolves its root from `payload.cwd || PWD || cwd()` — the
681
- SAME class of weakness — and is deliberately left unfixed here (named, not silently patched); see
682
- that file's own comments and the project backlog.
778
+ (`codex-hooks-assets.ts:232`/`:373`) resolves its root from `payload.cwd || PWD || cwd()` via a
779
+ single shared `resolveHookRoot(payload)` and now NAMES a not-found root on stderr instead of a
780
+ silent early return (`codex-hook-root-provenance`, below); a live probe (T1, codex-cli 0.154.0,
781
+ re-run correctly in fix-round 1 — see that feature's own change manifest) found `payload.cwd`
782
+ always present and equal to `PWD`/`cwd()` across 3 scenarios / 6 captures, so no explicit-override
783
+ knob was added — a SCOPED finding, not a claim that no producer could ever send a different value.
784
+
785
+ #### Codex hook root provenance (`codex-hook-root-provenance`)
786
+
787
+ Both Codex hooks (`dz-codex-veto`, `dz-codex-recall`) share ONE `resolveHookRoot(payload)` instead of
788
+ two copies of the `payload.cwd || PWD || cwd()` fallback, and a `root === null` early return now
789
+ prints one line to stderr — `[dz-codex-<hook>] skipped reason=no-project-root start=<startDir>
790
+ (<source>)` — so a hook that never found a project is distinguishable from one that is silently dead;
791
+ a found root stays silent unless `DZ_CODEX_HOOK_DEBUG` is set. Interpolated paths are escaped (C0
792
+ control range + DEL) before printing, so a hostile `cwd` cannot turn the "ONE line" promise into
793
+ several — the tradeoff is that the line still discloses the absolute directory the hook was asked
794
+ about (which can embed a username or a project name) to stderr, accepted because it is a diagnostic
795
+ for the person running the hook, not a return value. A live probe (three real `codex exec` sessions —
796
+ project root, a nested subdirectory, a directory with no `.dz` anywhere — each with BOTH hook events,
797
+ `PreToolUse` and `UserPromptSubmit`, captured separately; corrected in fix-round 1 after the original
798
+ reproducer's unexported `BASE` measured the wrong file) found `payload.cwd` always correct in every
799
+ one of the 6 captures, so there is no `DZ_PROJECT_ROOT`-style override — stated as "not observed on
800
+ codex-cli 0.154.0 across 3 scenarios / 6 captures", not as a claim that the "hook read the wrong
801
+ project" class cannot exist elsewhere.
683
802
 
684
803
  ### Store write-generation counter (`store-generation-counter`, `agentdb-index.ts`/`vector-tier.ts`)
685
804
 
@@ -804,6 +923,16 @@ deeper.
804
923
  flake, not a correctness defect; the daemon's own `HOOK_RECALL_BUDGET_MS` is independently
805
924
  widenable, and `probeApplyLeg`'s `env` option exists for exactly this in tests. A real single
806
925
  `dz doctor`/`dz parity` invocation never contends with 100+ concurrent test files.
926
+ - **Every probe beacon now carries an owner and a TTL (feature `apply-leg-daemon-hygiene`, FR-3).**
927
+ The pre-probe scavenger used to delete EVERY `apply-leg-probe`-domain record unconditionally,
928
+ which was safe against a probe killed mid-flight but WRONG the moment two probes from two
929
+ different sessions run against the same store concurrently — the second probe's scavenge could
930
+ delete the first probe's still-in-flight beacon, producing a false-red `dz doctor`/`dz parity`
931
+ parity check with no real defect behind it. Each beacon now embeds `probe-owner=<pid>:<startedMs>`
932
+ in its own text (no schema change), and `scavengeStaleProbeBeacons` (exported) removes ONLY a
933
+ beacon whose owner is dead (`process.kill(pid, 0)` ⇒ ESRCH) or older than 60 s — a live, in-budget
934
+ beacon from a different concurrent probe is left untouched. A scavenge failure is now a named fact
935
+ (`ApplyLegProbeResult.scavengeError`), never a swallowed exception.
807
936
 
808
937
  ## Run a plan without the Claude host
809
938
 
@@ -864,6 +993,15 @@ Four changes, each replacing a statement derived from configuration with one der
864
993
  it REPLACED the lexical one, and exact matches on rare identifiers vanished. The lexical top-1 now
865
994
  keeps a reserved seat (taken from the weakest non-`both` place, never from a hit both legs found),
866
995
  and ties break by evidence rather than by the id alphabet.
996
+ - **Recall's ordering contract is score DESC → evidence → `dzId` ASC everywhere, via one exported
997
+ `compareHybridHits`** (`feature recall-parity-tie-break`) — `mergeHybridHits`, `dampQuarantined`,
998
+ and `enhance()`'s reinforcement/bandit re-rank (via `orderHitsForReRank`, its pre-sort) all share
999
+ it, and the comparator is total even for `NaN`/±Infinity keys (a `NaN` sorts deterministically
1000
+ last, never a coincidental tie). The daemon's own `recall-hit` exposure write stays
1001
+ fire-and-forget — awaiting it cost +55–131 ms per recall against a 500 ms hook budget without
1002
+ fully closing the race anyway; parity between the daemon and `dz recall` is instead proven with a
1003
+ byte-level store snapshot taken before the comparison, so no write can reach the read it's
1004
+ compared against.
867
1005
  - **`HybridRecall` carries `semanticCandidates` and `semanticRanked`** — what the engine returned and
868
1006
  what actually entered the merge, so a caller can tell "the tier is empty" from "the tier returned
869
1007
  only stale ids", which need different fixes.
@@ -1015,12 +1153,17 @@ the second case, so a broken store no longer looks like plain "fewer lessons".
1015
1153
 
1016
1154
  ## Embedder cache (`agentdb-index.ts` — `resolveAgentdbEmbedder`)
1017
1155
 
1018
- `resolveAgentdbEmbedder(projectRoot)` resolves agentdb's `EmbeddingService` and stands up the
1019
- `@huggingface/transformers` pipeline behind it — MEASURED 2026-09-14 at 2-3.6s per call
1020
- (`features/agentdb-embedder-cache/00_complexity_assessment.md`), because the model+dim for a
1021
- project never changes within one process. It is now cached at module scope, keyed by
1022
- `${agentdbDir}|${model}|${dim}` (so a `DZ_EMBED_MODEL`/`.dz/config.json` change — a different
1023
- `resolveEmbedModel` result — gets its own entry rather than reusing a stale pipeline):
1156
+ `resolveAgentdbEmbedder(projectRoot, dbPath?)` builds the `@huggingface/transformers` pipeline
1157
+ **directly** — MEASURED 2026-09-14 at 2-3.6s per call
1158
+ (`features/agentdb-embedder-cache/00_complexity_assessment.md`), because the model+dim+dtype for a
1159
+ project never changes within one process. Feature `embed-daemon-memory` (ADR-001 D1) moved this off
1160
+ agentdb's own `EmbeddingService` class: the same `pipeline(text, { pooling: 'mean', normalize: true })`
1161
+ call `EmbeddingService.embed` makes internally, called straight from core, so a process building
1162
+ BOTH the write path (`indexPatternsToAgentdb`) and the read path (`searchAgentdbPatterns`) — or the
1163
+ embed daemon, when it can reach this same cache (see below) — stands up **one** pipeline instance,
1164
+ not one per code path. Cached at module scope, keyed by `${agentdbDir}|${model}|${dim}|${dtype}` (so
1165
+ a `DZ_EMBED_MODEL`/`.dz/config.json` change — a different `resolveEmbedModel` result, or a different
1166
+ dtype — gets its own entry rather than reusing a stale pipeline):
1024
1167
 
1025
1168
  - Repeat calls for the same key return the **same object**, not a re-initialized one — MEASURED
1026
1169
  2026-09-14 (`test/agentdb-embedder-cache.test.ts`, live model): cold call `2117ms`, warm call
@@ -1034,6 +1177,41 @@ project never changes within one process. It is now cached at module scope, keye
1034
1177
  tests and future warm-start use only — `vector-tier.ts`/`backlog.ts` call sites are unaffected.
1035
1178
  - `getAgentdbEmbedderCacheStats()` returns `{ entries, initializations }` (`entries` = currently
1036
1179
  cached successful pipelines, `initializations` = pipelines actually started since the last reset).
1180
+ - **COMPAT FALLBACK** (dtype `fp32` only): when `@huggingface/transformers`/`@xenova/transformers`
1181
+ cannot be resolved directly from the project (nor via `agentdb`'s own declared dependency), this
1182
+ falls back to agentdb's `EmbeddingService` — the pre-T2 behaviour — so a project whose only route
1183
+ to an embedder is through agentdb's own installed copy still works. A requested `dtype: 'q8'` never
1184
+ takes this fallback (NFR-4): a quantized store with no reachable transformers install is a hard
1185
+ `{error}` naming the model/dtype, never a silent fp32 downgrade.
1186
+
1187
+ ### dtype: `fp32` vs `q8` (`memory.embed.dtype`, ADR-001 D2)
1188
+
1189
+ A single fp32 instance of the default model (`Xenova/paraphrase-multilingual-MiniLM-L12-v2`) costs
1190
+ ~1.2 GB RSS once warm (816 MB immediately, ~1219 MB 1-2s after the first embed — a second, native
1191
+ weight read inside `onnxruntime`, library behaviour, not something this package controls). The
1192
+ quantized `q8` variant (`{ dtype: 'q8' }` at pipeline construction, needs `model_quantized.onnx` in
1193
+ the transformers cache) costs ~0.55 GB — roughly half — at a MEASURED (2026-09-16, 14-lesson RU/EN
1194
+ fixture, `features/embed-daemon-memory/07_code_changes/measure-q8-parity.mjs`) cosine parity of
1195
+ min=0.9900/mean=0.9929 against fp32 for the SAME text, and top-1 query agreement 5/5.
1196
+
1197
+ - `memory.embed.dtype` in `.dz/config.json` (`'fp32'` default, `'q8'` opt-in) selects the dtype for a
1198
+ **new** index or a `dz vector reindex` — it does **not** retroactively change an existing store.
1199
+ - The **store's manifest** (`<dbFile>.embed-manifest.json`) records the dtype it was actually built
1200
+ with; every query is embedded with the **manifest's** dtype (`resolveStoreEmbedDtype`), never
1201
+ blindly with the configured one — a manifest with no `dtype` field (every store written before
1202
+ this feature) reads as `fp32`, so existing stores are unaffected.
1203
+ - `guardEmbedSpace` now checks dtype the same way it already checks model/dim: a store built at one
1204
+ dtype and configured for the other is refused with `embedding dtype mismatch: index built with
1205
+ <m>, configured <c>; run <reindexHint>` — never a silent mixed-dtype read (store fp32, query q8),
1206
+ which the ADR names as the concrete risk a default flip would have created.
1207
+ - Switching a store to `q8` is exactly `dz vector reindex` after setting `memory.embed.dtype: 'q8'`
1208
+ — the ONLY point the dtype actually changes; an ordinary incremental index always embeds new rows
1209
+ in the store's EXISTING dtype, never the config's, so two partial writes can never leave one store
1210
+ split across two embedding spaces.
1211
+ - `q8` needs `model_quantized.onnx` already in the transformers cache — a fresh install with no
1212
+ cached weights would need network access on first use; confirm the file is present under the
1213
+ transformers cache dir (or that the machine has network access) before flipping `memory.embed.dtype`
1214
+ to `q8` and running `dz vector reindex`.
1037
1215
 
1038
1216
  ## Consistent pre-reindex snapshot + rollback (`agentdb-snapshot.ts`)
1039
1217
 
@@ -1230,8 +1408,115 @@ had NO mutual exclusion, and only the 10-minute grace period above stood between
1230
1408
  `rotatePreReindexSnapshots`, which always takes the snapshot lock. Exporting the unlocked primitive
1231
1409
  would hand outside callers a way to rotate with no mutual exclusion at all.
1232
1410
 
1411
+ ## Findings ledger + machine-readable QE verdict (`qe-findings.ts`, ADR-001 `qe-findings-record`)
1412
+
1413
+ `readQeGrade` (`score.ts`) used to read only PROSE — a report fixed after a `Grade: C` round-1
1414
+ review, ending `Grade: B`, read back as `ambiguous`, and a report whose only "Grade" mention was a
1415
+ stray round-1 line read as a confidently-WRONG `unique C` even when its final verdict was `B`
1416
+ (MEASURED, `00_complexity_assessment.md`: 41 of 100 feature reports came back ambiguous; `dz score`
1417
+ returned `C` for a report whose stated outcome was `B`). This feature adds ONE machine-readable
1418
+ surface on top of the prose, never replacing it:
1419
+
1420
+ - **A verdict line**: `QE-VERDICT: <A|A-|A+|B|B+|B-|C|C+|C-|D>` (`QE_VERDICT_RE`, `qe-findings.ts`).
1421
+ `readQeGrade` checks it FIRST: exactly one → `{status:'unique', source:'verdict-line'}`; more than
1422
+ one → `{status:'ambiguous', source:'verdict-line'}` (never "last wins", even when both name the
1423
+ same grade — two lines is a fact about the report); zero → the pre-existing prose scan runs
1424
+ exactly as before, tagged `source:'prose'` (or `'none'`). `GradeReading` gained the `source` field;
1425
+ every prior caller of `readQeGrade`/`extractQeGrade` is unaffected (NFR-1).
1426
+ - **A Findings ledger table** under the exact header `QE_FINDINGS_HEADER` = `| Finding | Severity |
1427
+ Status | Round | Author | Title |`. `parseQeFindings(md)` returns `{status:'absent'}` when no such
1428
+ table exists (406 pre-existing reports; the common case, and the ONLY case for anything written
1429
+ before this feature), or `{status:'present', hollow, rows, refused, summary}`. Three closed
1430
+ dictionaries — Severity `BLOCKER|CRITICAL|HIGH|MEDIUM|LOW|INFO`, Status
1431
+ `confirmed|fixed|partial|refuted|named-limit|open`, Author `codex|claude|lead` — plus Round (an
1432
+ integer ≥ 1) and a whitespace-free Finding id. **A row outside any dictionary is REFUSED
1433
+ (`{line, text, reason}`), never coerced to the nearest known value** — coercion would make the
1434
+ resulting severity/status tally unprovable (ADR-001 D2). **A second table is refused WHOLE**, its
1435
+ header the anchor, reason `duplicate table` — its rows are never parsed individually. **A
1436
+ header-only table is `hollow: true`** — worse than no table at all (ADR-001 D3, the same principle
1437
+ `readMutationEvidence`'s `present-unproven` already applies to the mutation-gate table).
1438
+ - `qe-findings.ts` is PURE (no `node:fs`) — file reads stay in the CLI, guarded by the same
1439
+ `core-boundary.ts` IO ratchet every other core module answers to.
1440
+ - `scoreRun` (`score.ts`) now returns `gradeSource` and `findings` (a lighter `{status, hollow?,
1441
+ summary?, refused?}` projection of `parseQeFindings`'s full result) — both ADDITIVE, the same
1442
+ optional-field discipline `mutationEvidence` already uses. `renderScorecard`/`renderFindingsLine`
1443
+ print a one-line findings summary (`findings: 3 HIGH / 2 MEDIUM; 1 refused (line 84: severity
1444
+ "Major" not in dictionary)`) only when a table exists — the common case (no table) stays silent.
1445
+ - `dz feature-adr-record --kind ledger --stage full` enriches the row with `findings`/`gradeSource`
1446
+ computed by this parser over the row's own `features/<slug>/08_qe_report.md` (fill-only-null,
1447
+ best-effort — a missing/unreadable report never blocks the write). See the CLI README for the
1448
+ producer-side prompt wiring (Step 8's `QE-VERDICT:` + table instruction) and the six September
1449
+ reports hand-marked from their own prose as the real-corpus proof (`qe-findings-corpus.test.ts`).
1450
+
1233
1451
  ## Status
1234
1452
 
1453
+ `next` — staged, not yet versioned or published. Feature `measurement-integrity` (ADR-001, tier M): five
1454
+ measurement holes in the SDD pipeline get an explicit status instead of a convenient number. **D1/D2** land in
1455
+ `cost-ledger.ts` and the new `feature-adr-stage-canon.ts` (see the module table above) — canonical stage
1456
+ taxonomy + `INCOMPLETE_INVENTORY`/`orphanTranscripts`. **D3** is the new `codex-rollouts.ts` (module table
1457
+ above) — a pure Codex rollout-log reader. **D3/D4** land in `run-records.ts`: `decideRecordWrite` gains an
1458
+ OPT-IN `enrich?: LedgerEnrichInput` (`{rollouts?, window?, cwd?, prices?}`) — a ledger row with a codex-family
1459
+ `coder`/`reviewer` and `tokens: null`, given a time window, is enriched via `matchCodexRollouts`:
1460
+ `status:'one'` fills `tokens`/`minutes`/`tokensSource:'codex-rollout'`/`rolloutId`; `'none'`/`'ambiguous'`
1461
+ stamp `tokensSource:'codex-rollout:'+status` with NO number (NFR-3 — never a guess); no `window` at all
1462
+ stamps `tokensSource:'unavailable'` rather than attempting nothing silently. Independently, ANY ledger row
1463
+ given a `prices` table gets a `prices: {snapshotAt, table: {model: {prompt, completion, cachedInput}},
1464
+ unknown?: [model,…]}` snapshot for every model it names (`coder`/`reviewer`/`envelope.chosen.stages`) — a
1465
+ longest-prefix match against the CALLER'S OWN table (never `cost-scoring.ts`'s live constant), so a future
1466
+ repricing never rewrites a historical row (ADR-001 D4). Omitting `enrich` entirely is byte-identical to
1467
+ before this feature. **D5** lands in `round.ts`: `closeRound` gains `grade?`/`reviewSidecar?
1468
+ (RoundReviewSidecar: {gradedBy, elapsedMs, grade?})`. `grade` is now MANDATORY for `outcome ∈
1469
+ {shipped,refuted}` (refusal exit 2, `'grade required for a finished review'`) and a warned-and-DROPPED no-op
1470
+ for `blocked|abandoned` (the close still succeeds; `result.warnings` names why nothing was written).
1471
+ `RoundLedgerRow.grade` is `string | null` (was always `null`). When `--reviewer` is absent, `reviewer` /
1472
+ `reviewMinutes` (`elapsedMs / 60000`, 1 decimal) / `reviewSource:'qe-bridge'` are filled from the sidecar; a
1473
+ `--grade` that disagrees with the sidecar's OWN grade is refused naming both. The CLI (`dz round close
1474
+ --grade`) reads the sidecar from `features/<slug>/.fa-state/qe-bridge/signoff-*.json`, latest by `emittedAt`
1475
+ — `round.ts` itself opens no file. `dz feature-adr-record --kind ledger` grows `--window-from/--window-to`
1476
+ + `--codex-sessions <dir>` (default `~/.codex/sessions`) to drive the FR-5 enrichment; every ledger write
1477
+ always carries the FR-6 price snapshot.
1478
+
1479
+ `plan-inherits-requirements`: the pure halves of the
1480
+ feature-adr plan-repair round now live here and are body-pinned against the workflow's inline copies by the
1481
+ drift guard — `shellQuote` (POSIX single-quote escaping), `planBackupCmd` / `planRestoreCmd` /
1482
+ `planArchiveBackupCmd` (backup before the ONE repair round, proven restore on rejection, archive into
1483
+ `.fa-state/` on acceptance), `planSnapshotCmd` (byte length, POSIX `cksum`, the `EXPECTED_CODE_TARGETS` lines
1484
+ and the task-heading lines), `snapshotBlock` / `snapshotNumber` (whole-line markers, first start to last end —
1485
+ a plan line spelling a marker lands inside its block) and `parsePlanSnapshot` (null = the probe never completed;
1486
+ the caller rejects on null, never reads it as "nothing to compare"). `planCompletenessGateCmd` gained
1487
+ `opts.requireRequirements`, which emits `--require-requirements` so the K2 gate's new C8 (every id declared in
1488
+ `01_requirements.md` is referenced by the plan) fails per id instead of warning with a count.
1489
+ Also in this staged release (feature `coder-reads-and-recall`): the decision-recall kind union gained
1490
+ `'code-implementation'` (stage `step-7`, bandit context `feature-adr-decision-code-implementation`) so the
1491
+ Step-7 coder receives the same ≤3-lesson recall block the Step-6 planner already gets, bundled into the code
1492
+ stage's checkpoint composite exactly like `planComposite` — a resumed stage restores the recalled prompt for
1493
+ the training-pair capture instead of re-spending recall. Measured motive: after the by-name input directive,
1494
+ Claude coders opened `01_requirements.md` in 5 of 7 runs (39 % before), while 37 of 48 post-directive coders
1495
+ were Codex, whose file reads are invisible to the transcript instrument.
1496
+
1497
+ `0.8.37` — this release (night 16→17.09). Five changes live in this package, each through the full pipeline with a
1498
+ cross-family Codex review: **qe-findings** — every Step-8 report now carries one machine-readable `QE-VERDICT:` line
1499
+ and a `## Findings ledger` table in a closed vocabulary (`parseQeFindings`, `readQeGrade` with its source; masks for
1500
+ fenced code, blockquotes incl. CommonMark lazy continuation, HTML comments/blockquotes; a near-miss table is refused
1501
+ loudly, never read as absent); **cross-family-control** — `diffFamilyFindings`/`aggregateByFamily` for `dz
1502
+ control-review`: two independent reviews over one tree, automatic pairs are CANDIDATES (maximum-cardinality
1503
+ matching), confirmed overlap only by adjudication, incomplete rows excluded from measured figures;
1504
+ **measurement-integrity** — canonical stages, `INCOMPLETE_INVENTORY`, per-turn Codex rollout deltas, price
1505
+ snapshots, `round close --grade` mandatory for shipped/refuted; **experiment-envelope** — every automatic ledger row
1506
+ carries `envelope.{taskKind,priority,arms,chosen,evaluator}` (verified live: 1 of 1 new auto rows); **qe-bridge** —
1507
+ the reviewer prompt demands ONE grade letter with no +/− suffix (a live `B-` was refused as no-grade-marker).
1508
+
1509
+ `0.8.36` — (night 15→16.09). Three changes live in this package: `recallHybrid` orders equal-scoring
1510
+ hits through ONE comparator (score desc → evidence rank → `dzId` asc, NaN last), so the embed daemon and
1511
+ `dz recall` agree; the apply-leg test helpers prove a daemon stop by OBSERVING `/proc` until nothing serves the
1512
+ root and gate every SIGKILL on a freshly-read identity plus containment under the test root (three-valued —
1513
+ "could not read" is never "does not match"), while the probe beacon carries its own owner, raw start ticks and
1514
+ expiry; and both generated Codex hooks resolve the project root through one `resolveHookRoot`, with
1515
+ `reportRootProvenance` naming the source and start directory on stderr, silent on the success path and with
1516
+ control characters escaped. The environment override originally planned for the hooks was dropped after a live
1517
+ measurement showed the payload's `cwd` present and correct in all six captures (3 scenarios × 2 hook events,
1518
+ codex-cli 0.154.0).
1519
+
1235
1520
  `dz guard check --op publish` now warns when either release line disagrees with the core/CLI package versions, and a registry-confirmed live core or CLI publish synchronizes the first such line in both release READMEs: each README is rewritten atomically; the pair is not one transaction (dry-run and bump-only never write them).
1236
1521
 
1237
1522
  `0.8.12` — **staged, not published.** The `/feature-adr` phase panel + per-phase ledger telemetry,
@@ -1517,6 +1802,46 @@ comment claimed historical notes were safe; they were, except for the one releas
1517
1802
  cites most — the one it supersedes. A backticked version opening a line before a dash is now treated
1518
1803
  as an entry and left alone; footers, badges, install examples and pins still move.
1519
1804
 
1805
+ **Feature publish-readme-stamp-scope (2026-09-15, fix-round 1 2026-09-15): the sync outside the
1806
+ changelog region is a real POSITIVE ALLOWLIST — not a denylist, and not a denylist that calls itself
1807
+ one.** MEASURED 2026-09-15: a live `dz publish --yes` rewrote five historical lines — a second `##
1808
+ Status` region's own changelog entry (`changelogRegion` protected only the FIRST run, so a `memory`
1809
+ README's later `0.2.21` entry sat bare) and four prose CITATIONS of the outgoing version as a past
1810
+ fact (`MEASURED on 0.8.25`, two `Previous release (vA / v0.8.25)` parentheticals, `on 0.8.10 and
1811
+ 0.8.25 alike`). The FIRST fix (same day) replaced that with a denylist of exactly those three phrase
1812
+ shapes (`isCitationContext`) — narrower than the incident, but still a denylist: any FOURTH prose
1813
+ shape citing the outgoing version ("since X", "measured against X", "X behaviour", a bare "X" in a
1814
+ sentence) would have rewritten by default until someone thought to deny it too, and the README
1815
+ documented it as an "allowlist" while the code rewrote by default — a cross-model review caught both.
1816
+ `planReadmeVersionSync(text, old, new)` now inverts the default: outside a changelog region, a token
1817
+ rewrites ONLY when `isAllowlistedRewriteContext` recognises one of six shapes — the lock-step feature
1818
+ this sync exists for, and nothing beyond it:
1819
+
1820
+ 1. a release-line token — `` `harness-core vX` · `harness-cli vY` `` and any generalised
1821
+ `` `<name> vX` `` on the same line, including a trailing `` · `memory vZ` `` segment
1822
+ (`release-line.ts` `isReleaseLineToken`/`GENERIC_RELEASE_TOKEN_RE`, unchanged since the earlier fix).
1823
+ 2. a current-release FOOTER prefix — `Status:`/`Version:`/`Current release:`/`Current status:`/
1824
+ `Released as` (case-insensitive, optional leading `**`/`-`), POSITION-aware: only the token
1825
+ immediately after the label is allowed, so a footer sentence that also cites an unrelated older
1826
+ release later in the same line (`Current release: X. (Previous release (vA / vB) …)`) allows the
1827
+ first token and still protects the second.
1828
+ 3. an install/dependency-pin context — the token immediately follows `@` (`npm i
1829
+ @dzhechkov/harness-core@X`), or sits in a JSON-pin shape `"<package-name>": "X"`.
1830
+ 4. the `dz publish: tarball <name>@X sha256:…` example line.
1831
+ 5. a `<!-- dz:version -->` marker on the line — forces the rewrite regardless of EVERY other
1832
+ protection, including the changelog region (the author's explicit override, AC-3).
1833
+ 6. a shields.io-style badge URL segment — `badge/npm-vX-…` / `badge/version-X-…`.
1834
+
1835
+ Every OTHER shape — whatever prose it is written in, today or in the future — is HISTORY by default,
1836
+ the same as a changelog entry. `syncReadmeVersion` stays a thin, atomic-write wrapper around the
1837
+ plan, returning exactly what it always returned (the pre-sync text, or `undefined` when nothing
1838
+ moved) — every existing caller is byte-compatible. The plan itself is never silent, and locates BOTH
1839
+ sides of its report: `dz publish` prints `readme sync <pkg>: would rewrite N line(s) (L…); M version
1840
+ token(s) kept as history (L…)` on a dry run and `rewrote N line(s) …` on a live publish — attached on
1841
+ every publish path (the main live publish, `--bump-only`, and the packed-transport batch), and
1842
+ `--json` carries the same `readmeSync` summary (`rewrittenLines`, `lines`, `skippedHistorical`,
1843
+ `historyLines`) per package.
1844
+
1520
1845
  Also: skill-enrichment ownership is anchored at the skill dir rather than searched across the whole
1521
1846
  absolute path, and enrichment is excluded from canonical SELECTION as well as from the destination
1522
1847
  set — otherwise a `--auto` canonical could propagate one target's metadata into every copy.
@@ -13,6 +13,7 @@
13
13
  */
14
14
  import { type SnapshotRotationReport } from './agentdb-snapshot-rotation.js';
15
15
  import { type SnapshotMethod } from './agentdb-snapshot.js';
16
+ import { type EmbedDtype } from './embedding-config.js';
16
17
  /** One record to index. `text` is stored as `approach` AND embedded (`${taskType}: ${text}`). */
17
18
  export interface AgentdbRow {
18
19
  readonly taskType: string;
@@ -26,7 +27,15 @@ export interface AgentdbRow {
26
27
  }
27
28
  /** Outcome of {@link indexPatternsToAgentdb}. `generationBumped`/`generationReason` are present only
28
29
  * when a store write actually happened (`indexed > 0`) — FR-4: a failed counter write NEVER fails
29
- * the indexing call itself, it is only reported so a caller (`dz doctor`, telemetry) can see it. */
30
+ * the indexing call itself, it is only reported so a caller (`dz doctor`, telemetry) can see it.
31
+ *
32
+ * Fix-round 1 (CRITICAL, item 1a): `indexed`/`generationBumped`/`generationReason` and `error` are
33
+ * NOT mutually exclusive. A failure AFTER the row commit (today, only `writeEmbedManifest` throwing)
34
+ * reports the REAL `indexed` count and the REAL bump outcome alongside `error` — it never collapses
35
+ * back to `{indexed: 0, error}` once rows are already on disk. Collapsing to `indexed: 0` after a
36
+ * real commit was the CRITICAL finding: a caller reading `indexed === 0` as "nothing happened" would
37
+ * skip its own rescue-bump logic even though the store had genuinely changed — an under-bump C-1
38
+ * forbids. */
30
39
  export interface AgentdbIndexResult {
31
40
  readonly indexed: number;
32
41
  readonly error?: string | undefined;
@@ -72,7 +81,17 @@ export declare function ensureAgentdbSchema(projectRoot: string, dbPath?: string
72
81
  * to decide validity, only to convert an already-validated string.
73
82
  */
74
83
  export declare function readStoreGeneration(projectRoot: string, dbPath?: string): number;
75
- export declare function bumpStoreGeneration(projectRoot: string, dbPath?: string): {
84
+ /** Fix-round 1, item 5: `String(err)` itself can throw if `err` carries a poisoned `toString` (or
85
+ * `Error.prototype.message` getter). `bumpStoreGeneration`'s "never throws" contract (FR-4) is
86
+ * ABSOLUTE, so every place in this function that turns a caught error into a string goes through
87
+ * this ONE protected helper — never a bare `err instanceof Error ? err.message : String(err)`.
88
+ * Exported test-only (same convention as {@link needsRescueBump}/{@link resetAgentdbEmbedderCache}).
89
+ */
90
+ export declare function safeErrorMessage(err: unknown): string;
91
+ export declare function bumpStoreGeneration(projectRoot: string, dbPath?: string,
92
+ /** T2: the ONLY signature extension the plan permits — injectable wall clock, default `Date.now`,
93
+ * so AC-3 (a rolled-back system clock) can be reproduced without touching the real clock. */
94
+ now?: () => number): {
76
95
  readonly ok: true;
77
96
  readonly generation: number;
78
97
  } | {
@@ -131,17 +150,60 @@ type Embedder = {
131
150
  export declare function resetAgentdbEmbedderCache(): void;
132
151
  /** `entries` = cached keys right now — a SUCCESSFUL pipeline or an IN-FLIGHT initialization (the promise is
133
152
  * cached before it settles, FR-4; a failed one is evicted, FR-3); `initializations` = pipelines actually
134
- * started since the last reset. (Codex round-1, 2026-09-14: the earlier wording said "successful" only.) */
153
+ * started since the last reset. (Codex round-1, 2026-09-14: the earlier wording said "successful" only.)
154
+ * `pipelinesBuilt` (fix round 1, F6) = the count of REAL pipeline-construction primitives that actually
155
+ * ran (`pipeline()` or `EmbeddingService.initialize()`), never merely the number of times the resolver
156
+ * was entered — see {@link embedderCachePipelinesBuilt}'s own doc comment for why the two can diverge. */
135
157
  export declare function getAgentdbEmbedderCacheStats(): {
136
158
  entries: number;
137
159
  initializations: number;
160
+ pipelinesBuilt: number;
161
+ };
162
+ /**
163
+ * C-3 (`embed-daemon-memory`): the SAME resolution order the daemon's own `resolveDeps` uses
164
+ * (`.claude/helpers/dz-embed-daemon.mjs`) — project `package.json` first, then `agentdb`'s own
165
+ * declared dependency (possibly hoisted elsewhere) — so core and the daemon agree on which install
166
+ * of transformers they find, in a monorepo or a plain install alike. Each candidate root is
167
+ * confirmed by {@link findAncestorWithModule} BEFORE `require.resolve` is trusted (see its own doc
168
+ * comment for why the plain try/catch this replaced was not safe under this package's test runner).
169
+ *
170
+ * Exported (fix round 1, F4, same convention as {@link safeErrorMessage}/{@link resetAgentdbEmbedderCache}):
171
+ * `embedder-single-owner.test.ts`'s live-dep skip gate needs the SAME resolution order the production
172
+ * code uses to decide, BEFORE running, whether a live embedder failure is a dependency gap (named skip)
173
+ * or a real defect (must fail) — a text-matching heuristic on the error message cannot tell those apart.
174
+ */
175
+ export declare function resolveTransformersModule(projectRoot: string): {
176
+ url: string;
177
+ } | {
178
+ error: string;
138
179
  };
139
180
  /**
140
- * Resolve agentdb's `EmbeddingService` from the PROJECT (same dynamic-resolution discipline as
141
- * {@link indexPatternsToAgentdb}); every dz call site uses the same resolved model so query and row
142
- * vectors stay in the same space. Cached per process — see {@link embedderCache} above.
181
+ * D2 (`embed-daemon-memory`): the dtype a QUERY/write is embedded with is the STORE's own dtype
182
+ * (its manifest) when the store already exists, falling back to the CONFIGURED dtype only for a
183
+ * store that does not exist yet (its first-ever write picks up the config). This is the ONE place
184
+ * that decision is made — {@link resolveAgentdbEmbedder} calls it so every caller (search, an
185
+ * ordinary incremental index) agrees; `reindexAgentdbRows` is the sole exception (T2/plan): it
186
+ * stamps the manifest with the NEW configured dtype BEFORE it re-embeds, so by the time this
187
+ * function runs during a reindex the manifest already names the new dtype — config and manifest
188
+ * necessarily agree at that point, which is what makes reindex "the one place dtype changes".
143
189
  */
144
- export declare function resolveAgentdbEmbedder(projectRoot: string): Promise<Embedder>;
190
+ export declare function resolveStoreEmbedDtype(projectRoot: string, dbPath?: string): EmbedDtype | {
191
+ error: string;
192
+ };
193
+ /**
194
+ * Resolve the shared embedder from the PROJECT (same dynamic-resolution discipline as
195
+ * {@link indexPatternsToAgentdb}); every dz call site uses the same resolved model/dtype so query
196
+ * and row vectors stay in the same space. Cached per process — see {@link embedderCache} above.
197
+ *
198
+ * Fix round 1 (F1, doc correction — the prior wording was misleading): `dbPath` is passed straight
199
+ * to {@link resolveStoreEmbedDtype}, which calls {@link resolveAgentdbPath}`(projectRoot, dbPath)` —
200
+ * and THAT function already returns the project's DEFAULT store path (`<project>/.dz/agentdb.db`,
201
+ * or `AGENTDB_PATH`) when `dbPath` is omitted, not "no path". So an omitted `dbPath` still reads the
202
+ * default store's OWN manifest when one exists; the CONFIGURED dtype is used only as the fallback
203
+ * for a store that has no manifest yet (i.e. does not exist, or predates this feature) — never as
204
+ * the default behaviour for "no dbPath given".
205
+ */
206
+ export declare function resolveAgentdbEmbedder(projectRoot: string, dbPath?: string): Promise<Embedder>;
145
207
  /**
146
208
  * Cosine similarity in [-1, 1] over two embeddings. Exported (was file-private) so
147
209
  * `harmonizeVectorStore` scores near-duplicate pairs with the IDENTICAL math the semantic search
@@ -248,6 +310,24 @@ export declare function bumpAgentdbUses(projectRoot: string, dzIds: readonly str
248
310
  bumped: number;
249
311
  error?: string;
250
312
  };
313
+ /**
314
+ * Fix-round 1 (CRITICAL, item 1b — the belt): whether {@link reindexAgentdbRows} must run its own
315
+ * rescue bump, given the DELETE's own observed `changes` count and the nested
316
+ * {@link indexPatternsToAgentdb} call's result. Exported test-only (same convention as
317
+ * {@link resetAgentdbEmbedderCache}) so the DECISION can be exercised directly and deterministically,
318
+ * independent of forcing a real concurrent bump-lock race.
319
+ *
320
+ * `!nestedBumped && (deleteChanges > 0 || indexed.indexed > 0 || indexed.error !== undefined)`:
321
+ * - `deleteChanges > 0` — the DELETE genuinely removed rows; the store changed regardless of the
322
+ * nested call's outcome.
323
+ * - `indexed.indexed > 0` — the nested call committed rows itself but its OWN bump failed
324
+ * (`generationBumped: false`) or was never attempted.
325
+ * - `indexed.error !== undefined` — the nested call's post-write state is UNKNOWN (item 1a: an error
326
+ * here may still carry accurate `indexed`/`generationBumped` facts, but a caller must not assume a
327
+ * future error path will). C-1: when in doubt, bump — an extra bump only over-invalidates a cache
328
+ * (safe), a missed one serves stale data (not safe).
329
+ */
330
+ export declare function needsRescueBump(deleteChanges: number, indexed: AgentdbIndexResult): boolean;
251
331
  export declare function reindexAgentdbRows(projectRoot: string, rows: readonly AgentdbRow[], opts?: {
252
332
  dbPath?: string;
253
333
  taskTypes?: readonly string[];