@dzhechkov/harness-core 0.8.35 → 0.8.37

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (135) hide show
  1. package/.dz-manifest.json +224 -104
  2. package/README.md +335 -10
  3. package/dist/agentdb-index.d.ts +87 -7
  4. package/dist/agentdb-index.d.ts.map +1 -1
  5. package/dist/agentdb-index.js +416 -57
  6. package/dist/agentdb-index.js.map +1 -1
  7. package/dist/apply-leg.d.ts +57 -1
  8. package/dist/apply-leg.d.ts.map +1 -1
  9. package/dist/apply-leg.js +450 -52
  10. package/dist/apply-leg.js.map +1 -1
  11. package/dist/codex-hooks-assets.d.ts.map +1 -1
  12. package/dist/codex-hooks-assets.js +67 -5
  13. package/dist/codex-hooks-assets.js.map +1 -1
  14. package/dist/codex-hooks.d.ts +13 -1
  15. package/dist/codex-hooks.d.ts.map +1 -1
  16. package/dist/codex-hooks.js +13 -1
  17. package/dist/codex-hooks.js.map +1 -1
  18. package/dist/codex-rollouts.d.ts +118 -0
  19. package/dist/codex-rollouts.d.ts.map +1 -0
  20. package/dist/codex-rollouts.js +297 -0
  21. package/dist/codex-rollouts.js.map +1 -0
  22. package/dist/cost-ledger.d.ts +56 -4
  23. package/dist/cost-ledger.d.ts.map +1 -1
  24. package/dist/cost-ledger.js +176 -20
  25. package/dist/cost-ledger.js.map +1 -1
  26. package/dist/cross-family-control.d.ts +345 -0
  27. package/dist/cross-family-control.d.ts.map +1 -0
  28. package/dist/cross-family-control.js +802 -0
  29. package/dist/cross-family-control.js.map +1 -0
  30. package/dist/debt-ratchet.d.ts +53 -0
  31. package/dist/debt-ratchet.d.ts.map +1 -0
  32. package/dist/debt-ratchet.js +107 -0
  33. package/dist/debt-ratchet.js.map +1 -0
  34. package/dist/embedding-config.d.ts +42 -0
  35. package/dist/embedding-config.d.ts.map +1 -1
  36. package/dist/embedding-config.js +106 -10
  37. package/dist/embedding-config.js.map +1 -1
  38. package/dist/feature-adr-checkpoints.d.ts +6 -0
  39. package/dist/feature-adr-checkpoints.d.ts.map +1 -1
  40. package/dist/feature-adr-checkpoints.js +29 -0
  41. package/dist/feature-adr-checkpoints.js.map +1 -1
  42. package/dist/feature-adr-decision-recall.d.ts +2 -2
  43. package/dist/feature-adr-decision-recall.d.ts.map +1 -1
  44. package/dist/feature-adr-decision-recall.js +5 -3
  45. package/dist/feature-adr-decision-recall.js.map +1 -1
  46. package/dist/feature-adr-envelope.d.ts +96 -0
  47. package/dist/feature-adr-envelope.d.ts.map +1 -0
  48. package/dist/feature-adr-envelope.js +183 -0
  49. package/dist/feature-adr-envelope.js.map +1 -0
  50. package/dist/feature-adr-routing.d.ts +64 -0
  51. package/dist/feature-adr-routing.d.ts.map +1 -1
  52. package/dist/feature-adr-routing.js +122 -2
  53. package/dist/feature-adr-routing.js.map +1 -1
  54. package/dist/feature-adr-stage-canon.d.ts +79 -0
  55. package/dist/feature-adr-stage-canon.d.ts.map +1 -0
  56. package/dist/feature-adr-stage-canon.js +117 -0
  57. package/dist/feature-adr-stage-canon.js.map +1 -0
  58. package/dist/index.d.ts +23 -12
  59. package/dist/index.d.ts.map +1 -1
  60. package/dist/index.js +15 -7
  61. package/dist/index.js.map +1 -1
  62. package/dist/loop-blobs.generated.js +4 -4
  63. package/dist/loop-blobs.generated.js.map +1 -1
  64. package/dist/mutation-gate.d.ts +51 -0
  65. package/dist/mutation-gate.d.ts.map +1 -1
  66. package/dist/mutation-gate.js +295 -0
  67. package/dist/mutation-gate.js.map +1 -1
  68. package/dist/operations.d.ts +1 -0
  69. package/dist/operations.d.ts.map +1 -1
  70. package/dist/operations.js +18 -2
  71. package/dist/operations.js.map +1 -1
  72. package/dist/publish.d.ts +59 -7
  73. package/dist/publish.d.ts.map +1 -1
  74. package/dist/publish.js +205 -32
  75. package/dist/publish.js.map +1 -1
  76. package/dist/qe-bridge.d.ts.map +1 -1
  77. package/dist/qe-bridge.js +4 -2
  78. package/dist/qe-bridge.js.map +1 -1
  79. package/dist/qe-findings.d.ts +107 -0
  80. package/dist/qe-findings.d.ts.map +1 -0
  81. package/dist/qe-findings.js +417 -0
  82. package/dist/qe-findings.js.map +1 -0
  83. package/dist/recap.d.ts +1 -1
  84. package/dist/recap.d.ts.map +1 -1
  85. package/dist/recap.js +4 -2
  86. package/dist/recap.js.map +1 -1
  87. package/dist/release-line.d.ts +16 -0
  88. package/dist/release-line.d.ts.map +1 -1
  89. package/dist/release-line.js +31 -0
  90. package/dist/release-line.js.map +1 -1
  91. package/dist/round.d.ts +74 -1
  92. package/dist/round.d.ts.map +1 -1
  93. package/dist/round.js +112 -4
  94. package/dist/round.js.map +1 -1
  95. package/dist/run-records.d.ts +60 -0
  96. package/dist/run-records.d.ts.map +1 -1
  97. package/dist/run-records.js +244 -2
  98. package/dist/run-records.js.map +1 -1
  99. package/dist/score.d.ts +44 -1
  100. package/dist/score.d.ts.map +1 -1
  101. package/dist/score.js +78 -5
  102. package/dist/score.js.map +1 -1
  103. package/dist/vector-tier.d.ts +34 -3
  104. package/dist/vector-tier.d.ts.map +1 -1
  105. package/dist/vector-tier.js +105 -14
  106. package/dist/vector-tier.js.map +1 -1
  107. package/package.json +2 -2
  108. package/sbom.json +403 -103
  109. package/src/agentdb-index.ts +423 -60
  110. package/src/apply-leg.ts +469 -50
  111. package/src/codex-hooks-assets.ts +67 -5
  112. package/src/codex-hooks.ts +13 -1
  113. package/src/codex-rollouts.ts +374 -0
  114. package/src/cost-ledger.ts +232 -24
  115. package/src/cross-family-control.ts +960 -0
  116. package/src/debt-ratchet.ts +143 -0
  117. package/src/embedding-config.ts +131 -10
  118. package/src/feature-adr-checkpoints.ts +29 -0
  119. package/src/feature-adr-decision-recall.ts +6 -4
  120. package/src/feature-adr-envelope.ts +242 -0
  121. package/src/feature-adr-routing.ts +139 -2
  122. package/src/feature-adr-stage-canon.ts +141 -0
  123. package/src/index.ts +66 -7
  124. package/src/loop-blobs.generated.ts +4 -4
  125. package/src/mutation-gate.ts +316 -0
  126. package/src/operations.ts +18 -3
  127. package/src/publish.ts +247 -30
  128. package/src/qe-bridge.ts +4 -2
  129. package/src/qe-findings.ts +463 -0
  130. package/src/recap.ts +10 -3
  131. package/src/release-line.ts +32 -0
  132. package/src/round.ts +165 -6
  133. package/src/run-records.ts +282 -2
  134. package/src/score.ts +115 -6
  135. package/src/vector-tier.ts +127 -14
package/src/score.ts CHANGED
@@ -21,6 +21,7 @@
21
21
  */
22
22
 
23
23
  import { amendmentIdsIn } from './amendment-trace.js';
24
+ import { QE_SEVERITIES, findInvalidQeVerdictLines, parseQeFindings, readQeVerdictLines, type QeFindingsRefusedRow, type QeFindingsSummary } from './qe-findings.js';
24
25
 
25
26
  export type DisciplineVerdict = 'pass' | 'partial' | 'absent';
26
27
 
@@ -45,11 +46,29 @@ export interface RunScorecard {
45
46
  * stay distinguishable.
46
47
  */
47
48
  readonly mutationEvidence?: MutationEvidence;
49
+ /**
50
+ * qe-findings-record FR-4: where `qeGrade` came from — a machine-written `QE-VERDICT:` line, the
51
+ * pre-existing prose scan, or neither. ADDITIVE for the same reason as `mutationEvidence`: every
52
+ * scorecard built before this field existed still satisfies `RunScorecard`.
53
+ */
54
+ readonly gradeSource?: GradeSource;
55
+ /**
56
+ * qe-findings-record FR-4: the report's Findings ledger, when one exists. `absent` for the 406
57
+ * pre-existing reports that carry neither the table nor the heading (NFR-1) — never omitted, for
58
+ * the same "gate never ran vs a field an older scorer never wrote" reason `mutationEvidence` gives.
59
+ */
60
+ readonly findings?: QeFindingsScoreView;
48
61
  readonly passed: number;
49
62
  readonly total: number;
50
63
  readonly summary: string;
51
64
  }
52
65
 
66
+ /** A lighter projection of `QeFindingsResult` for the scorecard — summary + refused + hollow, never
67
+ * the full row list (that stays in `parseQeFindings`'s own return value for a caller that wants it). */
68
+ export type QeFindingsScoreView =
69
+ | { readonly status: 'absent' }
70
+ | { readonly status: 'present'; readonly hollow: boolean; readonly summary: QeFindingsSummary; readonly refused: readonly QeFindingsRefusedRow[] };
71
+
53
72
  /** The artifact texts of one run, keyed by RELATIVE path under `features/<slug>/`. */
54
73
  /**
55
74
  * The exact heading Step 5 asks for, and the exact heading the check looks for — ONE constant, so
@@ -306,7 +325,22 @@ function normaliseGradeSign(grade: string): string {
306
325
  return grade.replace('\u2212', '-');
307
326
  }
308
327
 
309
- export type GradeReadStatus = 'unique' | 'ambiguous' | 'none';
328
+ /**
329
+ * fix-round-1 (codex-r1-verdict finding 2): `'invalid'` is a report that ATTEMPTED a machine verdict
330
+ * (a line starting `QE-VERDICT`, any case/spacing) and got the grammar wrong — `QE-VERDICT: B – final`
331
+ * (en dash + trailing prose), wrong case, no colon, a grade outside A-D. That is a DIFFERENT fact from
332
+ * `'none'` (no attempt at all, legacy prose scan still applies): a malformed attempt must never fall
333
+ * back to guessing a grade from prose — the malformed line is itself evidence the report is unreliable
334
+ * here, and guessing past it would silently launder that unreliability into a confident number.
335
+ */
336
+ export type GradeReadStatus = 'unique' | 'ambiguous' | 'none' | 'invalid';
337
+
338
+ /**
339
+ * qe-findings-record (ADR-001 D1): where the grade came from. `'verdict-line'` — a machine-written
340
+ * `QE-VERDICT:` line, the source of truth when present. `'prose'` — the pre-existing GRADE_RE scan,
341
+ * unchanged, used only when NO verdict line exists. `'none'` — neither surface names a grade.
342
+ */
343
+ export type GradeSource = 'verdict-line' | 'prose' | 'none';
310
344
 
311
345
  export interface GradeReading {
312
346
  readonly status: GradeReadStatus;
@@ -314,6 +348,7 @@ export interface GradeReading {
314
348
  readonly grade: string | null;
315
349
  /** Every distinct grade found, normalised — what makes an `ambiguous` verdict inspectable. */
316
350
  readonly found: readonly string[];
351
+ readonly source: GradeSource;
317
352
  }
318
353
 
319
354
  /**
@@ -329,6 +364,44 @@ export interface GradeReading {
329
364
  * report is ambiguous, which is a fact about the report, not a missing number.
330
365
  */
331
366
  export function readQeGrade(qeText: string): GradeReading {
367
+ // ADR-001 D1: a machine-written verdict line is the source of truth WHEN it exists — checked
368
+ // first, and the prose scan below never runs when it does. Exactly one -> unique; more than one
369
+ // -> ambiguous ("две строки — ambiguous, никогда «последняя побеждает»", even if both name the
370
+ // SAME grade — two lines is a fact about the report, not a number to reconcile); zero -> the
371
+ // prose scan runs exactly as it always has (NFR-1: 406 pre-existing reports are unaffected) —
372
+ // UNLESS a malformed attempt exists (fix-round-1 finding 2, checked next): the report is `invalid`
373
+ // and the prose fallback is refused, never silently reached.
374
+ const verdictLines = readQeVerdictLines(qeText);
375
+ // Lead delta after Codex r2 (new HIGH #1): a malformed declaration next to a valid one is NOT a
376
+ // unique verdict — `QE-VERDICT: A` + `qe-verdict: D` used to read as unique A. Invalid lines are
377
+ // checked FIRST, whatever the count of valid ones.
378
+ const invalidFirst = findInvalidQeVerdictLines(qeText);
379
+ if (invalidFirst.length > 0) {
380
+ const reasons = invalidFirst.map(
381
+ (l) => `line ${l.line}: ${JSON.stringify(l.text.trim())} does not match "QE-VERDICT: <A|B|C|D><+|-> "`,
382
+ );
383
+ return { status: 'invalid', grade: null, found: [...verdictLines, ...reasons], source: 'verdict-line' };
384
+ }
385
+ if (verdictLines.length === 1) {
386
+ return { status: 'unique', grade: verdictLines[0] as string, found: verdictLines, source: 'verdict-line' };
387
+ }
388
+ if (verdictLines.length > 1) {
389
+ const distinct: string[] = [];
390
+ for (const g of verdictLines) if (!distinct.includes(g)) distinct.push(g);
391
+ return { status: 'ambiguous', grade: null, found: distinct, source: 'verdict-line' };
392
+ }
393
+
394
+ // fix-round-1 finding 2: zero GRAMMATICALLY VALID verdict lines is not yet "no attempt" — a line
395
+ // that clearly opens a verdict declaration but botches the grammar (wrong dash, wrong case, no
396
+ // colon, a grade outside A-D) is `invalid`, and the legacy prose scan below is FORBIDDEN for it.
397
+ // The reason lives inside `found` (GradeReading keeps the same 4-field shape for every status).
398
+ // Lead delta after Codex r2 (#2 PARTIAL): the prose fallback is for LEGACY reports only. A report
399
+ // that already carries the new-format ledger heading but no verdict line is a NEW-format report
400
+ // missing its verdict — `invalid`, never a prose guess.
401
+ if (/^## Findings ledger\s*$/m.test(qeText)) {
402
+ return { status: 'invalid', grade: null, found: ['new-format report (has "## Findings ledger") without a QE-VERDICT line'], source: 'verdict-line' };
403
+ }
404
+
332
405
  const found: string[] = [];
333
406
  const lines = qeText.split('\n');
334
407
  // Line offsets once, so each match maps to ITS line for the negation screen (67d7883d: the
@@ -352,9 +425,9 @@ export function readQeGrade(qeText: string): GradeReading {
352
425
  const g = normaliseGradeSign(m[1] as string);
353
426
  if (!found.includes(g)) found.push(g);
354
427
  }
355
- if (found.length === 0) return { status: 'none', grade: null, found: [] };
356
- if (found.length === 1) return { status: 'unique', grade: found[0] as string, found };
357
- return { status: 'ambiguous', grade: null, found };
428
+ if (found.length === 0) return { status: 'none', grade: null, found: [], source: 'none' };
429
+ if (found.length === 1) return { status: 'unique', grade: found[0] as string, found, source: 'prose' };
430
+ return { status: 'ambiguous', grade: null, found, source: 'prose' };
358
431
  }
359
432
 
360
433
  export function extractQeGrade(qeText: string): string | null {
@@ -451,7 +524,13 @@ export function scoreRun(slug: string, artifacts: RunArtifacts): RunScorecard {
451
524
  );
452
525
 
453
526
  // 3. Cross-model QE — an independent family reviewed it, and a grade exists.
454
- const grade = extractQeGrade(qeText);
527
+ const gradeReading = readQeGrade(qeText);
528
+ const grade = gradeReading.grade;
529
+ const findingsParsed = parseQeFindings(qeText);
530
+ const findings: QeFindingsScoreView =
531
+ findingsParsed.status === 'absent'
532
+ ? { status: 'absent' }
533
+ : { status: 'present', hollow: findingsParsed.hollow, summary: findingsParsed.summary, refused: findingsParsed.refused };
455
534
  if (qeText === '') {
456
535
  add('cross-model-qe', 'independent cross-model review with a grade', 'absent', 'no 08_qe_report.md artifact');
457
536
  } else {
@@ -559,11 +638,36 @@ export function scoreRun(slug: string, artifacts: RunArtifacts): RunScorecard {
559
638
  (grade !== null ? ` · QE grade ${grade}` : ' · no QE grade') +
560
639
  (worst.length > 0 ? ` · absent: ${worst.join(', ')}` : '');
561
640
 
562
- return { slug, disciplines, qeGrade: grade, mutationEvidence, passed, total, summary };
641
+ return { slug, disciplines, qeGrade: grade, gradeSource: gradeReading.source, findings, mutationEvidence, passed, total, summary };
563
642
  }
564
643
 
565
644
  const MARK: Record<DisciplineVerdict, string> = { pass: '✓', partial: '◐', absent: '✗' };
566
645
 
646
+ /** qe-findings-record FR-4: the one-line summary `dz score` prints for a report's Findings ledger —
647
+ * "findings: 3 HIGH / 2 MEDIUM; 1 refused (line 84: severity "Major" not in dictionary)". Ordered by
648
+ * QE_SEVERITIES (BLOCKER first) so the worst finding always reads first. `null` when there is no
649
+ * table at all — the common case, which earns no noise (same rule as mutationEvidence above). */
650
+ export function renderFindingsLine(findings: QeFindingsScoreView | undefined): string | null {
651
+ if (findings === undefined || findings.status === 'absent') return null;
652
+ // fix-round-1 finding 6: hollow used to short-circuit BEFORE the refused check, so a report whose
653
+ // real ledger table (with CRITICAL rows) was refused as duplicate/outside-section alongside an
654
+ // empty accepted table read as pure "EMPTY" — the refused rows vanished from the printed line, not
655
+ // merely from the tally. Both facts are ALWAYS reported together now; neither hides the other.
656
+ const refusedSuffix = ((): string => {
657
+ if (findings.refused.length === 0) return '';
658
+ const first = findings.refused[0] as QeFindingsRefusedRow;
659
+ const more = findings.refused.length > 1 ? `, +${findings.refused.length - 1} more` : '';
660
+ return `; ${findings.refused.length} refused (line ${first.line}: ${first.reason})${more}`;
661
+ })();
662
+ if (findings.hollow) {
663
+ return `findings: table present but EMPTY (hollow) — worse than no table at all${refusedSuffix}`;
664
+ }
665
+ const bySev = findings.summary.bySeverity;
666
+ const parts = QE_SEVERITIES.filter((s) => (bySev[s] ?? 0) > 0).map((s) => `${bySev[s]} ${s}`);
667
+ const head = parts.length > 0 ? parts.join(' / ') : 'no rows';
668
+ return `findings: ${head}${refusedSuffix}`;
669
+ }
670
+
567
671
  export function renderScorecard(card: RunScorecard): string {
568
672
  const out: string[] = [];
569
673
  out.push(`dz score — ${card.slug} (process scorecard; descriptive-only, never a gate)`);
@@ -581,6 +685,11 @@ export function renderScorecard(card: RunScorecard): string {
581
685
  out.push('');
582
686
  out.push(` ${card.mutationEvidence.evidence}`);
583
687
  }
688
+ const findingsLine = renderFindingsLine(card.findings);
689
+ if (findingsLine !== null) {
690
+ out.push('');
691
+ out.push(` ${findingsLine}${card.gradeSource !== undefined && card.gradeSource !== 'none' ? ` (grade source: ${card.gradeSource})` : ''}`);
692
+ }
584
693
  out.push('');
585
694
  out.push(` ${card.summary}`);
586
695
  return out.join('\n');
@@ -1158,13 +1158,87 @@ export interface RankedPattern {
1158
1158
 
1159
1159
  const RRF_K = 60;
1160
1160
 
1161
+ /**
1162
+ * One entry in the total order every post-merge ranking step shares (feature
1163
+ * `recall-parity-tie-break`, FR-1). `evidence` is the SAME three-way rank `mergeHybridHits` has
1164
+ * always used (`both` < lexical-only < semantic-only — lower is stronger), computed once by
1165
+ * {@link evidenceRank} from a hit's `backend`.
1166
+ */
1167
+ export interface HybridOrderKey {
1168
+ readonly score: number;
1169
+ readonly evidence: number;
1170
+ readonly dzId: string;
1171
+ }
1172
+
1173
+ /** `both` outranks lexical-only outranks semantic-only (mergeHybridHits' own rule, ADR-001 AM-4). */
1174
+ export function evidenceRank(backend: RecallHit['backend']): number {
1175
+ return backend === 'both' ? 0 : backend === 'vector' ? 2 : 1;
1176
+ }
1177
+
1178
+ /**
1179
+ * The ONE deterministic total order recall uses at every step where a tie can occur: fused score
1180
+ * DESC, then EVIDENCE (both > lexical-only > semantic-only), then `dzId` ASC. `mergeHybridHits`
1181
+ * always applied exactly this rule inline; it is exported here (FR-1) so `dampQuarantined`,
1182
+ * `orderHitsForReRank` (the pre-sort `enhance()` runs before its reinforcement/bandit re-rank) and
1183
+ * any other post-merge sort can share the SAME tiebreak instead of an ad hoc score-only comparator
1184
+ * that is deterministic only because its input already arrived pre-ordered — a property that
1185
+ * silently breaks the moment an upstream step feeds it hits in a different order
1186
+ * (recall-parity-tie-break T0: MEASURED, `apply-leg-recall-parity.test.ts` AM-4, a racy
1187
+ * reinforcement-signal read, not a comparator defect, actually explained the observed tail swap —
1188
+ * this comparator is hardening kept from that round; the actual causal fix, per fix-round 1, is the
1189
+ * byte-level store snapshot AM-4 now takes, not this comparator and not an awaited flush).
1190
+ *
1191
+ * FINITE-NUMBER INVARIANT (fix-round 1, LOW finding): plain subtraction (`b.score - a.score`) is
1192
+ * NOT total over `number` — `NaN - x` is `NaN`, and the `||` chain treats a `NaN` term as falsy,
1193
+ * silently SKIPPING it and falling through to the next key as if score had never been compared.
1194
+ * Both terms below use explicit `>`/`<` comparisons instead (correct as-is for ±Infinity — IEEE 754
1195
+ * orders infinities correctly) plus an explicit NaN case: a `NaN` score or evidence is the WEAKEST
1196
+ * possible value on its own axis, so it sorts deterministically LAST, never a coincidental tie.
1197
+ */
1198
+ function compareScoreDesc(a: number, b: number): number {
1199
+ if (Number.isNaN(a) || Number.isNaN(b)) return Number.isNaN(a) && Number.isNaN(b) ? 0 : Number.isNaN(a) ? 1 : -1;
1200
+ return a > b ? -1 : a < b ? 1 : 0;
1201
+ }
1202
+ function compareEvidenceAsc(a: number, b: number): number {
1203
+ if (Number.isNaN(a) || Number.isNaN(b)) return Number.isNaN(a) && Number.isNaN(b) ? 0 : Number.isNaN(a) ? 1 : -1;
1204
+ return a < b ? -1 : a > b ? 1 : 0;
1205
+ }
1206
+ export function compareHybridHits(a: HybridOrderKey, b: HybridOrderKey): number {
1207
+ return compareScoreDesc(a.score, b.score)
1208
+ || compareEvidenceAsc(a.evidence, b.evidence)
1209
+ || (a.dzId < b.dzId ? -1 : a.dzId > b.dzId ? 1 : 0);
1210
+ }
1211
+
1212
+ /**
1213
+ * Sorts `hits` into the shared total order {@link compareHybridHits} defines — used by `enhance()`
1214
+ * BEFORE its reinforcement/bandit re-rank runs (`applyLearningSignalsWithTerms` et al.,
1215
+ * `learning-backend.ts`, out of this fix's edit scope). That re-rank sorts by an ADJUSTED score
1216
+ * with a STABLE tie-break on each hit's ORIGINAL array position — so pre-ordering the input here
1217
+ * makes any tie in the adjusted score resolve in the SAME evidence/dzId order `compareHybridHits`
1218
+ * would give directly, without touching the re-rank's own internals.
1219
+ *
1220
+ * This closes the ordering gap `enhance()` had (recall-parity-tie-break fix-round 1, HIGH finding):
1221
+ * `dampQuarantined` only ran {@link compareHybridHits} when `memory.learning.quarantine` was ON;
1222
+ * with it OFF (the default), `enhance()`'s final order was whatever the re-rank's own
1223
+ * original-index tie-break happened to preserve — invisible from the printed `score` column,
1224
+ * because the re-rank reorders the hit array but never rewrites `.score`. Real (non-tied) score
1225
+ * differences from reinforcement/bandit re-ranking are UNCHANGED by this — it only decides ties.
1226
+ */
1227
+ export function orderHitsForReRank(hits: readonly HybridHit[], idOf: (p: PatternRecord) => string): HybridHit[] {
1228
+ return [...hits].sort((a, b) => compareHybridHits(
1229
+ { score: a.score, evidence: evidenceRank(a.backend), dzId: idOf(a.pattern) },
1230
+ { score: b.score, evidence: evidenceRank(b.backend), dzId: idOf(b.pattern) },
1231
+ ));
1232
+ }
1233
+
1161
1234
  /**
1162
1235
  * Reciprocal Rank Fusion merge: `score(p) = Σ 1/(60 + rank)` over the lists containing `p`
1163
1236
  * (semantic ranks weighted by `semanticWeight`). Dedup by id; `backend: 'both'` when a pattern
1164
1237
  *
1165
- * Ordering: fused score, then EVIDENCE (`both` before lexical-only before semantic-only), then id.
1166
- * When `semanticWeight > 1` the lexical top-1 is guaranteed a place in the result, taken from the
1167
- * last seat unless that seat holds a `both` hit. See `features/semantic-keeps-exact-hits`.
1238
+ * Ordering: fused score, then EVIDENCE (`both` before lexical-only before semantic-only), then id
1239
+ * — {@link compareHybridHits}. When `semanticWeight > 1` the lexical top-1 is guaranteed a place in
1240
+ * the result, taken from the last seat unless that seat holds a `both` hit. See
1241
+ * `features/semantic-keeps-exact-hits`.
1168
1242
  *
1169
1243
  * appears in both lists. DETERMINISTIC (AC-6): ties break on id, so fixed inputs always yield
1170
1244
  * the same ordering. Pure — no I/O.
@@ -1196,12 +1270,14 @@ export function mergeHybridHits(
1196
1270
  });
1197
1271
  // Ties break by EVIDENCE, not by the id alphabet: a hit both legs found outranks one only a single
1198
1272
  // leg found. Before this, an exact-term match lost a tie to an arbitrary semantic hit purely
1199
- // because its id sorted later (ADR-001 AM-4).
1200
- const evidence = (v: Acc): number => (v.lex !== undefined && v.sem ? 0 : v.lex !== undefined ? 1 : 2);
1273
+ // because its id sorted later (ADR-001 AM-4). Now routed through the shared {@link
1274
+ // compareHybridHits} (FR-1) — same three-part rule, no behaviour change.
1275
+ const evidenceOfAcc = (v: Acc): number => evidenceRank(v.lex !== undefined && v.sem ? 'both' : v.lex ?? 'vector');
1201
1276
  const ordered = [...acc.entries()]
1202
- .sort((a, b) => b[1].score - a[1].score
1203
- || evidence(a[1]) - evidence(b[1])
1204
- || (a[0] < b[0] ? -1 : a[0] > b[0] ? 1 : 0));
1277
+ .sort((a, b) => compareHybridHits(
1278
+ { score: a[1].score, evidence: evidenceOfAcc(a[1]), dzId: a[0] },
1279
+ { score: b[1].score, evidence: evidenceOfAcc(b[1]), dzId: b[0] },
1280
+ ));
1205
1281
  const toHit = ([, v]: [string, Acc]): HybridHit => ({
1206
1282
  pattern: v.pattern,
1207
1283
  backend: v.lex !== undefined && v.sem ? ('both' as const) : v.lex ?? ('vector' as const),
@@ -1256,6 +1332,27 @@ export function mergeHybridHits(
1256
1332
  *
1257
1333
  * `bandit` is passed ONLY when `memory.learning.banditRerank` is armed; when it is absent this
1258
1334
  * function is byte-identical to its pre-feature self — no state file, no lock, no allocation.
1335
+ *
1336
+ * `backend.train()` fires FIRE-AND-FORGET (`void backend.train().catch(() => undefined)`) — this is
1337
+ * a REVERT (recall-parity-tie-break, fix-round 1, MEDIUM finding). An intermediate version of this
1338
+ * fix AWAITED the flush, on the theory that a caller's own reinforcement write landing before it got
1339
+ * an answer would remove the race `apply-leg-recall-parity.test.ts` AM-4 was catching (two tail hits
1340
+ * swapping order between the daemon's `op:recall` reply and a `dz recall --json` invoked a moment
1341
+ * later — MEASURED byte-identical `score` fields, only `uses` differed, so the merge/comparator was
1342
+ * never the cause). MEASURED (lead, 2026-09-15 15:57, temp project, 8 lexical hits, 5 warm runs):
1343
+ * `recallHybrid` took 2 ms with `onRecallHits:false` (no flush at all) vs 57–131 ms with the flush
1344
+ * AWAITED — landing INSIDE the hook's 500 ms `HOOK_RECALL_BUDGET_MS` but a real, avoidable tax on
1345
+ * every recall, for a property the await did not even fully deliver: the awaited write still let a
1346
+ * CLI invoked immediately after the daemon see the DAEMON'S OWN just-computed exposure for that same
1347
+ * query, answering a subtly different question than the daemon had. AM-4 now proves parity by taking
1348
+ * a byte-level SNAPSHOT of the store BEFORE each query's daemon call and pointing `dz recall --json`
1349
+ * at the frozen snapshot (`--project <snapshot>`) — both sides then answer the identical question
1350
+ * from the identical state under PRODUCTION defaults (`onRecallHits` ON), and no write, awaited or
1351
+ * not, can reach the CLI's read. That snapshot is what actually closes the race; this function stays
1352
+ * fire-and-forget, exactly as it always was, because the snapshot makes its timing irrelevant to the
1353
+ * test. `test/lesson-bandit-byte-identity.test.ts` documents the same underlying race in its own
1354
+ * fixture comment and works around it with `onRecallHits: false` there — a narrower, still-valid
1355
+ * isolation for a different test's needs.
1259
1356
  */
1260
1357
  function markRecallHits(
1261
1358
  projectRoot: string,
@@ -1356,7 +1453,15 @@ export async function recallHybrid(
1356
1453
  const q = rec !== undefined && readQuarantineState(rec).quarantined;
1357
1454
  return q ? { ...h, score: h.score * memCfg.quarantineDamp, quarantined: true as const } : h;
1358
1455
  })
1359
- .sort((a, b) => b.score - a.score);
1456
+ // FR-1 (recall-parity-tie-break): score-only used to rely on the INCOMING array already
1457
+ // being pre-ordered (native Array.sort is stable, so a genuine tie only stayed put by
1458
+ // accident of arrival order). Routed through the same {@link compareHybridHits} the merge
1459
+ // uses, so damping two equally-scored hits can never reorder them differently from how the
1460
+ // merge itself would have.
1461
+ .sort((a, b) => compareHybridHits(
1462
+ { score: a.score, evidence: evidenceRank(a.backend), dzId: idOf(a.pattern) },
1463
+ { score: b.score, evidence: evidenceRank(b.backend), dzId: idOf(b.pattern) },
1464
+ ));
1360
1465
  };
1361
1466
  // lesson-bandit-rerank (ADR-001): the payoff axis. Resolved ONCE per recall; `enabled:false` ⇒
1362
1467
  // the Lesson Payoff context is NEVER CONSTRUCTED — the branch is taken BEFORE any work, so the
@@ -1366,7 +1471,15 @@ export async function recallHybrid(
1366
1471
  let banditReport: BanditRecallReport | undefined;
1367
1472
  let banditExplored: readonly string[] = [];
1368
1473
  const enhance = (hits: readonly HybridHit[]): HybridHit[] => {
1369
- const candidates = hits.map((h) => {
1474
+ // FR-1 (recall-parity-tie-break, fix-round 1, HIGH finding): pre-sort into the shared total
1475
+ // order BEFORE the reinforcement/bandit re-rank runs — see {@link orderHitsForReRank}.
1476
+ // Why the pre-sort is sufficient and not "reliance on a stable sort" (Codex round 2): the
1477
+ // re-rank's own sort in learning-backend.ts is `b.adjusted - a.adjusted || a.i - b.i` — an
1478
+ // EXPLICIT tie-break on the incoming index, so an adjusted-score tie resolves to exactly the
1479
+ // order built here (score → evidence → dzId), by construction, on any engine. That generic
1480
+ // function only knows `score`, so it cannot call compareHybridHits itself.
1481
+ const ordered = orderHitsForReRank(hits, idOf);
1482
+ const candidates = ordered.map((h) => {
1370
1483
  const dzId = idOf(h.pattern);
1371
1484
  const rec = idToRecord.get(dzId);
1372
1485
  return { dzId, score: h.score, reinforcement: rec !== undefined ? readReinforcementState(rec) : undefined };
@@ -1392,8 +1505,8 @@ export async function recallHybrid(
1392
1505
  ? []
1393
1506
  : [{ id: 'delta', byIndex: candidates.map((c) => deltaMap.get(c.dzId) ?? 0), cap: REINFORCE_RRF_CAP }];
1394
1507
  // The SAME ranking without the payoff term — the only honest way to say what the term moved.
1395
- const before = dampQuarantined(applyLearningSignalsWithTerms(hits, learning, candidates, REINFORCE_RRF_CAP, baseTerms));
1396
- const after = dampQuarantined(applyLearningSignalsWithTerms(hits, learning, candidates, REINFORCE_RRF_CAP, [
1508
+ const before = dampQuarantined(applyLearningSignalsWithTerms(ordered, learning, candidates, REINFORCE_RRF_CAP, baseTerms));
1509
+ const after = dampQuarantined(applyLearningSignalsWithTerms(ordered, learning, candidates, REINFORCE_RRF_CAP, [
1397
1510
  ...baseTerms,
1398
1511
  // ADDED, never assigned, and pre-bounded to [-1,+1] by the ACL — so `squash` is identity and
1399
1512
  // `cap` is an EXACT bound on this term's contribution (INV-4).
@@ -1423,9 +1536,9 @@ export async function recallHybrid(
1423
1536
  }
1424
1537
  if (deltaMap !== undefined) {
1425
1538
  const deltaByIndex = candidates.map((c) => deltaMap.get(c.dzId) ?? 0);
1426
- return dampQuarantined(applyLearningSignalsWithDelta(hits, learning, candidates, REINFORCE_RRF_CAP, deltaByIndex, REINFORCE_RRF_CAP));
1539
+ return dampQuarantined(applyLearningSignalsWithDelta(ordered, learning, candidates, REINFORCE_RRF_CAP, deltaByIndex, REINFORCE_RRF_CAP));
1427
1540
  }
1428
- return dampQuarantined(applyLearningSignals(hits, learning, candidates, REINFORCE_RRF_CAP));
1541
+ return dampQuarantined(applyLearningSignals(ordered, learning, candidates, REINFORCE_RRF_CAP));
1429
1542
  };
1430
1543
  /** The exposure/telemetry payload for `markRecallHits` — `undefined` while disarmed (INV-1). */
1431
1544
  const banditEmission = (): { readonly contextKey: string; readonly explored: readonly string[]; readonly moved: number; readonly arms: number; readonly deferred?: boolean } | undefined =>