@dzhechkov/harness-core 0.8.35 → 0.8.37
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.dz-manifest.json +224 -104
- package/README.md +335 -10
- package/dist/agentdb-index.d.ts +87 -7
- package/dist/agentdb-index.d.ts.map +1 -1
- package/dist/agentdb-index.js +416 -57
- package/dist/agentdb-index.js.map +1 -1
- package/dist/apply-leg.d.ts +57 -1
- package/dist/apply-leg.d.ts.map +1 -1
- package/dist/apply-leg.js +450 -52
- package/dist/apply-leg.js.map +1 -1
- package/dist/codex-hooks-assets.d.ts.map +1 -1
- package/dist/codex-hooks-assets.js +67 -5
- package/dist/codex-hooks-assets.js.map +1 -1
- package/dist/codex-hooks.d.ts +13 -1
- package/dist/codex-hooks.d.ts.map +1 -1
- package/dist/codex-hooks.js +13 -1
- package/dist/codex-hooks.js.map +1 -1
- package/dist/codex-rollouts.d.ts +118 -0
- package/dist/codex-rollouts.d.ts.map +1 -0
- package/dist/codex-rollouts.js +297 -0
- package/dist/codex-rollouts.js.map +1 -0
- package/dist/cost-ledger.d.ts +56 -4
- package/dist/cost-ledger.d.ts.map +1 -1
- package/dist/cost-ledger.js +176 -20
- package/dist/cost-ledger.js.map +1 -1
- package/dist/cross-family-control.d.ts +345 -0
- package/dist/cross-family-control.d.ts.map +1 -0
- package/dist/cross-family-control.js +802 -0
- package/dist/cross-family-control.js.map +1 -0
- package/dist/debt-ratchet.d.ts +53 -0
- package/dist/debt-ratchet.d.ts.map +1 -0
- package/dist/debt-ratchet.js +107 -0
- package/dist/debt-ratchet.js.map +1 -0
- package/dist/embedding-config.d.ts +42 -0
- package/dist/embedding-config.d.ts.map +1 -1
- package/dist/embedding-config.js +106 -10
- package/dist/embedding-config.js.map +1 -1
- package/dist/feature-adr-checkpoints.d.ts +6 -0
- package/dist/feature-adr-checkpoints.d.ts.map +1 -1
- package/dist/feature-adr-checkpoints.js +29 -0
- package/dist/feature-adr-checkpoints.js.map +1 -1
- package/dist/feature-adr-decision-recall.d.ts +2 -2
- package/dist/feature-adr-decision-recall.d.ts.map +1 -1
- package/dist/feature-adr-decision-recall.js +5 -3
- package/dist/feature-adr-decision-recall.js.map +1 -1
- package/dist/feature-adr-envelope.d.ts +96 -0
- package/dist/feature-adr-envelope.d.ts.map +1 -0
- package/dist/feature-adr-envelope.js +183 -0
- package/dist/feature-adr-envelope.js.map +1 -0
- package/dist/feature-adr-routing.d.ts +64 -0
- package/dist/feature-adr-routing.d.ts.map +1 -1
- package/dist/feature-adr-routing.js +122 -2
- package/dist/feature-adr-routing.js.map +1 -1
- package/dist/feature-adr-stage-canon.d.ts +79 -0
- package/dist/feature-adr-stage-canon.d.ts.map +1 -0
- package/dist/feature-adr-stage-canon.js +117 -0
- package/dist/feature-adr-stage-canon.js.map +1 -0
- package/dist/index.d.ts +23 -12
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +15 -7
- package/dist/index.js.map +1 -1
- package/dist/loop-blobs.generated.js +4 -4
- package/dist/loop-blobs.generated.js.map +1 -1
- package/dist/mutation-gate.d.ts +51 -0
- package/dist/mutation-gate.d.ts.map +1 -1
- package/dist/mutation-gate.js +295 -0
- package/dist/mutation-gate.js.map +1 -1
- package/dist/operations.d.ts +1 -0
- package/dist/operations.d.ts.map +1 -1
- package/dist/operations.js +18 -2
- package/dist/operations.js.map +1 -1
- package/dist/publish.d.ts +59 -7
- package/dist/publish.d.ts.map +1 -1
- package/dist/publish.js +205 -32
- package/dist/publish.js.map +1 -1
- package/dist/qe-bridge.d.ts.map +1 -1
- package/dist/qe-bridge.js +4 -2
- package/dist/qe-bridge.js.map +1 -1
- package/dist/qe-findings.d.ts +107 -0
- package/dist/qe-findings.d.ts.map +1 -0
- package/dist/qe-findings.js +417 -0
- package/dist/qe-findings.js.map +1 -0
- package/dist/recap.d.ts +1 -1
- package/dist/recap.d.ts.map +1 -1
- package/dist/recap.js +4 -2
- package/dist/recap.js.map +1 -1
- package/dist/release-line.d.ts +16 -0
- package/dist/release-line.d.ts.map +1 -1
- package/dist/release-line.js +31 -0
- package/dist/release-line.js.map +1 -1
- package/dist/round.d.ts +74 -1
- package/dist/round.d.ts.map +1 -1
- package/dist/round.js +112 -4
- package/dist/round.js.map +1 -1
- package/dist/run-records.d.ts +60 -0
- package/dist/run-records.d.ts.map +1 -1
- package/dist/run-records.js +244 -2
- package/dist/run-records.js.map +1 -1
- package/dist/score.d.ts +44 -1
- package/dist/score.d.ts.map +1 -1
- package/dist/score.js +78 -5
- package/dist/score.js.map +1 -1
- package/dist/vector-tier.d.ts +34 -3
- package/dist/vector-tier.d.ts.map +1 -1
- package/dist/vector-tier.js +105 -14
- package/dist/vector-tier.js.map +1 -1
- package/package.json +2 -2
- package/sbom.json +403 -103
- package/src/agentdb-index.ts +423 -60
- package/src/apply-leg.ts +469 -50
- package/src/codex-hooks-assets.ts +67 -5
- package/src/codex-hooks.ts +13 -1
- package/src/codex-rollouts.ts +374 -0
- package/src/cost-ledger.ts +232 -24
- package/src/cross-family-control.ts +960 -0
- package/src/debt-ratchet.ts +143 -0
- package/src/embedding-config.ts +131 -10
- package/src/feature-adr-checkpoints.ts +29 -0
- package/src/feature-adr-decision-recall.ts +6 -4
- package/src/feature-adr-envelope.ts +242 -0
- package/src/feature-adr-routing.ts +139 -2
- package/src/feature-adr-stage-canon.ts +141 -0
- package/src/index.ts +66 -7
- package/src/loop-blobs.generated.ts +4 -4
- package/src/mutation-gate.ts +316 -0
- package/src/operations.ts +18 -3
- package/src/publish.ts +247 -30
- package/src/qe-bridge.ts +4 -2
- package/src/qe-findings.ts +463 -0
- package/src/recap.ts +10 -3
- package/src/release-line.ts +32 -0
- package/src/round.ts +165 -6
- package/src/run-records.ts +282 -2
- package/src/score.ts +115 -6
- package/src/vector-tier.ts +127 -14
package/src/score.ts
CHANGED
|
@@ -21,6 +21,7 @@
|
|
|
21
21
|
*/
|
|
22
22
|
|
|
23
23
|
import { amendmentIdsIn } from './amendment-trace.js';
|
|
24
|
+
import { QE_SEVERITIES, findInvalidQeVerdictLines, parseQeFindings, readQeVerdictLines, type QeFindingsRefusedRow, type QeFindingsSummary } from './qe-findings.js';
|
|
24
25
|
|
|
25
26
|
export type DisciplineVerdict = 'pass' | 'partial' | 'absent';
|
|
26
27
|
|
|
@@ -45,11 +46,29 @@ export interface RunScorecard {
|
|
|
45
46
|
* stay distinguishable.
|
|
46
47
|
*/
|
|
47
48
|
readonly mutationEvidence?: MutationEvidence;
|
|
49
|
+
/**
|
|
50
|
+
* qe-findings-record FR-4: where `qeGrade` came from — a machine-written `QE-VERDICT:` line, the
|
|
51
|
+
* pre-existing prose scan, or neither. ADDITIVE for the same reason as `mutationEvidence`: every
|
|
52
|
+
* scorecard built before this field existed still satisfies `RunScorecard`.
|
|
53
|
+
*/
|
|
54
|
+
readonly gradeSource?: GradeSource;
|
|
55
|
+
/**
|
|
56
|
+
* qe-findings-record FR-4: the report's Findings ledger, when one exists. `absent` for the 406
|
|
57
|
+
* pre-existing reports that carry neither the table nor the heading (NFR-1) — never omitted, for
|
|
58
|
+
* the same "gate never ran vs a field an older scorer never wrote" reason `mutationEvidence` gives.
|
|
59
|
+
*/
|
|
60
|
+
readonly findings?: QeFindingsScoreView;
|
|
48
61
|
readonly passed: number;
|
|
49
62
|
readonly total: number;
|
|
50
63
|
readonly summary: string;
|
|
51
64
|
}
|
|
52
65
|
|
|
66
|
+
/** A lighter projection of `QeFindingsResult` for the scorecard — summary + refused + hollow, never
|
|
67
|
+
* the full row list (that stays in `parseQeFindings`'s own return value for a caller that wants it). */
|
|
68
|
+
export type QeFindingsScoreView =
|
|
69
|
+
| { readonly status: 'absent' }
|
|
70
|
+
| { readonly status: 'present'; readonly hollow: boolean; readonly summary: QeFindingsSummary; readonly refused: readonly QeFindingsRefusedRow[] };
|
|
71
|
+
|
|
53
72
|
/** The artifact texts of one run, keyed by RELATIVE path under `features/<slug>/`. */
|
|
54
73
|
/**
|
|
55
74
|
* The exact heading Step 5 asks for, and the exact heading the check looks for — ONE constant, so
|
|
@@ -306,7 +325,22 @@ function normaliseGradeSign(grade: string): string {
|
|
|
306
325
|
return grade.replace('\u2212', '-');
|
|
307
326
|
}
|
|
308
327
|
|
|
309
|
-
|
|
328
|
+
/**
|
|
329
|
+
* fix-round-1 (codex-r1-verdict finding 2): `'invalid'` is a report that ATTEMPTED a machine verdict
|
|
330
|
+
* (a line starting `QE-VERDICT`, any case/spacing) and got the grammar wrong — `QE-VERDICT: B – final`
|
|
331
|
+
* (en dash + trailing prose), wrong case, no colon, a grade outside A-D. That is a DIFFERENT fact from
|
|
332
|
+
* `'none'` (no attempt at all, legacy prose scan still applies): a malformed attempt must never fall
|
|
333
|
+
* back to guessing a grade from prose — the malformed line is itself evidence the report is unreliable
|
|
334
|
+
* here, and guessing past it would silently launder that unreliability into a confident number.
|
|
335
|
+
*/
|
|
336
|
+
export type GradeReadStatus = 'unique' | 'ambiguous' | 'none' | 'invalid';
|
|
337
|
+
|
|
338
|
+
/**
|
|
339
|
+
* qe-findings-record (ADR-001 D1): where the grade came from. `'verdict-line'` — a machine-written
|
|
340
|
+
* `QE-VERDICT:` line, the source of truth when present. `'prose'` — the pre-existing GRADE_RE scan,
|
|
341
|
+
* unchanged, used only when NO verdict line exists. `'none'` — neither surface names a grade.
|
|
342
|
+
*/
|
|
343
|
+
export type GradeSource = 'verdict-line' | 'prose' | 'none';
|
|
310
344
|
|
|
311
345
|
export interface GradeReading {
|
|
312
346
|
readonly status: GradeReadStatus;
|
|
@@ -314,6 +348,7 @@ export interface GradeReading {
|
|
|
314
348
|
readonly grade: string | null;
|
|
315
349
|
/** Every distinct grade found, normalised — what makes an `ambiguous` verdict inspectable. */
|
|
316
350
|
readonly found: readonly string[];
|
|
351
|
+
readonly source: GradeSource;
|
|
317
352
|
}
|
|
318
353
|
|
|
319
354
|
/**
|
|
@@ -329,6 +364,44 @@ export interface GradeReading {
|
|
|
329
364
|
* report is ambiguous, which is a fact about the report, not a missing number.
|
|
330
365
|
*/
|
|
331
366
|
export function readQeGrade(qeText: string): GradeReading {
|
|
367
|
+
// ADR-001 D1: a machine-written verdict line is the source of truth WHEN it exists — checked
|
|
368
|
+
// first, and the prose scan below never runs when it does. Exactly one -> unique; more than one
|
|
369
|
+
// -> ambiguous ("две строки — ambiguous, никогда «последняя побеждает»", even if both name the
|
|
370
|
+
// SAME grade — two lines is a fact about the report, not a number to reconcile); zero -> the
|
|
371
|
+
// prose scan runs exactly as it always has (NFR-1: 406 pre-existing reports are unaffected) —
|
|
372
|
+
// UNLESS a malformed attempt exists (fix-round-1 finding 2, checked next): the report is `invalid`
|
|
373
|
+
// and the prose fallback is refused, never silently reached.
|
|
374
|
+
const verdictLines = readQeVerdictLines(qeText);
|
|
375
|
+
// Lead delta after Codex r2 (new HIGH #1): a malformed declaration next to a valid one is NOT a
|
|
376
|
+
// unique verdict — `QE-VERDICT: A` + `qe-verdict: D` used to read as unique A. Invalid lines are
|
|
377
|
+
// checked FIRST, whatever the count of valid ones.
|
|
378
|
+
const invalidFirst = findInvalidQeVerdictLines(qeText);
|
|
379
|
+
if (invalidFirst.length > 0) {
|
|
380
|
+
const reasons = invalidFirst.map(
|
|
381
|
+
(l) => `line ${l.line}: ${JSON.stringify(l.text.trim())} does not match "QE-VERDICT: <A|B|C|D><+|-> "`,
|
|
382
|
+
);
|
|
383
|
+
return { status: 'invalid', grade: null, found: [...verdictLines, ...reasons], source: 'verdict-line' };
|
|
384
|
+
}
|
|
385
|
+
if (verdictLines.length === 1) {
|
|
386
|
+
return { status: 'unique', grade: verdictLines[0] as string, found: verdictLines, source: 'verdict-line' };
|
|
387
|
+
}
|
|
388
|
+
if (verdictLines.length > 1) {
|
|
389
|
+
const distinct: string[] = [];
|
|
390
|
+
for (const g of verdictLines) if (!distinct.includes(g)) distinct.push(g);
|
|
391
|
+
return { status: 'ambiguous', grade: null, found: distinct, source: 'verdict-line' };
|
|
392
|
+
}
|
|
393
|
+
|
|
394
|
+
// fix-round-1 finding 2: zero GRAMMATICALLY VALID verdict lines is not yet "no attempt" — a line
|
|
395
|
+
// that clearly opens a verdict declaration but botches the grammar (wrong dash, wrong case, no
|
|
396
|
+
// colon, a grade outside A-D) is `invalid`, and the legacy prose scan below is FORBIDDEN for it.
|
|
397
|
+
// The reason lives inside `found` (GradeReading keeps the same 4-field shape for every status).
|
|
398
|
+
// Lead delta after Codex r2 (#2 PARTIAL): the prose fallback is for LEGACY reports only. A report
|
|
399
|
+
// that already carries the new-format ledger heading but no verdict line is a NEW-format report
|
|
400
|
+
// missing its verdict — `invalid`, never a prose guess.
|
|
401
|
+
if (/^## Findings ledger\s*$/m.test(qeText)) {
|
|
402
|
+
return { status: 'invalid', grade: null, found: ['new-format report (has "## Findings ledger") without a QE-VERDICT line'], source: 'verdict-line' };
|
|
403
|
+
}
|
|
404
|
+
|
|
332
405
|
const found: string[] = [];
|
|
333
406
|
const lines = qeText.split('\n');
|
|
334
407
|
// Line offsets once, so each match maps to ITS line for the negation screen (67d7883d: the
|
|
@@ -352,9 +425,9 @@ export function readQeGrade(qeText: string): GradeReading {
|
|
|
352
425
|
const g = normaliseGradeSign(m[1] as string);
|
|
353
426
|
if (!found.includes(g)) found.push(g);
|
|
354
427
|
}
|
|
355
|
-
if (found.length === 0) return { status: 'none', grade: null, found: [] };
|
|
356
|
-
if (found.length === 1) return { status: 'unique', grade: found[0] as string, found };
|
|
357
|
-
return { status: 'ambiguous', grade: null, found };
|
|
428
|
+
if (found.length === 0) return { status: 'none', grade: null, found: [], source: 'none' };
|
|
429
|
+
if (found.length === 1) return { status: 'unique', grade: found[0] as string, found, source: 'prose' };
|
|
430
|
+
return { status: 'ambiguous', grade: null, found, source: 'prose' };
|
|
358
431
|
}
|
|
359
432
|
|
|
360
433
|
export function extractQeGrade(qeText: string): string | null {
|
|
@@ -451,7 +524,13 @@ export function scoreRun(slug: string, artifacts: RunArtifacts): RunScorecard {
|
|
|
451
524
|
);
|
|
452
525
|
|
|
453
526
|
// 3. Cross-model QE — an independent family reviewed it, and a grade exists.
|
|
454
|
-
const
|
|
527
|
+
const gradeReading = readQeGrade(qeText);
|
|
528
|
+
const grade = gradeReading.grade;
|
|
529
|
+
const findingsParsed = parseQeFindings(qeText);
|
|
530
|
+
const findings: QeFindingsScoreView =
|
|
531
|
+
findingsParsed.status === 'absent'
|
|
532
|
+
? { status: 'absent' }
|
|
533
|
+
: { status: 'present', hollow: findingsParsed.hollow, summary: findingsParsed.summary, refused: findingsParsed.refused };
|
|
455
534
|
if (qeText === '') {
|
|
456
535
|
add('cross-model-qe', 'independent cross-model review with a grade', 'absent', 'no 08_qe_report.md artifact');
|
|
457
536
|
} else {
|
|
@@ -559,11 +638,36 @@ export function scoreRun(slug: string, artifacts: RunArtifacts): RunScorecard {
|
|
|
559
638
|
(grade !== null ? ` · QE grade ${grade}` : ' · no QE grade') +
|
|
560
639
|
(worst.length > 0 ? ` · absent: ${worst.join(', ')}` : '');
|
|
561
640
|
|
|
562
|
-
return { slug, disciplines, qeGrade: grade, mutationEvidence, passed, total, summary };
|
|
641
|
+
return { slug, disciplines, qeGrade: grade, gradeSource: gradeReading.source, findings, mutationEvidence, passed, total, summary };
|
|
563
642
|
}
|
|
564
643
|
|
|
565
644
|
const MARK: Record<DisciplineVerdict, string> = { pass: '✓', partial: '◐', absent: '✗' };
|
|
566
645
|
|
|
646
|
+
/** qe-findings-record FR-4: the one-line summary `dz score` prints for a report's Findings ledger —
|
|
647
|
+
* "findings: 3 HIGH / 2 MEDIUM; 1 refused (line 84: severity "Major" not in dictionary)". Ordered by
|
|
648
|
+
* QE_SEVERITIES (BLOCKER first) so the worst finding always reads first. `null` when there is no
|
|
649
|
+
* table at all — the common case, which earns no noise (same rule as mutationEvidence above). */
|
|
650
|
+
export function renderFindingsLine(findings: QeFindingsScoreView | undefined): string | null {
|
|
651
|
+
if (findings === undefined || findings.status === 'absent') return null;
|
|
652
|
+
// fix-round-1 finding 6: hollow used to short-circuit BEFORE the refused check, so a report whose
|
|
653
|
+
// real ledger table (with CRITICAL rows) was refused as duplicate/outside-section alongside an
|
|
654
|
+
// empty accepted table read as pure "EMPTY" — the refused rows vanished from the printed line, not
|
|
655
|
+
// merely from the tally. Both facts are ALWAYS reported together now; neither hides the other.
|
|
656
|
+
const refusedSuffix = ((): string => {
|
|
657
|
+
if (findings.refused.length === 0) return '';
|
|
658
|
+
const first = findings.refused[0] as QeFindingsRefusedRow;
|
|
659
|
+
const more = findings.refused.length > 1 ? `, +${findings.refused.length - 1} more` : '';
|
|
660
|
+
return `; ${findings.refused.length} refused (line ${first.line}: ${first.reason})${more}`;
|
|
661
|
+
})();
|
|
662
|
+
if (findings.hollow) {
|
|
663
|
+
return `findings: table present but EMPTY (hollow) — worse than no table at all${refusedSuffix}`;
|
|
664
|
+
}
|
|
665
|
+
const bySev = findings.summary.bySeverity;
|
|
666
|
+
const parts = QE_SEVERITIES.filter((s) => (bySev[s] ?? 0) > 0).map((s) => `${bySev[s]} ${s}`);
|
|
667
|
+
const head = parts.length > 0 ? parts.join(' / ') : 'no rows';
|
|
668
|
+
return `findings: ${head}${refusedSuffix}`;
|
|
669
|
+
}
|
|
670
|
+
|
|
567
671
|
export function renderScorecard(card: RunScorecard): string {
|
|
568
672
|
const out: string[] = [];
|
|
569
673
|
out.push(`dz score — ${card.slug} (process scorecard; descriptive-only, never a gate)`);
|
|
@@ -581,6 +685,11 @@ export function renderScorecard(card: RunScorecard): string {
|
|
|
581
685
|
out.push('');
|
|
582
686
|
out.push(` ${card.mutationEvidence.evidence}`);
|
|
583
687
|
}
|
|
688
|
+
const findingsLine = renderFindingsLine(card.findings);
|
|
689
|
+
if (findingsLine !== null) {
|
|
690
|
+
out.push('');
|
|
691
|
+
out.push(` ${findingsLine}${card.gradeSource !== undefined && card.gradeSource !== 'none' ? ` (grade source: ${card.gradeSource})` : ''}`);
|
|
692
|
+
}
|
|
584
693
|
out.push('');
|
|
585
694
|
out.push(` ${card.summary}`);
|
|
586
695
|
return out.join('\n');
|
package/src/vector-tier.ts
CHANGED
|
@@ -1158,13 +1158,87 @@ export interface RankedPattern {
|
|
|
1158
1158
|
|
|
1159
1159
|
const RRF_K = 60;
|
|
1160
1160
|
|
|
1161
|
+
/**
|
|
1162
|
+
* One entry in the total order every post-merge ranking step shares (feature
|
|
1163
|
+
* `recall-parity-tie-break`, FR-1). `evidence` is the SAME three-way rank `mergeHybridHits` has
|
|
1164
|
+
* always used (`both` < lexical-only < semantic-only — lower is stronger), computed once by
|
|
1165
|
+
* {@link evidenceRank} from a hit's `backend`.
|
|
1166
|
+
*/
|
|
1167
|
+
export interface HybridOrderKey {
|
|
1168
|
+
readonly score: number;
|
|
1169
|
+
readonly evidence: number;
|
|
1170
|
+
readonly dzId: string;
|
|
1171
|
+
}
|
|
1172
|
+
|
|
1173
|
+
/** `both` outranks lexical-only outranks semantic-only (mergeHybridHits' own rule, ADR-001 AM-4). */
|
|
1174
|
+
export function evidenceRank(backend: RecallHit['backend']): number {
|
|
1175
|
+
return backend === 'both' ? 0 : backend === 'vector' ? 2 : 1;
|
|
1176
|
+
}
|
|
1177
|
+
|
|
1178
|
+
/**
|
|
1179
|
+
* The ONE deterministic total order recall uses at every step where a tie can occur: fused score
|
|
1180
|
+
* DESC, then EVIDENCE (both > lexical-only > semantic-only), then `dzId` ASC. `mergeHybridHits`
|
|
1181
|
+
* always applied exactly this rule inline; it is exported here (FR-1) so `dampQuarantined`,
|
|
1182
|
+
* `orderHitsForReRank` (the pre-sort `enhance()` runs before its reinforcement/bandit re-rank) and
|
|
1183
|
+
* any other post-merge sort can share the SAME tiebreak instead of an ad hoc score-only comparator
|
|
1184
|
+
* that is deterministic only because its input already arrived pre-ordered — a property that
|
|
1185
|
+
* silently breaks the moment an upstream step feeds it hits in a different order
|
|
1186
|
+
* (recall-parity-tie-break T0: MEASURED, `apply-leg-recall-parity.test.ts` AM-4, a racy
|
|
1187
|
+
* reinforcement-signal read, not a comparator defect, actually explained the observed tail swap —
|
|
1188
|
+
* this comparator is hardening kept from that round; the actual causal fix, per fix-round 1, is the
|
|
1189
|
+
* byte-level store snapshot AM-4 now takes, not this comparator and not an awaited flush).
|
|
1190
|
+
*
|
|
1191
|
+
* FINITE-NUMBER INVARIANT (fix-round 1, LOW finding): plain subtraction (`b.score - a.score`) is
|
|
1192
|
+
* NOT total over `number` — `NaN - x` is `NaN`, and the `||` chain treats a `NaN` term as falsy,
|
|
1193
|
+
* silently SKIPPING it and falling through to the next key as if score had never been compared.
|
|
1194
|
+
* Both terms below use explicit `>`/`<` comparisons instead (correct as-is for ±Infinity — IEEE 754
|
|
1195
|
+
* orders infinities correctly) plus an explicit NaN case: a `NaN` score or evidence is the WEAKEST
|
|
1196
|
+
* possible value on its own axis, so it sorts deterministically LAST, never a coincidental tie.
|
|
1197
|
+
*/
|
|
1198
|
+
function compareScoreDesc(a: number, b: number): number {
|
|
1199
|
+
if (Number.isNaN(a) || Number.isNaN(b)) return Number.isNaN(a) && Number.isNaN(b) ? 0 : Number.isNaN(a) ? 1 : -1;
|
|
1200
|
+
return a > b ? -1 : a < b ? 1 : 0;
|
|
1201
|
+
}
|
|
1202
|
+
function compareEvidenceAsc(a: number, b: number): number {
|
|
1203
|
+
if (Number.isNaN(a) || Number.isNaN(b)) return Number.isNaN(a) && Number.isNaN(b) ? 0 : Number.isNaN(a) ? 1 : -1;
|
|
1204
|
+
return a < b ? -1 : a > b ? 1 : 0;
|
|
1205
|
+
}
|
|
1206
|
+
export function compareHybridHits(a: HybridOrderKey, b: HybridOrderKey): number {
|
|
1207
|
+
return compareScoreDesc(a.score, b.score)
|
|
1208
|
+
|| compareEvidenceAsc(a.evidence, b.evidence)
|
|
1209
|
+
|| (a.dzId < b.dzId ? -1 : a.dzId > b.dzId ? 1 : 0);
|
|
1210
|
+
}
|
|
1211
|
+
|
|
1212
|
+
/**
|
|
1213
|
+
* Sorts `hits` into the shared total order {@link compareHybridHits} defines — used by `enhance()`
|
|
1214
|
+
* BEFORE its reinforcement/bandit re-rank runs (`applyLearningSignalsWithTerms` et al.,
|
|
1215
|
+
* `learning-backend.ts`, out of this fix's edit scope). That re-rank sorts by an ADJUSTED score
|
|
1216
|
+
* with a STABLE tie-break on each hit's ORIGINAL array position — so pre-ordering the input here
|
|
1217
|
+
* makes any tie in the adjusted score resolve in the SAME evidence/dzId order `compareHybridHits`
|
|
1218
|
+
* would give directly, without touching the re-rank's own internals.
|
|
1219
|
+
*
|
|
1220
|
+
* This closes the ordering gap `enhance()` had (recall-parity-tie-break fix-round 1, HIGH finding):
|
|
1221
|
+
* `dampQuarantined` only ran {@link compareHybridHits} when `memory.learning.quarantine` was ON;
|
|
1222
|
+
* with it OFF (the default), `enhance()`'s final order was whatever the re-rank's own
|
|
1223
|
+
* original-index tie-break happened to preserve — invisible from the printed `score` column,
|
|
1224
|
+
* because the re-rank reorders the hit array but never rewrites `.score`. Real (non-tied) score
|
|
1225
|
+
* differences from reinforcement/bandit re-ranking are UNCHANGED by this — it only decides ties.
|
|
1226
|
+
*/
|
|
1227
|
+
export function orderHitsForReRank(hits: readonly HybridHit[], idOf: (p: PatternRecord) => string): HybridHit[] {
|
|
1228
|
+
return [...hits].sort((a, b) => compareHybridHits(
|
|
1229
|
+
{ score: a.score, evidence: evidenceRank(a.backend), dzId: idOf(a.pattern) },
|
|
1230
|
+
{ score: b.score, evidence: evidenceRank(b.backend), dzId: idOf(b.pattern) },
|
|
1231
|
+
));
|
|
1232
|
+
}
|
|
1233
|
+
|
|
1161
1234
|
/**
|
|
1162
1235
|
* Reciprocal Rank Fusion merge: `score(p) = Σ 1/(60 + rank)` over the lists containing `p`
|
|
1163
1236
|
* (semantic ranks weighted by `semanticWeight`). Dedup by id; `backend: 'both'` when a pattern
|
|
1164
1237
|
*
|
|
1165
|
-
* Ordering: fused score, then EVIDENCE (`both` before lexical-only before semantic-only), then id
|
|
1166
|
-
* When `semanticWeight > 1` the lexical top-1 is guaranteed a place in
|
|
1167
|
-
* last seat unless that seat holds a `both` hit. See
|
|
1238
|
+
* Ordering: fused score, then EVIDENCE (`both` before lexical-only before semantic-only), then id
|
|
1239
|
+
* — {@link compareHybridHits}. When `semanticWeight > 1` the lexical top-1 is guaranteed a place in
|
|
1240
|
+
* the result, taken from the last seat unless that seat holds a `both` hit. See
|
|
1241
|
+
* `features/semantic-keeps-exact-hits`.
|
|
1168
1242
|
*
|
|
1169
1243
|
* appears in both lists. DETERMINISTIC (AC-6): ties break on id, so fixed inputs always yield
|
|
1170
1244
|
* the same ordering. Pure — no I/O.
|
|
@@ -1196,12 +1270,14 @@ export function mergeHybridHits(
|
|
|
1196
1270
|
});
|
|
1197
1271
|
// Ties break by EVIDENCE, not by the id alphabet: a hit both legs found outranks one only a single
|
|
1198
1272
|
// leg found. Before this, an exact-term match lost a tie to an arbitrary semantic hit purely
|
|
1199
|
-
// because its id sorted later (ADR-001 AM-4).
|
|
1200
|
-
|
|
1273
|
+
// because its id sorted later (ADR-001 AM-4). Now routed through the shared {@link
|
|
1274
|
+
// compareHybridHits} (FR-1) — same three-part rule, no behaviour change.
|
|
1275
|
+
const evidenceOfAcc = (v: Acc): number => evidenceRank(v.lex !== undefined && v.sem ? 'both' : v.lex ?? 'vector');
|
|
1201
1276
|
const ordered = [...acc.entries()]
|
|
1202
|
-
.sort((a, b) =>
|
|
1203
|
-
|
|
1204
|
-
|
|
1277
|
+
.sort((a, b) => compareHybridHits(
|
|
1278
|
+
{ score: a[1].score, evidence: evidenceOfAcc(a[1]), dzId: a[0] },
|
|
1279
|
+
{ score: b[1].score, evidence: evidenceOfAcc(b[1]), dzId: b[0] },
|
|
1280
|
+
));
|
|
1205
1281
|
const toHit = ([, v]: [string, Acc]): HybridHit => ({
|
|
1206
1282
|
pattern: v.pattern,
|
|
1207
1283
|
backend: v.lex !== undefined && v.sem ? ('both' as const) : v.lex ?? ('vector' as const),
|
|
@@ -1256,6 +1332,27 @@ export function mergeHybridHits(
|
|
|
1256
1332
|
*
|
|
1257
1333
|
* `bandit` is passed ONLY when `memory.learning.banditRerank` is armed; when it is absent this
|
|
1258
1334
|
* function is byte-identical to its pre-feature self — no state file, no lock, no allocation.
|
|
1335
|
+
*
|
|
1336
|
+
* `backend.train()` fires FIRE-AND-FORGET (`void backend.train().catch(() => undefined)`) — this is
|
|
1337
|
+
* a REVERT (recall-parity-tie-break, fix-round 1, MEDIUM finding). An intermediate version of this
|
|
1338
|
+
* fix AWAITED the flush, on the theory that a caller's own reinforcement write landing before it got
|
|
1339
|
+
* an answer would remove the race `apply-leg-recall-parity.test.ts` AM-4 was catching (two tail hits
|
|
1340
|
+
* swapping order between the daemon's `op:recall` reply and a `dz recall --json` invoked a moment
|
|
1341
|
+
* later — MEASURED byte-identical `score` fields, only `uses` differed, so the merge/comparator was
|
|
1342
|
+
* never the cause). MEASURED (lead, 2026-09-15 15:57, temp project, 8 lexical hits, 5 warm runs):
|
|
1343
|
+
* `recallHybrid` took 2 ms with `onRecallHits:false` (no flush at all) vs 57–131 ms with the flush
|
|
1344
|
+
* AWAITED — landing INSIDE the hook's 500 ms `HOOK_RECALL_BUDGET_MS` but a real, avoidable tax on
|
|
1345
|
+
* every recall, for a property the await did not even fully deliver: the awaited write still let a
|
|
1346
|
+
* CLI invoked immediately after the daemon see the DAEMON'S OWN just-computed exposure for that same
|
|
1347
|
+
* query, answering a subtly different question than the daemon had. AM-4 now proves parity by taking
|
|
1348
|
+
* a byte-level SNAPSHOT of the store BEFORE each query's daemon call and pointing `dz recall --json`
|
|
1349
|
+
* at the frozen snapshot (`--project <snapshot>`) — both sides then answer the identical question
|
|
1350
|
+
* from the identical state under PRODUCTION defaults (`onRecallHits` ON), and no write, awaited or
|
|
1351
|
+
* not, can reach the CLI's read. That snapshot is what actually closes the race; this function stays
|
|
1352
|
+
* fire-and-forget, exactly as it always was, because the snapshot makes its timing irrelevant to the
|
|
1353
|
+
* test. `test/lesson-bandit-byte-identity.test.ts` documents the same underlying race in its own
|
|
1354
|
+
* fixture comment and works around it with `onRecallHits: false` there — a narrower, still-valid
|
|
1355
|
+
* isolation for a different test's needs.
|
|
1259
1356
|
*/
|
|
1260
1357
|
function markRecallHits(
|
|
1261
1358
|
projectRoot: string,
|
|
@@ -1356,7 +1453,15 @@ export async function recallHybrid(
|
|
|
1356
1453
|
const q = rec !== undefined && readQuarantineState(rec).quarantined;
|
|
1357
1454
|
return q ? { ...h, score: h.score * memCfg.quarantineDamp, quarantined: true as const } : h;
|
|
1358
1455
|
})
|
|
1359
|
-
|
|
1456
|
+
// FR-1 (recall-parity-tie-break): score-only used to rely on the INCOMING array already
|
|
1457
|
+
// being pre-ordered (native Array.sort is stable, so a genuine tie only stayed put by
|
|
1458
|
+
// accident of arrival order). Routed through the same {@link compareHybridHits} the merge
|
|
1459
|
+
// uses, so damping two equally-scored hits can never reorder them differently from how the
|
|
1460
|
+
// merge itself would have.
|
|
1461
|
+
.sort((a, b) => compareHybridHits(
|
|
1462
|
+
{ score: a.score, evidence: evidenceRank(a.backend), dzId: idOf(a.pattern) },
|
|
1463
|
+
{ score: b.score, evidence: evidenceRank(b.backend), dzId: idOf(b.pattern) },
|
|
1464
|
+
));
|
|
1360
1465
|
};
|
|
1361
1466
|
// lesson-bandit-rerank (ADR-001): the payoff axis. Resolved ONCE per recall; `enabled:false` ⇒
|
|
1362
1467
|
// the Lesson Payoff context is NEVER CONSTRUCTED — the branch is taken BEFORE any work, so the
|
|
@@ -1366,7 +1471,15 @@ export async function recallHybrid(
|
|
|
1366
1471
|
let banditReport: BanditRecallReport | undefined;
|
|
1367
1472
|
let banditExplored: readonly string[] = [];
|
|
1368
1473
|
const enhance = (hits: readonly HybridHit[]): HybridHit[] => {
|
|
1369
|
-
|
|
1474
|
+
// FR-1 (recall-parity-tie-break, fix-round 1, HIGH finding): pre-sort into the shared total
|
|
1475
|
+
// order BEFORE the reinforcement/bandit re-rank runs — see {@link orderHitsForReRank}.
|
|
1476
|
+
// Why the pre-sort is sufficient and not "reliance on a stable sort" (Codex round 2): the
|
|
1477
|
+
// re-rank's own sort in learning-backend.ts is `b.adjusted - a.adjusted || a.i - b.i` — an
|
|
1478
|
+
// EXPLICIT tie-break on the incoming index, so an adjusted-score tie resolves to exactly the
|
|
1479
|
+
// order built here (score → evidence → dzId), by construction, on any engine. That generic
|
|
1480
|
+
// function only knows `score`, so it cannot call compareHybridHits itself.
|
|
1481
|
+
const ordered = orderHitsForReRank(hits, idOf);
|
|
1482
|
+
const candidates = ordered.map((h) => {
|
|
1370
1483
|
const dzId = idOf(h.pattern);
|
|
1371
1484
|
const rec = idToRecord.get(dzId);
|
|
1372
1485
|
return { dzId, score: h.score, reinforcement: rec !== undefined ? readReinforcementState(rec) : undefined };
|
|
@@ -1392,8 +1505,8 @@ export async function recallHybrid(
|
|
|
1392
1505
|
? []
|
|
1393
1506
|
: [{ id: 'delta', byIndex: candidates.map((c) => deltaMap.get(c.dzId) ?? 0), cap: REINFORCE_RRF_CAP }];
|
|
1394
1507
|
// The SAME ranking without the payoff term — the only honest way to say what the term moved.
|
|
1395
|
-
const before = dampQuarantined(applyLearningSignalsWithTerms(
|
|
1396
|
-
const after = dampQuarantined(applyLearningSignalsWithTerms(
|
|
1508
|
+
const before = dampQuarantined(applyLearningSignalsWithTerms(ordered, learning, candidates, REINFORCE_RRF_CAP, baseTerms));
|
|
1509
|
+
const after = dampQuarantined(applyLearningSignalsWithTerms(ordered, learning, candidates, REINFORCE_RRF_CAP, [
|
|
1397
1510
|
...baseTerms,
|
|
1398
1511
|
// ADDED, never assigned, and pre-bounded to [-1,+1] by the ACL — so `squash` is identity and
|
|
1399
1512
|
// `cap` is an EXACT bound on this term's contribution (INV-4).
|
|
@@ -1423,9 +1536,9 @@ export async function recallHybrid(
|
|
|
1423
1536
|
}
|
|
1424
1537
|
if (deltaMap !== undefined) {
|
|
1425
1538
|
const deltaByIndex = candidates.map((c) => deltaMap.get(c.dzId) ?? 0);
|
|
1426
|
-
return dampQuarantined(applyLearningSignalsWithDelta(
|
|
1539
|
+
return dampQuarantined(applyLearningSignalsWithDelta(ordered, learning, candidates, REINFORCE_RRF_CAP, deltaByIndex, REINFORCE_RRF_CAP));
|
|
1427
1540
|
}
|
|
1428
|
-
return dampQuarantined(applyLearningSignals(
|
|
1541
|
+
return dampQuarantined(applyLearningSignals(ordered, learning, candidates, REINFORCE_RRF_CAP));
|
|
1429
1542
|
};
|
|
1430
1543
|
/** The exposure/telemetry payload for `markRecallHits` — `undefined` while disarmed (INV-1). */
|
|
1431
1544
|
const banditEmission = (): { readonly contextKey: string; readonly explored: readonly string[]; readonly moved: number; readonly arms: number; readonly deferred?: boolean } | undefined =>
|