@dzhechkov/harness-core 0.8.36 → 0.8.37
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.dz-manifest.json +195 -75
- package/README.md +235 -8
- package/dist/agentdb-index.d.ts +87 -7
- package/dist/agentdb-index.d.ts.map +1 -1
- package/dist/agentdb-index.js +416 -57
- package/dist/agentdb-index.js.map +1 -1
- package/dist/apply-leg.d.ts +19 -1
- package/dist/apply-leg.d.ts.map +1 -1
- package/dist/apply-leg.js +187 -36
- package/dist/apply-leg.js.map +1 -1
- package/dist/codex-rollouts.d.ts +118 -0
- package/dist/codex-rollouts.d.ts.map +1 -0
- package/dist/codex-rollouts.js +297 -0
- package/dist/codex-rollouts.js.map +1 -0
- package/dist/cost-ledger.d.ts +56 -4
- package/dist/cost-ledger.d.ts.map +1 -1
- package/dist/cost-ledger.js +176 -20
- package/dist/cost-ledger.js.map +1 -1
- package/dist/cross-family-control.d.ts +345 -0
- package/dist/cross-family-control.d.ts.map +1 -0
- package/dist/cross-family-control.js +802 -0
- package/dist/cross-family-control.js.map +1 -0
- package/dist/debt-ratchet.d.ts +53 -0
- package/dist/debt-ratchet.d.ts.map +1 -0
- package/dist/debt-ratchet.js +107 -0
- package/dist/debt-ratchet.js.map +1 -0
- package/dist/embedding-config.d.ts +42 -0
- package/dist/embedding-config.d.ts.map +1 -1
- package/dist/embedding-config.js +106 -10
- package/dist/embedding-config.js.map +1 -1
- package/dist/feature-adr-checkpoints.d.ts +6 -0
- package/dist/feature-adr-checkpoints.d.ts.map +1 -1
- package/dist/feature-adr-checkpoints.js +29 -0
- package/dist/feature-adr-checkpoints.js.map +1 -1
- package/dist/feature-adr-decision-recall.d.ts +2 -2
- package/dist/feature-adr-decision-recall.d.ts.map +1 -1
- package/dist/feature-adr-decision-recall.js +5 -3
- package/dist/feature-adr-decision-recall.js.map +1 -1
- package/dist/feature-adr-envelope.d.ts +96 -0
- package/dist/feature-adr-envelope.d.ts.map +1 -0
- package/dist/feature-adr-envelope.js +183 -0
- package/dist/feature-adr-envelope.js.map +1 -0
- package/dist/feature-adr-routing.d.ts +64 -0
- package/dist/feature-adr-routing.d.ts.map +1 -1
- package/dist/feature-adr-routing.js +122 -2
- package/dist/feature-adr-routing.js.map +1 -1
- package/dist/feature-adr-stage-canon.d.ts +79 -0
- package/dist/feature-adr-stage-canon.d.ts.map +1 -0
- package/dist/feature-adr-stage-canon.js +117 -0
- package/dist/feature-adr-stage-canon.js.map +1 -0
- package/dist/index.d.ts +19 -9
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +13 -5
- package/dist/index.js.map +1 -1
- package/dist/loop-blobs.generated.js +4 -4
- package/dist/loop-blobs.generated.js.map +1 -1
- package/dist/mutation-gate.d.ts +51 -0
- package/dist/mutation-gate.d.ts.map +1 -1
- package/dist/mutation-gate.js +295 -0
- package/dist/mutation-gate.js.map +1 -1
- package/dist/qe-bridge.d.ts.map +1 -1
- package/dist/qe-bridge.js +4 -2
- package/dist/qe-bridge.js.map +1 -1
- package/dist/qe-findings.d.ts +107 -0
- package/dist/qe-findings.d.ts.map +1 -0
- package/dist/qe-findings.js +417 -0
- package/dist/qe-findings.js.map +1 -0
- package/dist/recap.d.ts +1 -1
- package/dist/recap.d.ts.map +1 -1
- package/dist/recap.js +4 -2
- package/dist/recap.js.map +1 -1
- package/dist/round.d.ts +74 -1
- package/dist/round.d.ts.map +1 -1
- package/dist/round.js +112 -4
- package/dist/round.js.map +1 -1
- package/dist/run-records.d.ts +60 -0
- package/dist/run-records.d.ts.map +1 -1
- package/dist/run-records.js +244 -2
- package/dist/run-records.js.map +1 -1
- package/dist/score.d.ts +44 -1
- package/dist/score.d.ts.map +1 -1
- package/dist/score.js +78 -5
- package/dist/score.js.map +1 -1
- package/package.json +1 -1
- package/sbom.json +374 -74
- package/src/agentdb-index.ts +423 -60
- package/src/apply-leg.ts +187 -36
- package/src/codex-rollouts.ts +374 -0
- package/src/cost-ledger.ts +232 -24
- package/src/cross-family-control.ts +960 -0
- package/src/debt-ratchet.ts +143 -0
- package/src/embedding-config.ts +131 -10
- package/src/feature-adr-checkpoints.ts +29 -0
- package/src/feature-adr-decision-recall.ts +6 -4
- package/src/feature-adr-envelope.ts +242 -0
- package/src/feature-adr-routing.ts +139 -2
- package/src/feature-adr-stage-canon.ts +141 -0
- package/src/index.ts +60 -6
- package/src/loop-blobs.generated.ts +4 -4
- package/src/mutation-gate.ts +316 -0
- package/src/qe-bridge.ts +4 -2
- package/src/qe-findings.ts +463 -0
- package/src/recap.ts +10 -3
- package/src/round.ts +165 -6
- package/src/run-records.ts +282 -2
- package/src/score.ts +115 -6
package/src/score.ts
CHANGED
|
@@ -21,6 +21,7 @@
|
|
|
21
21
|
*/
|
|
22
22
|
|
|
23
23
|
import { amendmentIdsIn } from './amendment-trace.js';
|
|
24
|
+
import { QE_SEVERITIES, findInvalidQeVerdictLines, parseQeFindings, readQeVerdictLines, type QeFindingsRefusedRow, type QeFindingsSummary } from './qe-findings.js';
|
|
24
25
|
|
|
25
26
|
export type DisciplineVerdict = 'pass' | 'partial' | 'absent';
|
|
26
27
|
|
|
@@ -45,11 +46,29 @@ export interface RunScorecard {
|
|
|
45
46
|
* stay distinguishable.
|
|
46
47
|
*/
|
|
47
48
|
readonly mutationEvidence?: MutationEvidence;
|
|
49
|
+
/**
|
|
50
|
+
* qe-findings-record FR-4: where `qeGrade` came from — a machine-written `QE-VERDICT:` line, the
|
|
51
|
+
* pre-existing prose scan, or neither. ADDITIVE for the same reason as `mutationEvidence`: every
|
|
52
|
+
* scorecard built before this field existed still satisfies `RunScorecard`.
|
|
53
|
+
*/
|
|
54
|
+
readonly gradeSource?: GradeSource;
|
|
55
|
+
/**
|
|
56
|
+
* qe-findings-record FR-4: the report's Findings ledger, when one exists. `absent` for the 406
|
|
57
|
+
* pre-existing reports that carry neither the table nor the heading (NFR-1) — never omitted, for
|
|
58
|
+
* the same "gate never ran vs a field an older scorer never wrote" reason `mutationEvidence` gives.
|
|
59
|
+
*/
|
|
60
|
+
readonly findings?: QeFindingsScoreView;
|
|
48
61
|
readonly passed: number;
|
|
49
62
|
readonly total: number;
|
|
50
63
|
readonly summary: string;
|
|
51
64
|
}
|
|
52
65
|
|
|
66
|
+
/** A lighter projection of `QeFindingsResult` for the scorecard — summary + refused + hollow, never
|
|
67
|
+
* the full row list (that stays in `parseQeFindings`'s own return value for a caller that wants it). */
|
|
68
|
+
export type QeFindingsScoreView =
|
|
69
|
+
| { readonly status: 'absent' }
|
|
70
|
+
| { readonly status: 'present'; readonly hollow: boolean; readonly summary: QeFindingsSummary; readonly refused: readonly QeFindingsRefusedRow[] };
|
|
71
|
+
|
|
53
72
|
/** The artifact texts of one run, keyed by RELATIVE path under `features/<slug>/`. */
|
|
54
73
|
/**
|
|
55
74
|
* The exact heading Step 5 asks for, and the exact heading the check looks for — ONE constant, so
|
|
@@ -306,7 +325,22 @@ function normaliseGradeSign(grade: string): string {
|
|
|
306
325
|
return grade.replace('\u2212', '-');
|
|
307
326
|
}
|
|
308
327
|
|
|
309
|
-
|
|
328
|
+
/**
|
|
329
|
+
* fix-round-1 (codex-r1-verdict finding 2): `'invalid'` is a report that ATTEMPTED a machine verdict
|
|
330
|
+
* (a line starting `QE-VERDICT`, any case/spacing) and got the grammar wrong — `QE-VERDICT: B – final`
|
|
331
|
+
* (en dash + trailing prose), wrong case, no colon, a grade outside A-D. That is a DIFFERENT fact from
|
|
332
|
+
* `'none'` (no attempt at all, legacy prose scan still applies): a malformed attempt must never fall
|
|
333
|
+
* back to guessing a grade from prose — the malformed line is itself evidence the report is unreliable
|
|
334
|
+
* here, and guessing past it would silently launder that unreliability into a confident number.
|
|
335
|
+
*/
|
|
336
|
+
export type GradeReadStatus = 'unique' | 'ambiguous' | 'none' | 'invalid';
|
|
337
|
+
|
|
338
|
+
/**
|
|
339
|
+
* qe-findings-record (ADR-001 D1): where the grade came from. `'verdict-line'` — a machine-written
|
|
340
|
+
* `QE-VERDICT:` line, the source of truth when present. `'prose'` — the pre-existing GRADE_RE scan,
|
|
341
|
+
* unchanged, used only when NO verdict line exists. `'none'` — neither surface names a grade.
|
|
342
|
+
*/
|
|
343
|
+
export type GradeSource = 'verdict-line' | 'prose' | 'none';
|
|
310
344
|
|
|
311
345
|
export interface GradeReading {
|
|
312
346
|
readonly status: GradeReadStatus;
|
|
@@ -314,6 +348,7 @@ export interface GradeReading {
|
|
|
314
348
|
readonly grade: string | null;
|
|
315
349
|
/** Every distinct grade found, normalised — what makes an `ambiguous` verdict inspectable. */
|
|
316
350
|
readonly found: readonly string[];
|
|
351
|
+
readonly source: GradeSource;
|
|
317
352
|
}
|
|
318
353
|
|
|
319
354
|
/**
|
|
@@ -329,6 +364,44 @@ export interface GradeReading {
|
|
|
329
364
|
* report is ambiguous, which is a fact about the report, not a missing number.
|
|
330
365
|
*/
|
|
331
366
|
export function readQeGrade(qeText: string): GradeReading {
|
|
367
|
+
// ADR-001 D1: a machine-written verdict line is the source of truth WHEN it exists — checked
|
|
368
|
+
// first, and the prose scan below never runs when it does. Exactly one -> unique; more than one
|
|
369
|
+
// -> ambiguous ("две строки — ambiguous, никогда «последняя побеждает»", even if both name the
|
|
370
|
+
// SAME grade — two lines is a fact about the report, not a number to reconcile); zero -> the
|
|
371
|
+
// prose scan runs exactly as it always has (NFR-1: 406 pre-existing reports are unaffected) —
|
|
372
|
+
// UNLESS a malformed attempt exists (fix-round-1 finding 2, checked next): the report is `invalid`
|
|
373
|
+
// and the prose fallback is refused, never silently reached.
|
|
374
|
+
const verdictLines = readQeVerdictLines(qeText);
|
|
375
|
+
// Lead delta after Codex r2 (new HIGH #1): a malformed declaration next to a valid one is NOT a
|
|
376
|
+
// unique verdict — `QE-VERDICT: A` + `qe-verdict: D` used to read as unique A. Invalid lines are
|
|
377
|
+
// checked FIRST, whatever the count of valid ones.
|
|
378
|
+
const invalidFirst = findInvalidQeVerdictLines(qeText);
|
|
379
|
+
if (invalidFirst.length > 0) {
|
|
380
|
+
const reasons = invalidFirst.map(
|
|
381
|
+
(l) => `line ${l.line}: ${JSON.stringify(l.text.trim())} does not match "QE-VERDICT: <A|B|C|D><+|-> "`,
|
|
382
|
+
);
|
|
383
|
+
return { status: 'invalid', grade: null, found: [...verdictLines, ...reasons], source: 'verdict-line' };
|
|
384
|
+
}
|
|
385
|
+
if (verdictLines.length === 1) {
|
|
386
|
+
return { status: 'unique', grade: verdictLines[0] as string, found: verdictLines, source: 'verdict-line' };
|
|
387
|
+
}
|
|
388
|
+
if (verdictLines.length > 1) {
|
|
389
|
+
const distinct: string[] = [];
|
|
390
|
+
for (const g of verdictLines) if (!distinct.includes(g)) distinct.push(g);
|
|
391
|
+
return { status: 'ambiguous', grade: null, found: distinct, source: 'verdict-line' };
|
|
392
|
+
}
|
|
393
|
+
|
|
394
|
+
// fix-round-1 finding 2: zero GRAMMATICALLY VALID verdict lines is not yet "no attempt" — a line
|
|
395
|
+
// that clearly opens a verdict declaration but botches the grammar (wrong dash, wrong case, no
|
|
396
|
+
// colon, a grade outside A-D) is `invalid`, and the legacy prose scan below is FORBIDDEN for it.
|
|
397
|
+
// The reason lives inside `found` (GradeReading keeps the same 4-field shape for every status).
|
|
398
|
+
// Lead delta after Codex r2 (#2 PARTIAL): the prose fallback is for LEGACY reports only. A report
|
|
399
|
+
// that already carries the new-format ledger heading but no verdict line is a NEW-format report
|
|
400
|
+
// missing its verdict — `invalid`, never a prose guess.
|
|
401
|
+
if (/^## Findings ledger\s*$/m.test(qeText)) {
|
|
402
|
+
return { status: 'invalid', grade: null, found: ['new-format report (has "## Findings ledger") without a QE-VERDICT line'], source: 'verdict-line' };
|
|
403
|
+
}
|
|
404
|
+
|
|
332
405
|
const found: string[] = [];
|
|
333
406
|
const lines = qeText.split('\n');
|
|
334
407
|
// Line offsets once, so each match maps to ITS line for the negation screen (67d7883d: the
|
|
@@ -352,9 +425,9 @@ export function readQeGrade(qeText: string): GradeReading {
|
|
|
352
425
|
const g = normaliseGradeSign(m[1] as string);
|
|
353
426
|
if (!found.includes(g)) found.push(g);
|
|
354
427
|
}
|
|
355
|
-
if (found.length === 0) return { status: 'none', grade: null, found: [] };
|
|
356
|
-
if (found.length === 1) return { status: 'unique', grade: found[0] as string, found };
|
|
357
|
-
return { status: 'ambiguous', grade: null, found };
|
|
428
|
+
if (found.length === 0) return { status: 'none', grade: null, found: [], source: 'none' };
|
|
429
|
+
if (found.length === 1) return { status: 'unique', grade: found[0] as string, found, source: 'prose' };
|
|
430
|
+
return { status: 'ambiguous', grade: null, found, source: 'prose' };
|
|
358
431
|
}
|
|
359
432
|
|
|
360
433
|
export function extractQeGrade(qeText: string): string | null {
|
|
@@ -451,7 +524,13 @@ export function scoreRun(slug: string, artifacts: RunArtifacts): RunScorecard {
|
|
|
451
524
|
);
|
|
452
525
|
|
|
453
526
|
// 3. Cross-model QE — an independent family reviewed it, and a grade exists.
|
|
454
|
-
const
|
|
527
|
+
const gradeReading = readQeGrade(qeText);
|
|
528
|
+
const grade = gradeReading.grade;
|
|
529
|
+
const findingsParsed = parseQeFindings(qeText);
|
|
530
|
+
const findings: QeFindingsScoreView =
|
|
531
|
+
findingsParsed.status === 'absent'
|
|
532
|
+
? { status: 'absent' }
|
|
533
|
+
: { status: 'present', hollow: findingsParsed.hollow, summary: findingsParsed.summary, refused: findingsParsed.refused };
|
|
455
534
|
if (qeText === '') {
|
|
456
535
|
add('cross-model-qe', 'independent cross-model review with a grade', 'absent', 'no 08_qe_report.md artifact');
|
|
457
536
|
} else {
|
|
@@ -559,11 +638,36 @@ export function scoreRun(slug: string, artifacts: RunArtifacts): RunScorecard {
|
|
|
559
638
|
(grade !== null ? ` · QE grade ${grade}` : ' · no QE grade') +
|
|
560
639
|
(worst.length > 0 ? ` · absent: ${worst.join(', ')}` : '');
|
|
561
640
|
|
|
562
|
-
return { slug, disciplines, qeGrade: grade, mutationEvidence, passed, total, summary };
|
|
641
|
+
return { slug, disciplines, qeGrade: grade, gradeSource: gradeReading.source, findings, mutationEvidence, passed, total, summary };
|
|
563
642
|
}
|
|
564
643
|
|
|
565
644
|
const MARK: Record<DisciplineVerdict, string> = { pass: '✓', partial: '◐', absent: '✗' };
|
|
566
645
|
|
|
646
|
+
/** qe-findings-record FR-4: the one-line summary `dz score` prints for a report's Findings ledger —
|
|
647
|
+
* "findings: 3 HIGH / 2 MEDIUM; 1 refused (line 84: severity "Major" not in dictionary)". Ordered by
|
|
648
|
+
* QE_SEVERITIES (BLOCKER first) so the worst finding always reads first. `null` when there is no
|
|
649
|
+
* table at all — the common case, which earns no noise (same rule as mutationEvidence above). */
|
|
650
|
+
export function renderFindingsLine(findings: QeFindingsScoreView | undefined): string | null {
|
|
651
|
+
if (findings === undefined || findings.status === 'absent') return null;
|
|
652
|
+
// fix-round-1 finding 6: hollow used to short-circuit BEFORE the refused check, so a report whose
|
|
653
|
+
// real ledger table (with CRITICAL rows) was refused as duplicate/outside-section alongside an
|
|
654
|
+
// empty accepted table read as pure "EMPTY" — the refused rows vanished from the printed line, not
|
|
655
|
+
// merely from the tally. Both facts are ALWAYS reported together now; neither hides the other.
|
|
656
|
+
const refusedSuffix = ((): string => {
|
|
657
|
+
if (findings.refused.length === 0) return '';
|
|
658
|
+
const first = findings.refused[0] as QeFindingsRefusedRow;
|
|
659
|
+
const more = findings.refused.length > 1 ? `, +${findings.refused.length - 1} more` : '';
|
|
660
|
+
return `; ${findings.refused.length} refused (line ${first.line}: ${first.reason})${more}`;
|
|
661
|
+
})();
|
|
662
|
+
if (findings.hollow) {
|
|
663
|
+
return `findings: table present but EMPTY (hollow) — worse than no table at all${refusedSuffix}`;
|
|
664
|
+
}
|
|
665
|
+
const bySev = findings.summary.bySeverity;
|
|
666
|
+
const parts = QE_SEVERITIES.filter((s) => (bySev[s] ?? 0) > 0).map((s) => `${bySev[s]} ${s}`);
|
|
667
|
+
const head = parts.length > 0 ? parts.join(' / ') : 'no rows';
|
|
668
|
+
return `findings: ${head}${refusedSuffix}`;
|
|
669
|
+
}
|
|
670
|
+
|
|
567
671
|
export function renderScorecard(card: RunScorecard): string {
|
|
568
672
|
const out: string[] = [];
|
|
569
673
|
out.push(`dz score — ${card.slug} (process scorecard; descriptive-only, never a gate)`);
|
|
@@ -581,6 +685,11 @@ export function renderScorecard(card: RunScorecard): string {
|
|
|
581
685
|
out.push('');
|
|
582
686
|
out.push(` ${card.mutationEvidence.evidence}`);
|
|
583
687
|
}
|
|
688
|
+
const findingsLine = renderFindingsLine(card.findings);
|
|
689
|
+
if (findingsLine !== null) {
|
|
690
|
+
out.push('');
|
|
691
|
+
out.push(` ${findingsLine}${card.gradeSource !== undefined && card.gradeSource !== 'none' ? ` (grade source: ${card.gradeSource})` : ''}`);
|
|
692
|
+
}
|
|
584
693
|
out.push('');
|
|
585
694
|
out.push(` ${card.summary}`);
|
|
586
695
|
return out.join('\n');
|