@dzhechkov/harness-core 0.8.39 → 0.8.40
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.dz-manifest.json +76 -56
- package/README.md +48 -0
- package/dist/cmd-usage.d.ts.map +1 -1
- package/dist/cmd-usage.js +48 -2
- package/dist/cmd-usage.js.map +1 -1
- package/dist/compounding.d.ts +66 -0
- package/dist/compounding.d.ts.map +1 -1
- package/dist/compounding.js +76 -9
- package/dist/compounding.js.map +1 -1
- package/dist/doctor-instrument.d.ts +65 -0
- package/dist/doctor-instrument.d.ts.map +1 -0
- package/dist/doctor-instrument.js +91 -0
- package/dist/doctor-instrument.js.map +1 -0
- package/dist/feature-tier.d.ts.map +1 -1
- package/dist/feature-tier.js +13 -1
- package/dist/feature-tier.js.map +1 -1
- package/dist/guard.d.ts +17 -0
- package/dist/guard.d.ts.map +1 -1
- package/dist/guard.js +15 -0
- package/dist/guard.js.map +1 -1
- package/dist/index.d.ts +5 -3
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +3 -2
- package/dist/index.js.map +1 -1
- package/dist/mutation-gate.d.ts +30 -0
- package/dist/mutation-gate.d.ts.map +1 -1
- package/dist/mutation-gate.js +45 -1
- package/dist/mutation-gate.js.map +1 -1
- package/dist/operations.d.ts +13 -0
- package/dist/operations.d.ts.map +1 -1
- package/dist/operations.js +88 -2
- package/dist/operations.js.map +1 -1
- package/dist/registry.d.ts +58 -0
- package/dist/registry.d.ts.map +1 -1
- package/dist/registry.js +62 -5
- package/dist/registry.js.map +1 -1
- package/dist/release.d.ts +18 -0
- package/dist/release.d.ts.map +1 -1
- package/dist/release.js +30 -0
- package/dist/release.js.map +1 -1
- package/dist/round-exec.d.ts +10 -0
- package/dist/round-exec.d.ts.map +1 -1
- package/dist/round-exec.js +3 -2
- package/dist/round-exec.js.map +1 -1
- package/dist/round.d.ts +19 -0
- package/dist/round.d.ts.map +1 -1
- package/dist/round.js +1 -0
- package/dist/round.js.map +1 -1
- package/package.json +1 -1
- package/sbom.json +105 -55
- package/src/cmd-usage.ts +52 -2
- package/src/compounding.ts +112 -9
- package/src/doctor-instrument.ts +153 -0
- package/src/feature-tier.ts +14 -1
- package/src/guard.ts +24 -0
- package/src/index.ts +5 -2
- package/src/mutation-gate.ts +54 -1
- package/src/operations.ts +85 -3
- package/src/registry.ts +91 -1
- package/src/release.ts +36 -0
- package/src/round-exec.ts +13 -2
- package/src/round.ts +20 -0
package/src/cmd-usage.ts
CHANGED
|
@@ -153,6 +153,12 @@ interface RuleUsage {
|
|
|
153
153
|
readonly stats: ReadonlyMap<string, CmdUsageStat>;
|
|
154
154
|
readonly skipped: number;
|
|
155
155
|
readonly outOfRange: number;
|
|
156
|
+
/**
|
|
157
|
+
* Audit rows inside the window that recorded which rules they EVALUATED. Zero means this report
|
|
158
|
+
* has no evidence about rules at all — which is a different answer from "the rule is unused", and
|
|
159
|
+
* the distinction is the whole point of counting it (backlog 1bee49dd).
|
|
160
|
+
*/
|
|
161
|
+
readonly evaluationRows: number;
|
|
156
162
|
}
|
|
157
163
|
|
|
158
164
|
const REPO_BOUNDARY_IO = {
|
|
@@ -401,9 +407,13 @@ export function loadDeadwoodAllowlist(json: string): DeadwoodAllowlistEntry[] {
|
|
|
401
407
|
function parseGuardAuditUsage(text: string, weeks: number, now: Date): RuleUsage {
|
|
402
408
|
const auditTimestamps: string[] = [];
|
|
403
409
|
const hits: CmdUsageInvocationRecord[] = [];
|
|
410
|
+
let evaluationRows = 0;
|
|
404
411
|
let skipped = 0;
|
|
405
412
|
let outOfRange = 0;
|
|
406
413
|
const newestAllowed = now.getTime() + DEADWOOD_FUTURE_TOLERANCE_MS;
|
|
414
|
+
// Same window arithmetic `foldCmdUsage` uses, so "counted as evidence" and "counted as a run"
|
|
415
|
+
// cannot disagree about which rows are inside.
|
|
416
|
+
const windowStart = now.getTime() - Math.max(0, weeks) * 7 * DAY_MS;
|
|
407
417
|
for (const line of text.split('\n')) {
|
|
408
418
|
if (line.trim() === '') continue;
|
|
409
419
|
let row: Record<string, unknown>;
|
|
@@ -425,17 +435,40 @@ function parseGuardAuditUsage(text: string, weeks: number, now: Date): RuleUsage
|
|
|
425
435
|
continue;
|
|
426
436
|
}
|
|
427
437
|
auditTimestamps.push(row.ts);
|
|
438
|
+
// A rule's HEALTHY state is silence, so firing cannot measure whether it is alive. `evaluated`
|
|
439
|
+
// records the rules that actually got their turn; rows written before the field existed carry
|
|
440
|
+
// none, and they contribute evidence about firing only.
|
|
441
|
+
//
|
|
442
|
+
// Two conditions, both named by cross-family review (Codex gpt-5.6-sol, 2026-09-21), both of
|
|
443
|
+
// which turn this evidence into a false accusation if skipped:
|
|
444
|
+
// · the row must be INSIDE the window. An instrumented row older than `now - weeks` yields no
|
|
445
|
+
// in-window runs, so counting it as evidence would let one ancient row flip every absent
|
|
446
|
+
// rule from "cannot judge" to "dead".
|
|
447
|
+
// · the array must carry a USABLE id. `evaluated: []` is a row that recorded nothing; treating
|
|
448
|
+
// it as evidence is the same false accusation by a shorter path.
|
|
449
|
+
const evaluatedIds = new Set<string>();
|
|
450
|
+
if (Array.isArray(row.evaluated)) {
|
|
451
|
+
for (const value of row.evaluated) {
|
|
452
|
+
if (typeof value === 'string' && value.trim() !== '') evaluatedIds.add(value);
|
|
453
|
+
}
|
|
454
|
+
}
|
|
455
|
+
if (evaluatedIds.size > 0 && tsMs >= windowStart) evaluationRows += 1;
|
|
456
|
+
for (const id of evaluatedIds) {
|
|
457
|
+
hits.push({ kind: 'cmd', cmd: id, ts: row.ts, v: CMD_USAGE_SCHEMA });
|
|
458
|
+
}
|
|
428
459
|
const violations = Array.isArray(row.violations) ? row.violations : [];
|
|
429
460
|
for (const value of violations) {
|
|
430
461
|
const rule = typeof value === 'object' && value !== null
|
|
431
462
|
? (value as { rule?: unknown }).rule
|
|
432
463
|
: undefined;
|
|
433
|
-
|
|
464
|
+
// One guard run is ONE run. A rule that both evaluated and fired in the same row would be
|
|
465
|
+
// counted twice — 100 warning evaluations reported as 200 runs (same review, second finding).
|
|
466
|
+
if (typeof rule === 'string' && rule.trim() !== '' && !evaluatedIds.has(rule)) {
|
|
434
467
|
hits.push({ kind: 'cmd', cmd: rule, ts: row.ts, v: CMD_USAGE_SCHEMA });
|
|
435
468
|
}
|
|
436
469
|
}
|
|
437
470
|
}
|
|
438
|
-
return { auditTimestamps, stats: foldCmdUsage(hits, weeks, now), skipped, outOfRange };
|
|
471
|
+
return { auditTimestamps, stats: foldCmdUsage(hits, weeks, now), skipped, outOfRange, evaluationRows };
|
|
439
472
|
}
|
|
440
473
|
|
|
441
474
|
function timestampDepthDays(timestamps: readonly string[], now: Date): number {
|
|
@@ -607,6 +640,23 @@ export function buildDeadwoodReport(input: DeadwoodInput): DeadwoodReport {
|
|
|
607
640
|
});
|
|
608
641
|
continue;
|
|
609
642
|
}
|
|
643
|
+
// A rule the window has no EVALUATION evidence for is unjudged, not unused. Firing is the wrong
|
|
644
|
+
// signal for a guard (silence is its healthy state), so without `evaluated` rows the only honest
|
|
645
|
+
// answer is "this report cannot judge the rule" — exactly what the skill surface already says.
|
|
646
|
+
if (item.kind === 'rule' && rules.evaluationRows === 0
|
|
647
|
+
&& (rules.stats.get(item.surface)?.runsInWindow ?? 0) === 0
|
|
648
|
+
// An explicit allowlist entry is an operator's standing statement about this surface; it keeps
|
|
649
|
+
// its own wording. Only the ACCUSING path — "zero usage, consider deprecating" — is withdrawn.
|
|
650
|
+
&& !allowlist.has(allowlistKey(item.kind, item.surface))) {
|
|
651
|
+
noInstrumentation.push({
|
|
652
|
+
state: 'no-instrumentation',
|
|
653
|
+
surface: item.surface,
|
|
654
|
+
kind: item.kind,
|
|
655
|
+
reason: 'no guard-audit row in this window recorded which rules it evaluated; a rule that '
|
|
656
|
+
+ 'never fires may be a healthy safety net, so firing alone cannot judge it',
|
|
657
|
+
});
|
|
658
|
+
continue;
|
|
659
|
+
}
|
|
610
660
|
classifyInstrumented(
|
|
611
661
|
{ surface: item.surface, kind: item.kind },
|
|
612
662
|
item.kind === 'command'
|
package/src/compounding.ts
CHANGED
|
@@ -15,7 +15,7 @@
|
|
|
15
15
|
* Everything here is PURE: callers gather facts (files, store rows); this module only computes.
|
|
16
16
|
*/
|
|
17
17
|
|
|
18
|
-
import { EVENT_CHAIN_SCOPE, verifyEventChainText } from './event-chain.js';
|
|
18
|
+
import { EVENT_CHAIN_SCOPE, classifyChainDefects, verifyEventChainText } from './event-chain.js';
|
|
19
19
|
import {
|
|
20
20
|
isOffsetIsoTimestamp,
|
|
21
21
|
type PromotionAcceptanceEvidence,
|
|
@@ -160,15 +160,36 @@ export interface GuardEvent {
|
|
|
160
160
|
readonly verdict: string;
|
|
161
161
|
readonly rules: readonly string[]; // violated rule ids
|
|
162
162
|
readonly violations?: readonly { readonly rule: string; readonly contentAnchor?: string }[];
|
|
163
|
+
/**
|
|
164
|
+
* 1-based position of this record among the log's non-empty lines — the ONLY thing that can place
|
|
165
|
+
* it relative to a chain defect. Absent when the caller read the rows without a chain.
|
|
166
|
+
*/
|
|
167
|
+
readonly chainLine?: number;
|
|
163
168
|
}
|
|
164
169
|
|
|
165
170
|
export type FunnelEvidenceSource<T> =
|
|
166
171
|
| { readonly status: 'measured'; readonly rows: readonly T[] }
|
|
167
172
|
| { readonly status: 'not-measured'; readonly reason: string };
|
|
168
173
|
|
|
174
|
+
/**
|
|
175
|
+
* Where the guard journal's chain damage sits, so a PERIOD can be judged instead of the whole FILE.
|
|
176
|
+
*
|
|
177
|
+
* A log damaged once in March and unbroken since is not evidence against September's rows, and
|
|
178
|
+
* refusing to measure September because of March is the same "verdict answers a different question"
|
|
179
|
+
* defect the chain headline was fixed for (backlog b38dd3ba, MEASURED 2026-09-21: 28 defects, all
|
|
180
|
+
* before a run of 1169 unbroken records, suppressed BOTH measured months).
|
|
181
|
+
*/
|
|
182
|
+
export interface GuardAuditChainWindow {
|
|
183
|
+
/** First non-empty line of the current unbroken run: one past the last defect. */
|
|
184
|
+
readonly runFrom: number;
|
|
185
|
+
/** Total defects in the file. Zero means the window imposes nothing. */
|
|
186
|
+
readonly defects: number;
|
|
187
|
+
}
|
|
188
|
+
|
|
169
189
|
export interface LessonToRuleFunnelFacts {
|
|
170
190
|
readonly promotionRuns: FunnelEvidenceSource<PromotionRunEvidence>;
|
|
171
191
|
readonly guardAudits: FunnelEvidenceSource<GuardEvent>;
|
|
192
|
+
readonly guardAuditChain?: GuardAuditChainWindow;
|
|
172
193
|
readonly promotionAcceptances?: readonly PromotionAcceptanceEvidence[];
|
|
173
194
|
readonly truncatedPromotionPeriods?: readonly string[];
|
|
174
195
|
readonly acceptanceHistoryComplete?: boolean;
|
|
@@ -274,6 +295,20 @@ export interface EvidenceChainHealth {
|
|
|
274
295
|
readonly preChainPrefix: number;
|
|
275
296
|
readonly defects: number;
|
|
276
297
|
readonly defectKinds: readonly string[];
|
|
298
|
+
/**
|
|
299
|
+
* WHERE the defects sit relative to the log's current unbroken run, and HOW MUCH of a run that is.
|
|
300
|
+
* Without this a bare `FAILED` over a log whose damage is entirely historical reads as "today's
|
|
301
|
+
* numbers are garbage", while `dz chain` over the SAME file says "healed … verdicts over those are
|
|
302
|
+
* sound" — MEASURED 2026-09-20 on `.dz/guard-audit.jsonl`: 28 defects, all before the current run,
|
|
303
|
+
* 1095 unbroken records after them; one instrument printed FAILED, the other healed, both exit 0
|
|
304
|
+
* (backlog 79ce6262). Neither was lying; neither named its WINDOW. `event-chain.ts` says it
|
|
305
|
+
* outright: a caller that reports soundness without printing the run size overclaims on its behalf,
|
|
306
|
+
* and the same holds for a caller that reports damage without printing where the damage sits.
|
|
307
|
+
*/
|
|
308
|
+
readonly defectsBeforeRun: number;
|
|
309
|
+
readonly defectsInRun: number;
|
|
310
|
+
/** Records in the current unbroken run — the evidence behind any "sound for today" reading. */
|
|
311
|
+
readonly runRecords: number;
|
|
277
312
|
}
|
|
278
313
|
|
|
279
314
|
export interface InstrumentationHealth {
|
|
@@ -394,6 +429,13 @@ function executionMeasurement(
|
|
|
394
429
|
if (periodAudits.length === 0) {
|
|
395
430
|
return LESSON_TO_RULE_FUNNEL_POLICY.unavailable(`guard-audit-not-recorded:${period}`);
|
|
396
431
|
}
|
|
432
|
+
// Damage that PRECEDES this period's rows says nothing about them; damage that touches them does.
|
|
433
|
+
// A row with no position cannot be placed, and unplaceable is not the same as sound — it refuses.
|
|
434
|
+
const chain = facts.guardAuditChain;
|
|
435
|
+
if (chain !== undefined && chain.defects > 0
|
|
436
|
+
&& periodAudits.some((row) => row.chainLine === undefined || row.chainLine < chain.runFrom)) {
|
|
437
|
+
return LESSON_TO_RULE_FUNNEL_POLICY.unavailable(`guard-audit-chain-damaged:${period}`);
|
|
438
|
+
}
|
|
397
439
|
const audits = periodAudits.filter((row) => row.op === 'publish');
|
|
398
440
|
if (audits.length === 0) {
|
|
399
441
|
return LESSON_TO_RULE_FUNNEL_POLICY.unavailable(`guard-publish-not-recorded:${period}`);
|
|
@@ -594,6 +636,11 @@ export function assembleCompoundingReport(facts: CompoundingFacts): CompoundingR
|
|
|
594
636
|
preChainPrefix: v.preChainPrefix,
|
|
595
637
|
defects: v.defects.length,
|
|
596
638
|
defectKinds: [...new Set(v.defects.map((d) => d.kind))],
|
|
639
|
+
...((age) => ({
|
|
640
|
+
defectsBeforeRun: age.beforeRun.length,
|
|
641
|
+
defectsInRun: age.inRun.length,
|
|
642
|
+
runRecords: age.runRecords,
|
|
643
|
+
}))(classifyChainDefects(v, v.lines)),
|
|
597
644
|
};
|
|
598
645
|
});
|
|
599
646
|
|
|
@@ -623,13 +670,7 @@ export function assembleCompoundingReport(facts: CompoundingFacts): CompoundingR
|
|
|
623
670
|
trajectory.length > 0 ? `guard: ${improvedRules}/${trajectory.length} rules recur less in the later half` : 'guard: not enough history',
|
|
624
671
|
`cold-vs-warm: ${replay.verdict === 'insufficient-data' ? 'INSUFFICIENT DATA (accruing)' : 'READY to measure'}`,
|
|
625
672
|
instrumentation.applyLegLive ? 'apply leg: live' : 'apply leg: STALE — fix the instrumentation before trusting anything above',
|
|
626
|
-
...(chains.length === 0
|
|
627
|
-
? []
|
|
628
|
-
: [
|
|
629
|
-
instrumentation.chainsOk
|
|
630
|
-
? 'evidence chain: verified'
|
|
631
|
-
: 'evidence chain: CORRUPT — the numbers above are computed from a damaged log',
|
|
632
|
-
]),
|
|
673
|
+
...(chains.length === 0 ? [] : [chainHeadline(chains)]),
|
|
633
674
|
].join(' · ');
|
|
634
675
|
|
|
635
676
|
return { pool, guardTrajectory: trajectory, replay, instrumentation, lessonToRuleFunnel, verdict };
|
|
@@ -671,6 +712,68 @@ function renderFunnelPeriodMeasurements(row: LessonToRuleFunnelPeriod): string {
|
|
|
671
712
|
return `${renderPromotionMeasurements(row)} · ${renderFunnelMeasurement('executions', row.executions)}`;
|
|
672
713
|
}
|
|
673
714
|
|
|
715
|
+
/**
|
|
716
|
+
* The HEADLINE verdict over every evidence log — three-valued, because two values lied.
|
|
717
|
+
*
|
|
718
|
+
* MEASURED 2026-09-21 on `.dz/guard-audit.jsonl`: 28 defects, the LAST of them dated 2026-09-05,
|
|
719
|
+
* followed by more than a thousand unbroken records. The old headline read
|
|
720
|
+
* "CORRUPT — the numbers above are computed from a damaged log", which is true of the FILE'S
|
|
721
|
+
* HISTORY and false about the numbers it was printed next to. The distinction already existed one
|
|
722
|
+
* function below, in {@link chainVerdictPhrase}; it simply never reached the line a reader sees
|
|
723
|
+
* first. That is the same defect class this report exists to find: a verdict answering a different
|
|
724
|
+
* question than the one it appears to answer.
|
|
725
|
+
*/
|
|
726
|
+
/**
|
|
727
|
+
* Whether the report's OWN numbers may be trusted, as a value the caller can turn into an exit code.
|
|
728
|
+
*
|
|
729
|
+
* Backlog 79ce6262 named the defect: the report printed "the numbers above are computed from a
|
|
730
|
+
* damaged log" and exited 0 anyway — a tool announcing its own output untrustworthy and reporting
|
|
731
|
+
* success. That record offered two lawful cures and asked which applies. Both do, on different
|
|
732
|
+
* branches, and only the three-valued verdict lets them coexist: damage BEHIND the current run
|
|
733
|
+
* narrows the WORDING (the numbers stand, exit 0), damage INSIDE it makes the numbers genuinely
|
|
734
|
+
* unreliable and must reach the exit code.
|
|
735
|
+
*
|
|
736
|
+
* `'trusted'` ⇒ 0. `'unreliable'` ⇒ a non-zero the caller chooses — the run succeeded, the verdict
|
|
737
|
+
* cannot be relied on, which is this repository's INCONCLUSIVE shape, not its failure shape.
|
|
738
|
+
*/
|
|
739
|
+
export function chainTrust(chains: readonly EvidenceChainHealth[]): 'trusted' | 'unreliable' {
|
|
740
|
+
return chains.some((c) => !c.ok && c.defectsInRun > 0) ? 'unreliable' : 'trusted';
|
|
741
|
+
}
|
|
742
|
+
|
|
743
|
+
export function chainHeadline(chains: readonly EvidenceChainHealth[]): string {
|
|
744
|
+
const broken = chains.filter((c) => !c.ok);
|
|
745
|
+
if (broken.length === 0) return 'evidence chain: verified';
|
|
746
|
+
const live = broken.filter((c) => c.defectsInRun > 0);
|
|
747
|
+
if (live.length === 0) {
|
|
748
|
+
const runs = broken.reduce((n, c) => n + c.runRecords, 0);
|
|
749
|
+
const defects = broken.reduce((n, c) => n + c.defects, 0);
|
|
750
|
+
return `evidence chain: damaged EARLIER — ${defects} defect(s), none inside the current run of `
|
|
751
|
+
+ `${runs} unbroken record(s); the numbers above stand, the file's history does not`;
|
|
752
|
+
}
|
|
753
|
+
const inRun = live.reduce((n, c) => n + c.defectsInRun, 0);
|
|
754
|
+
return `evidence chain: CORRUPT — ${inRun} defect(s) INSIDE the current run; the numbers above are `
|
|
755
|
+
+ 'computed from a damaged log';
|
|
756
|
+
}
|
|
757
|
+
|
|
758
|
+
/**
|
|
759
|
+
* The verdict phrase for one evidence log, with its WINDOW named. A bare `FAILED` over damage that an
|
|
760
|
+
* unbroken run has already followed is true of the FILE and misleading about TODAY — see
|
|
761
|
+
* {@link EvidenceChainHealth.defectsBeforeRun}.
|
|
762
|
+
*/
|
|
763
|
+
export function chainVerdictPhrase(c: EvidenceChainHealth): string {
|
|
764
|
+
if (c.ok) return 'verified';
|
|
765
|
+
const kinds = `[${c.defectKinds.join(', ')}]`;
|
|
766
|
+
if (c.defectsInRun === 0) {
|
|
767
|
+
return `DAMAGED EARLIER — ${c.defects} defect(s) ${kinds}, all BEFORE the current run of `
|
|
768
|
+
+ `${c.runRecords} unbroken record(s); numbers over that run stand, the file's history does not`;
|
|
769
|
+
}
|
|
770
|
+
if (c.defectsBeforeRun === 0) {
|
|
771
|
+
return `FAILED — ${c.defects} defect(s) ${kinds} with NO sound records after them`;
|
|
772
|
+
}
|
|
773
|
+
return `FAILED — ${c.defects} defect(s) ${kinds}: ${c.defectsInRun} inside the current run of `
|
|
774
|
+
+ `${c.runRecords} record(s), ${c.defectsBeforeRun} before it`;
|
|
775
|
+
}
|
|
776
|
+
|
|
674
777
|
export function renderCompoundingReport(r: CompoundingReport): string {
|
|
675
778
|
const out: string[] = [];
|
|
676
779
|
out.push('dz compounding — does the learning loop pay? (honest report: gates without data say so)');
|
|
@@ -695,7 +798,7 @@ export function renderCompoundingReport(r: CompoundingReport): string {
|
|
|
695
798
|
);
|
|
696
799
|
for (const c of r.instrumentation.chains) {
|
|
697
800
|
out.push(
|
|
698
|
-
` EVIDENCE CHAIN ${c.log}: ${
|
|
801
|
+
` EVIDENCE CHAIN ${c.log}: ${chainVerdictPhrase(c)}` +
|
|
699
802
|
` · ${c.chained} chained · ${c.preChainPrefix} pre-chain (uncovered)`,
|
|
700
803
|
);
|
|
701
804
|
}
|
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
import { isAbsolute, relative, sep } from 'node:path';
|
|
2
|
+
|
|
3
|
+
export type InstrumentFreshness = 'same' | 'stale' | 'unknown';
|
|
4
|
+
|
|
5
|
+
export interface InstrumentCheckInput {
|
|
6
|
+
/** realpath of the running binary, or null when it could not be resolved. */
|
|
7
|
+
readonly binPath: string | null;
|
|
8
|
+
/** version from the package.json that owns binPath, or null. */
|
|
9
|
+
readonly binVersion: string | null;
|
|
10
|
+
/** version from packages/@dzhechkov/harness-cli/package.json, or null outside the monorepo. */
|
|
11
|
+
readonly treeVersion: string | null;
|
|
12
|
+
/** absolute, realpath'd project root. */
|
|
13
|
+
readonly projectRoot: string;
|
|
14
|
+
/**
|
|
15
|
+
* Whether `projectRoot` above really IS realpath'd. The caller resolves it and falls back to a
|
|
16
|
+
* plain resolve when that throws; with a symlinked root that fallback compares a realpath'd
|
|
17
|
+
* binary against a non-realpath'd root, and an IN-TREE binary then looks external. Containment is
|
|
18
|
+
* undecidable in that state, so it is answered `unknown` rather than guessed either way.
|
|
19
|
+
* Named by independent review (Claude Sonnet, 2026-09-20).
|
|
20
|
+
*/
|
|
21
|
+
readonly projectRootRealpathed: boolean;
|
|
22
|
+
readonly isMonorepo: boolean;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* `level` is the WHOLE verdict — the caller renders it and never re-derives one of its own. That is
|
|
27
|
+
* deliberate: the first wiring of this module answered `unknown` with `ok: false` while this decider
|
|
28
|
+
* answered `ok`, and two answers to one question is the defect class this repo pays for most often.
|
|
29
|
+
*
|
|
30
|
+
* Three values, because two would lie: `ok` (the instrument is the tree's, or the check does not
|
|
31
|
+
* apply here), `warn` (measured stale — worth saying loudly, never worth failing a health command
|
|
32
|
+
* that gates other people's CI), `unknown` (the evidence could not be gathered — never rendered as
|
|
33
|
+
* a pass, and never as a failure either, since absence of evidence is not a defect).
|
|
34
|
+
*/
|
|
35
|
+
export interface InstrumentCheckResult {
|
|
36
|
+
readonly freshness: InstrumentFreshness;
|
|
37
|
+
readonly level: 'ok' | 'warn' | 'unknown';
|
|
38
|
+
readonly detail: string;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
export interface RankingStateCheckInput {
|
|
42
|
+
readonly flagOn: boolean;
|
|
43
|
+
/** absolute path the state was looked for at. */
|
|
44
|
+
readonly statePath: string;
|
|
45
|
+
readonly stateExists: boolean;
|
|
46
|
+
/** resolved binary path, for the detail — null when unknown. */
|
|
47
|
+
readonly binPath: string | null;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/** Ranking state has no freshness concept, so it deliberately has its own result type. */
|
|
51
|
+
export interface RankingStateCheckResult {
|
|
52
|
+
readonly level: 'ok' | 'warn';
|
|
53
|
+
readonly detail: string;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/** Strip semver build metadata: `0.8.32+a1b2c3` and `0.8.32` are the same release. */
|
|
57
|
+
function withoutBuildMetadata(version: string): string {
|
|
58
|
+
const plus = version.indexOf('+');
|
|
59
|
+
return plus < 0 ? version : version.slice(0, plus);
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
function isInsideRoot(candidate: string, root: string): boolean {
|
|
63
|
+
const fromRoot = relative(root, candidate);
|
|
64
|
+
return fromRoot === '' || (!isAbsolute(fromRoot) && fromRoot !== '..' && !fromRoot.startsWith(`..${sep}`));
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* Decide whether the executable answering `dz doctor` is the workspace's current instrument.
|
|
69
|
+
*
|
|
70
|
+
* LIMITS NAMED BY INDEPENDENT REVIEW (Claude Sonnet, 2026-09-20), none of them hidden behind a
|
|
71
|
+
* passing test:
|
|
72
|
+
* - Containment is a case-SENSITIVE path comparison. On a case-insensitive filesystem, or where the
|
|
73
|
+
* same location is reachable under two path forms, an in-tree binary can read as external. This
|
|
74
|
+
* repo runs on Linux; the cost of being wrong is one extra `warn` line and never an exit code.
|
|
75
|
+
* - The caller attributes a version by walking up from the binary to the NEAREST `package.json`.
|
|
76
|
+
* A shim in package A that loads package B's code is attributed to A, and a broken install with
|
|
77
|
+
* no own manifest is attributed to whatever ancestor has one. The detail always prints the
|
|
78
|
+
* resolved binary path so a reader can see which file was actually measured.
|
|
79
|
+
*/
|
|
80
|
+
export function checkInstrumentFreshness(input: InstrumentCheckInput): InstrumentCheckResult {
|
|
81
|
+
const binary = input.binPath ?? '(unresolved)';
|
|
82
|
+
const binaryVersion = input.binVersion ?? 'unknown';
|
|
83
|
+
const treeVersion = input.treeVersion ?? 'unknown';
|
|
84
|
+
|
|
85
|
+
if (!input.isMonorepo) {
|
|
86
|
+
return {
|
|
87
|
+
freshness: 'unknown',
|
|
88
|
+
level: 'ok',
|
|
89
|
+
detail: `not applicable in a consumer project: no packages/@dzhechkov tree version to compare; resolved binary ${binary}; binary version ${binaryVersion}; tree version ${treeVersion}`,
|
|
90
|
+
};
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
if (input.binPath !== null && input.projectRootRealpathed && isInsideRoot(input.binPath, input.projectRoot)) {
|
|
94
|
+
return {
|
|
95
|
+
freshness: 'same',
|
|
96
|
+
level: 'ok',
|
|
97
|
+
detail: `resolved binary ${input.binPath} is inside project root ${input.projectRoot}; binary version ${binaryVersion}; tree version ${treeVersion}; this binary is the tree instrument`,
|
|
98
|
+
};
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
if (!input.projectRootRealpathed) {
|
|
102
|
+
return {
|
|
103
|
+
freshness: 'unknown',
|
|
104
|
+
level: 'unknown',
|
|
105
|
+
detail: `project root ${input.projectRoot} could not be resolved through its symlinks, so it cannot be told whether the answering binary is the tree's own; version could not be determined safely; resolved binary ${binary}; binary version ${binaryVersion}; tree version ${treeVersion}`,
|
|
106
|
+
};
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
if (input.binPath === null || input.binVersion === null || input.treeVersion === null) {
|
|
110
|
+
return {
|
|
111
|
+
freshness: 'unknown',
|
|
112
|
+
level: 'unknown',
|
|
113
|
+
detail: `instrument version could not be determined; resolved binary ${binary}; binary version ${binaryVersion}; tree version ${treeVersion}`,
|
|
114
|
+
};
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
// Semver says build metadata after `+` does not participate in equality, so a pipeline that
|
|
118
|
+
// stamps a commit hash onto the version must not read as a stale instrument.
|
|
119
|
+
if (withoutBuildMetadata(input.binVersion) === withoutBuildMetadata(input.treeVersion)) {
|
|
120
|
+
return {
|
|
121
|
+
freshness: 'same',
|
|
122
|
+
level: 'ok',
|
|
123
|
+
detail: `resolved binary ${input.binPath}; binary version ${input.binVersion}; tree version ${input.treeVersion}; versions match`,
|
|
124
|
+
};
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
return {
|
|
128
|
+
freshness: 'stale',
|
|
129
|
+
level: 'warn',
|
|
130
|
+
detail: `resolved binary ${input.binPath}; binary version ${input.binVersion}; tree version ${input.treeVersion}; the answering instrument is stale`,
|
|
131
|
+
};
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/** Decide whether enabled bandit re-ranking has the on-disk state needed to operate. */
|
|
135
|
+
export function checkRankingState(input: RankingStateCheckInput): RankingStateCheckResult {
|
|
136
|
+
const binary = input.binPath ?? '(unresolved)';
|
|
137
|
+
if (!input.flagOn) {
|
|
138
|
+
return {
|
|
139
|
+
level: 'ok',
|
|
140
|
+
detail: `bandit re-ranking feature is off; no state is expected at ${input.statePath}; answering binary ${binary}`,
|
|
141
|
+
};
|
|
142
|
+
}
|
|
143
|
+
if (input.stateExists) {
|
|
144
|
+
return {
|
|
145
|
+
level: 'ok',
|
|
146
|
+
detail: `bandit re-ranking is on and state is present at ${input.statePath}; answering binary ${binary}`,
|
|
147
|
+
};
|
|
148
|
+
}
|
|
149
|
+
return {
|
|
150
|
+
level: 'warn',
|
|
151
|
+
detail: `bandit re-ranking is on but state is absent at ${input.statePath}; answering binary ${binary}`,
|
|
152
|
+
};
|
|
153
|
+
}
|
package/src/feature-tier.ts
CHANGED
|
@@ -1,7 +1,20 @@
|
|
|
1
1
|
import type { FeatureTier } from './guard-volume.js';
|
|
2
2
|
|
|
3
|
+
/**
|
|
4
|
+
* Начало строки, на котором тир ещё считается ОБЪЯВЛЕННЫМ, а не упомянутым в прозе: необязательный
|
|
5
|
+
* маркер списка (`-`, `*`, `+`, `1.`), необязательный заголовок и необязательное выделение.
|
|
6
|
+
*
|
|
7
|
+
* Маркер списка добавлен 2026-09-21 по измерению: `00_complexity_assessment.md` фичи
|
|
8
|
+
* `amendment-seams` несёт строку `- **Tier: S** — …`, и парсер её НЕ видел. Следствие было не
|
|
9
|
+
* косметическим: `dz contract-check --slug amendment-seams` отвечал NOT-ESTABLISHED с диагнозом
|
|
10
|
+
* «required ADR directory cannot be resolved», хотя в самом приборе есть верная ветка «тиру S
|
|
11
|
+
* каталог ADR не требуется» — она просто никогда не исполнялась, потому что тир читался как
|
|
12
|
+
* неизвестный. То есть вердикт называл не ту причину и отправлял чинить не то (бэклог 5d436aa6).
|
|
13
|
+
*/
|
|
14
|
+
const TIER_LINE_START = /^\s*(?:[-*+]\s+|\d+[.)]\s+)?(?:##+\s*)?(?:\*\*)?(?:Tier|Тир)(?=$|[\s:*])/iu;
|
|
15
|
+
|
|
3
16
|
function tiersAfterMarkers(line: string): FeatureTier[] {
|
|
4
|
-
if (
|
|
17
|
+
if (!TIER_LINE_START.test(line)) return [];
|
|
5
18
|
const tiers: FeatureTier[] = [];
|
|
6
19
|
for (const marker of line.matchAll(/Tier|Тир/giu)) {
|
|
7
20
|
const markerIndex = marker.index ?? 0;
|
package/src/guard.ts
CHANGED
|
@@ -1338,6 +1338,18 @@ export interface GuardAuditRecord {
|
|
|
1338
1338
|
readonly observations?: readonly GuardObservation[];
|
|
1339
1339
|
/** set when the operator overrode a block with `--force <reason>` — the override is logged, never silent. */
|
|
1340
1340
|
readonly override?: { readonly forced: true; readonly reason: string };
|
|
1341
|
+
/**
|
|
1342
|
+
* Ids of every rule this run EVALUATED — `checked` plus `notEstablished`, because both mean the
|
|
1343
|
+
* rule was active for the op and got its turn.
|
|
1344
|
+
*
|
|
1345
|
+
* Why the log needs it (backlog 1bee49dd): a guard rule's zero-firing state is its HEALTHY state,
|
|
1346
|
+
* so `violations[]` cannot tell a working safety net from a dead rule. MEASURED 2026-09-21 on
|
|
1347
|
+
* `.dz/guard-audit.jsonl`: 1854 rows, 26 default rules, 6 of which never appear in any violations
|
|
1348
|
+
* array — and 4 of those 6 are not allowlisted, so the moment the report's history floor is met
|
|
1349
|
+
* they would be named dead for doing their job. The evaluation set was already computed in
|
|
1350
|
+
* `GuardResult`; only the record dropped it.
|
|
1351
|
+
*/
|
|
1352
|
+
readonly evaluated?: readonly string[];
|
|
1341
1353
|
}
|
|
1342
1354
|
|
|
1343
1355
|
/** Build the audit record for a guard evaluation (+ an optional forced-override reason). Pure. */
|
|
@@ -1350,9 +1362,21 @@ export function auditRecord(result: GuardResult, ts: string, override?: { reason
|
|
|
1350
1362
|
...(Array.isArray(result.notes) && result.notes.length > 0 ? { notes: result.notes } : {}),
|
|
1351
1363
|
...(Array.isArray(result.observations) && result.observations.length > 0 ? { observations: result.observations } : {}),
|
|
1352
1364
|
...(override && typeof override.reason === 'string' ? { override: { forced: true, reason: override.reason } } : {}),
|
|
1365
|
+
...(evaluatedRuleIds(result).length > 0 ? { evaluated: evaluatedRuleIds(result) } : {}),
|
|
1353
1366
|
};
|
|
1354
1367
|
}
|
|
1355
1368
|
|
|
1369
|
+
/**
|
|
1370
|
+
* Every rule that got its turn this run, sorted and de-duplicated. A rule with no input still RAN —
|
|
1371
|
+
* calling that "not evaluated" would reintroduce the very conflation this field exists to remove.
|
|
1372
|
+
*/
|
|
1373
|
+
export function evaluatedRuleIds(result: GuardResult): string[] {
|
|
1374
|
+
const ids = new Set<string>();
|
|
1375
|
+
for (const id of Array.isArray(result.checked) ? result.checked : []) if (typeof id === 'string' && id !== '') ids.add(id);
|
|
1376
|
+
for (const id of Array.isArray(result.notEstablished) ? result.notEstablished : []) if (typeof id === 'string' && id !== '') ids.add(id);
|
|
1377
|
+
return [...ids].sort();
|
|
1378
|
+
}
|
|
1379
|
+
|
|
1356
1380
|
/** The exit-code contract: a block is non-zero unless forced; a warn/pass is zero. */
|
|
1357
1381
|
export function guardExitCode(result: GuardResult, forced: boolean): number {
|
|
1358
1382
|
return result.verdict === 'block' && !forced ? 1 : 0;
|
package/src/index.ts
CHANGED
|
@@ -78,6 +78,8 @@ export {
|
|
|
78
78
|
} from './parity.js';
|
|
79
79
|
export type { RuntimeCapability, FeatureForm, ParityFeature, ParityCell, ParityReportCell, ParityMatrixRow, CapabilityEvidence, UnbackedCapability, ProbedRuntimeVersions } from './parity.js';
|
|
80
80
|
export * from './operations.js';
|
|
81
|
+
export { checkInstrumentFreshness, checkRankingState } from './doctor-instrument.js';
|
|
82
|
+
export type { InstrumentFreshness, InstrumentCheckInput, InstrumentCheckResult, RankingStateCheckInput, RankingStateCheckResult } from './doctor-instrument.js';
|
|
81
83
|
// workflows.ts: the ADR-005 templates are RETIRED (feature loop-designer, AM-6) — the module is a
|
|
82
84
|
// deprecation shim (empty WORKFLOW_NAMES). BREAKING for external harness-core consumers of
|
|
83
85
|
// WorkflowTemplate/WORKFLOWS/getWorkflow — deliberately channeled through the 0.x MINOR bump and
|
|
@@ -137,7 +139,7 @@ export type { RepoBoundaryIo } from './repo-boundary.js';
|
|
|
137
139
|
export type { LedgerBackfillPlan, LedgerBackfillRow, RunCostFacts } from './ledger-backfill.js';
|
|
138
140
|
export type { SweepResult, DriftedSkill, SyncResult, SyncCanonicalOptions } from './skill-drift.js';
|
|
139
141
|
export { benchmarkSkill, benchmarkSkills, compareSkills } from './benchmark.js';
|
|
140
|
-
export { buildRegistry, searchRegistry, filterByCategory, skillPackBaseDirs, discoverSkillPackDirs, discoverSkillCarryingDirs, discoverVerifiablePackDirs } from './registry.js';
|
|
142
|
+
export { buildRegistry, buildShowcaseRegistry, searchRegistry, filterByCategory, skillPackBaseDirs, discoverSkillPackDirs, discoverSkillCarryingDirs, discoverVerifiablePackDirs, packScope, verifiedScopeNote } from './registry.js';
|
|
141
143
|
export { tokenize, stemToken, stems } from './stem.js';
|
|
142
144
|
// Package skill-layout resolution (feature dz-install-npx-init) — the ONE seam that knows where an
|
|
143
145
|
// npm package keeps its skills (flat / templates/.claude/skills / skills). `cmdInstall` calls it;
|
|
@@ -782,6 +784,7 @@ export {
|
|
|
782
784
|
planReleaseGates,
|
|
783
785
|
classifyGateExecutions,
|
|
784
786
|
buildFailureIssue,
|
|
787
|
+
shouldRetryGhWithoutToken,
|
|
785
788
|
buildReleaseNotes,
|
|
786
789
|
releaseTagName,
|
|
787
790
|
firstOutputLine,
|
|
@@ -881,7 +884,7 @@ export type {
|
|
|
881
884
|
McpSeverity,
|
|
882
885
|
McpCapability,
|
|
883
886
|
} from './mcp-scan.js';
|
|
884
|
-
export type { RegistryEntry, Registry } from './registry.js';
|
|
887
|
+
export type { RegistryEntry, Registry, ShowcaseRegistry, ShowcaseSkill } from './registry.js';
|
|
885
888
|
export type { BenchmarkCheck, BenchmarkScore, BenchmarkReport, CompareResult } from './benchmark.js';
|
|
886
889
|
export {
|
|
887
890
|
specToOpts,
|
package/src/mutation-gate.ts
CHANGED
|
@@ -494,6 +494,12 @@ export interface MutationObservation {
|
|
|
494
494
|
readonly outputError?: string;
|
|
495
495
|
/** bounded log proving an internal runner failure received at most one retry. */
|
|
496
496
|
readonly internalAttemptLog?: string;
|
|
497
|
+
/**
|
|
498
|
+
* Whether any suite the entry's `testCommand` selects NAMES the mutated module — see
|
|
499
|
+
* {@link suiteSelectionNamesModule}. Only read on the UNDEFENDED path, and only to add a hint:
|
|
500
|
+
* `false` says "check the command before the tests", never "the property is fine".
|
|
501
|
+
*/
|
|
502
|
+
readonly suiteNamesModule?: boolean | 'unknown';
|
|
497
503
|
}
|
|
498
504
|
|
|
499
505
|
export interface MutationEntryResult {
|
|
@@ -1333,7 +1339,13 @@ function classifyMutationOutcomeWithoutAttemptLog(obs: MutationObservation): Mut
|
|
|
1333
1339
|
applied: true,
|
|
1334
1340
|
verdict: 'UNDEFENDED',
|
|
1335
1341
|
drop: false,
|
|
1336
|
-
detail: `suite stayed GREEN with the protection deleted — property UNDEFENDED: "${e.property}" (${e.file}). The suite would not notice this protection regressing (rule 2)
|
|
1342
|
+
detail: `suite stayed GREEN with the protection deleted — property UNDEFENDED: "${e.property}" (${e.file}). The suite would not notice this protection regressing (rule 2).`
|
|
1343
|
+
+ (obs.suiteNamesModule === false
|
|
1344
|
+
? ' HINT: no suite in this entry\'s testCommand NAMES this module, so check the COMMAND before the tests'
|
|
1345
|
+
+ ' — a suite that never loads the module cannot notice its protection (MEASURED 2026-09-04: two properties'
|
|
1346
|
+
+ ' read as UNDEFENDED for exactly this reason, and both tests existed). The hint matches the module stem,'
|
|
1347
|
+
+ ' so a transitive import would not be seen by it: it narrows where to look, it does not decide.'
|
|
1348
|
+
: ''),
|
|
1337
1349
|
};
|
|
1338
1350
|
}
|
|
1339
1351
|
|
|
@@ -1408,6 +1420,47 @@ function classifyMutationOutcomeWithoutAttemptLog(obs: MutationObservation): Mut
|
|
|
1408
1420
|
};
|
|
1409
1421
|
}
|
|
1410
1422
|
|
|
1423
|
+
/**
|
|
1424
|
+
* Suite paths a registry `testCommand` selects, in order. Tokens that are not suite files (the
|
|
1425
|
+
* runner, its flags) are ignored — the command is a shell line, not a schema, so this reads the
|
|
1426
|
+
* shape it actually has rather than assuming one.
|
|
1427
|
+
*/
|
|
1428
|
+
export function parseSuitePaths(testCommand: string): readonly string[] {
|
|
1429
|
+
return String(testCommand ?? '')
|
|
1430
|
+
.split(/\s+/)
|
|
1431
|
+
.filter((token) => /\.(?:test|spec)\.[cm]?[jt]sx?$/.test(token) && !token.startsWith('-'));
|
|
1432
|
+
}
|
|
1433
|
+
|
|
1434
|
+
/**
|
|
1435
|
+
* Does ANY suite the command selects even name the mutated module?
|
|
1436
|
+
*
|
|
1437
|
+
* Why this exists: `UNDEFENDED` reads as "this property has no test", and MEASURED 2026-09-04 that
|
|
1438
|
+
* reading was wrong twice in one run — both tests existed; the registry's `testCommand` simply did
|
|
1439
|
+
* not select the suites that import them (backlog 1f4e4f66). The author filed a finding about two
|
|
1440
|
+
* "unprotected properties" before checking the instrument, which is the failure this hint prevents.
|
|
1441
|
+
*
|
|
1442
|
+
* Deliberately a HINT, never a verdict: it matches the module's STEM in each suite's text, so a
|
|
1443
|
+
* suite that reaches the module through a transitive import is invisible to it. `'unknown'` when no
|
|
1444
|
+
* suite could be read — absence of evidence is not evidence, and a hint that guesses is worse than
|
|
1445
|
+
* no hint.
|
|
1446
|
+
*/
|
|
1447
|
+
export function suiteSelectionNamesModule(input: {
|
|
1448
|
+
readonly file: string;
|
|
1449
|
+
readonly suitePaths: readonly string[];
|
|
1450
|
+
readonly readSuite: (path: string) => string | null;
|
|
1451
|
+
}): boolean | 'unknown' {
|
|
1452
|
+
const stem = String(input.file ?? '').split('/').pop()?.replace(/\.[cm]?[jt]sx?$/, '') ?? '';
|
|
1453
|
+
if (stem === '') return 'unknown';
|
|
1454
|
+
let read = 0;
|
|
1455
|
+
for (const suite of input.suitePaths) {
|
|
1456
|
+
const text = input.readSuite(suite);
|
|
1457
|
+
if (text === null) continue;
|
|
1458
|
+
read += 1;
|
|
1459
|
+
if (text.includes(stem)) return true;
|
|
1460
|
+
}
|
|
1461
|
+
return read === 0 ? 'unknown' : false;
|
|
1462
|
+
}
|
|
1463
|
+
|
|
1411
1464
|
export function classifyMutationOutcome(obs: MutationObservation): MutationEntryResult {
|
|
1412
1465
|
const result = classifyMutationOutcomeWithoutAttemptLog(obs);
|
|
1413
1466
|
if (obs.internalAttemptLog === undefined) return result;
|