@dzhechkov/harness-core 0.8.39 → 0.8.40

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. package/.dz-manifest.json +76 -56
  2. package/README.md +48 -0
  3. package/dist/cmd-usage.d.ts.map +1 -1
  4. package/dist/cmd-usage.js +48 -2
  5. package/dist/cmd-usage.js.map +1 -1
  6. package/dist/compounding.d.ts +66 -0
  7. package/dist/compounding.d.ts.map +1 -1
  8. package/dist/compounding.js +76 -9
  9. package/dist/compounding.js.map +1 -1
  10. package/dist/doctor-instrument.d.ts +65 -0
  11. package/dist/doctor-instrument.d.ts.map +1 -0
  12. package/dist/doctor-instrument.js +91 -0
  13. package/dist/doctor-instrument.js.map +1 -0
  14. package/dist/feature-tier.d.ts.map +1 -1
  15. package/dist/feature-tier.js +13 -1
  16. package/dist/feature-tier.js.map +1 -1
  17. package/dist/guard.d.ts +17 -0
  18. package/dist/guard.d.ts.map +1 -1
  19. package/dist/guard.js +15 -0
  20. package/dist/guard.js.map +1 -1
  21. package/dist/index.d.ts +5 -3
  22. package/dist/index.d.ts.map +1 -1
  23. package/dist/index.js +3 -2
  24. package/dist/index.js.map +1 -1
  25. package/dist/mutation-gate.d.ts +30 -0
  26. package/dist/mutation-gate.d.ts.map +1 -1
  27. package/dist/mutation-gate.js +45 -1
  28. package/dist/mutation-gate.js.map +1 -1
  29. package/dist/operations.d.ts +13 -0
  30. package/dist/operations.d.ts.map +1 -1
  31. package/dist/operations.js +88 -2
  32. package/dist/operations.js.map +1 -1
  33. package/dist/registry.d.ts +58 -0
  34. package/dist/registry.d.ts.map +1 -1
  35. package/dist/registry.js +62 -5
  36. package/dist/registry.js.map +1 -1
  37. package/dist/release.d.ts +18 -0
  38. package/dist/release.d.ts.map +1 -1
  39. package/dist/release.js +30 -0
  40. package/dist/release.js.map +1 -1
  41. package/dist/round-exec.d.ts +10 -0
  42. package/dist/round-exec.d.ts.map +1 -1
  43. package/dist/round-exec.js +3 -2
  44. package/dist/round-exec.js.map +1 -1
  45. package/dist/round.d.ts +19 -0
  46. package/dist/round.d.ts.map +1 -1
  47. package/dist/round.js +1 -0
  48. package/dist/round.js.map +1 -1
  49. package/package.json +1 -1
  50. package/sbom.json +105 -55
  51. package/src/cmd-usage.ts +52 -2
  52. package/src/compounding.ts +112 -9
  53. package/src/doctor-instrument.ts +153 -0
  54. package/src/feature-tier.ts +14 -1
  55. package/src/guard.ts +24 -0
  56. package/src/index.ts +5 -2
  57. package/src/mutation-gate.ts +54 -1
  58. package/src/operations.ts +85 -3
  59. package/src/registry.ts +91 -1
  60. package/src/release.ts +36 -0
  61. package/src/round-exec.ts +13 -2
  62. package/src/round.ts +20 -0
package/src/cmd-usage.ts CHANGED
@@ -153,6 +153,12 @@ interface RuleUsage {
153
153
  readonly stats: ReadonlyMap<string, CmdUsageStat>;
154
154
  readonly skipped: number;
155
155
  readonly outOfRange: number;
156
+ /**
157
+ * Audit rows inside the window that recorded which rules they EVALUATED. Zero means this report
158
+ * has no evidence about rules at all — which is a different answer from "the rule is unused", and
159
+ * the distinction is the whole point of counting it (backlog 1bee49dd).
160
+ */
161
+ readonly evaluationRows: number;
156
162
  }
157
163
 
158
164
  const REPO_BOUNDARY_IO = {
@@ -401,9 +407,13 @@ export function loadDeadwoodAllowlist(json: string): DeadwoodAllowlistEntry[] {
401
407
  function parseGuardAuditUsage(text: string, weeks: number, now: Date): RuleUsage {
402
408
  const auditTimestamps: string[] = [];
403
409
  const hits: CmdUsageInvocationRecord[] = [];
410
+ let evaluationRows = 0;
404
411
  let skipped = 0;
405
412
  let outOfRange = 0;
406
413
  const newestAllowed = now.getTime() + DEADWOOD_FUTURE_TOLERANCE_MS;
414
+ // Same window arithmetic `foldCmdUsage` uses, so "counted as evidence" and "counted as a run"
415
+ // cannot disagree about which rows are inside.
416
+ const windowStart = now.getTime() - Math.max(0, weeks) * 7 * DAY_MS;
407
417
  for (const line of text.split('\n')) {
408
418
  if (line.trim() === '') continue;
409
419
  let row: Record<string, unknown>;
@@ -425,17 +435,40 @@ function parseGuardAuditUsage(text: string, weeks: number, now: Date): RuleUsage
425
435
  continue;
426
436
  }
427
437
  auditTimestamps.push(row.ts);
438
+ // A rule's HEALTHY state is silence, so firing cannot measure whether it is alive. `evaluated`
439
+ // records the rules that actually got their turn; rows written before the field existed carry
440
+ // none, and they contribute evidence about firing only.
441
+ //
442
+ // Two conditions, both named by cross-family review (Codex gpt-5.6-sol, 2026-09-21), both of
443
+ // which turn this evidence into a false accusation if skipped:
444
+ // · the row must be INSIDE the window. An instrumented row older than `now - weeks` yields no
445
+ // in-window runs, so counting it as evidence would let one ancient row flip every absent
446
+ // rule from "cannot judge" to "dead".
447
+ // · the array must carry a USABLE id. `evaluated: []` is a row that recorded nothing; treating
448
+ // it as evidence is the same false accusation by a shorter path.
449
+ const evaluatedIds = new Set<string>();
450
+ if (Array.isArray(row.evaluated)) {
451
+ for (const value of row.evaluated) {
452
+ if (typeof value === 'string' && value.trim() !== '') evaluatedIds.add(value);
453
+ }
454
+ }
455
+ if (evaluatedIds.size > 0 && tsMs >= windowStart) evaluationRows += 1;
456
+ for (const id of evaluatedIds) {
457
+ hits.push({ kind: 'cmd', cmd: id, ts: row.ts, v: CMD_USAGE_SCHEMA });
458
+ }
428
459
  const violations = Array.isArray(row.violations) ? row.violations : [];
429
460
  for (const value of violations) {
430
461
  const rule = typeof value === 'object' && value !== null
431
462
  ? (value as { rule?: unknown }).rule
432
463
  : undefined;
433
- if (typeof rule === 'string' && rule.trim() !== '') {
464
+ // One guard run is ONE run. A rule that both evaluated and fired in the same row would be
465
+ // counted twice — 100 warning evaluations reported as 200 runs (same review, second finding).
466
+ if (typeof rule === 'string' && rule.trim() !== '' && !evaluatedIds.has(rule)) {
434
467
  hits.push({ kind: 'cmd', cmd: rule, ts: row.ts, v: CMD_USAGE_SCHEMA });
435
468
  }
436
469
  }
437
470
  }
438
- return { auditTimestamps, stats: foldCmdUsage(hits, weeks, now), skipped, outOfRange };
471
+ return { auditTimestamps, stats: foldCmdUsage(hits, weeks, now), skipped, outOfRange, evaluationRows };
439
472
  }
440
473
 
441
474
  function timestampDepthDays(timestamps: readonly string[], now: Date): number {
@@ -607,6 +640,23 @@ export function buildDeadwoodReport(input: DeadwoodInput): DeadwoodReport {
607
640
  });
608
641
  continue;
609
642
  }
643
+ // A rule the window has no EVALUATION evidence for is unjudged, not unused. Firing is the wrong
644
+ // signal for a guard (silence is its healthy state), so without `evaluated` rows the only honest
645
+ // answer is "this report cannot judge the rule" — exactly what the skill surface already says.
646
+ if (item.kind === 'rule' && rules.evaluationRows === 0
647
+ && (rules.stats.get(item.surface)?.runsInWindow ?? 0) === 0
648
+ // An explicit allowlist entry is an operator's standing statement about this surface; it keeps
649
+ // its own wording. Only the ACCUSING path — "zero usage, consider deprecating" — is withdrawn.
650
+ && !allowlist.has(allowlistKey(item.kind, item.surface))) {
651
+ noInstrumentation.push({
652
+ state: 'no-instrumentation',
653
+ surface: item.surface,
654
+ kind: item.kind,
655
+ reason: 'no guard-audit row in this window recorded which rules it evaluated; a rule that '
656
+ + 'never fires may be a healthy safety net, so firing alone cannot judge it',
657
+ });
658
+ continue;
659
+ }
610
660
  classifyInstrumented(
611
661
  { surface: item.surface, kind: item.kind },
612
662
  item.kind === 'command'
@@ -15,7 +15,7 @@
15
15
  * Everything here is PURE: callers gather facts (files, store rows); this module only computes.
16
16
  */
17
17
 
18
- import { EVENT_CHAIN_SCOPE, verifyEventChainText } from './event-chain.js';
18
+ import { EVENT_CHAIN_SCOPE, classifyChainDefects, verifyEventChainText } from './event-chain.js';
19
19
  import {
20
20
  isOffsetIsoTimestamp,
21
21
  type PromotionAcceptanceEvidence,
@@ -160,15 +160,36 @@ export interface GuardEvent {
160
160
  readonly verdict: string;
161
161
  readonly rules: readonly string[]; // violated rule ids
162
162
  readonly violations?: readonly { readonly rule: string; readonly contentAnchor?: string }[];
163
+ /**
164
+ * 1-based position of this record among the log's non-empty lines — the ONLY thing that can place
165
+ * it relative to a chain defect. Absent when the caller read the rows without a chain.
166
+ */
167
+ readonly chainLine?: number;
163
168
  }
164
169
 
165
170
  export type FunnelEvidenceSource<T> =
166
171
  | { readonly status: 'measured'; readonly rows: readonly T[] }
167
172
  | { readonly status: 'not-measured'; readonly reason: string };
168
173
 
174
+ /**
175
+ * Where the guard journal's chain damage sits, so a PERIOD can be judged instead of the whole FILE.
176
+ *
177
+ * A log damaged once in March and unbroken since is not evidence against September's rows, and
178
+ * refusing to measure September because of March is the same "verdict answers a different question"
179
+ * defect the chain headline was fixed for (backlog b38dd3ba, MEASURED 2026-09-21: 28 defects, all
180
+ * before a run of 1169 unbroken records, suppressed BOTH measured months).
181
+ */
182
+ export interface GuardAuditChainWindow {
183
+ /** First non-empty line of the current unbroken run: one past the last defect. */
184
+ readonly runFrom: number;
185
+ /** Total defects in the file. Zero means the window imposes nothing. */
186
+ readonly defects: number;
187
+ }
188
+
169
189
  export interface LessonToRuleFunnelFacts {
170
190
  readonly promotionRuns: FunnelEvidenceSource<PromotionRunEvidence>;
171
191
  readonly guardAudits: FunnelEvidenceSource<GuardEvent>;
192
+ readonly guardAuditChain?: GuardAuditChainWindow;
172
193
  readonly promotionAcceptances?: readonly PromotionAcceptanceEvidence[];
173
194
  readonly truncatedPromotionPeriods?: readonly string[];
174
195
  readonly acceptanceHistoryComplete?: boolean;
@@ -274,6 +295,20 @@ export interface EvidenceChainHealth {
274
295
  readonly preChainPrefix: number;
275
296
  readonly defects: number;
276
297
  readonly defectKinds: readonly string[];
298
+ /**
299
+ * WHERE the defects sit relative to the log's current unbroken run, and HOW MUCH of a run that is.
300
+ * Without this a bare `FAILED` over a log whose damage is entirely historical reads as "today's
301
+ * numbers are garbage", while `dz chain` over the SAME file says "healed … verdicts over those are
302
+ * sound" — MEASURED 2026-09-20 on `.dz/guard-audit.jsonl`: 28 defects, all before the current run,
303
+ * 1095 unbroken records after them; one instrument printed FAILED, the other healed, both exit 0
304
+ * (backlog 79ce6262). Neither was lying; neither named its WINDOW. `event-chain.ts` says it
305
+ * outright: a caller that reports soundness without printing the run size overclaims on its behalf,
306
+ * and the same holds for a caller that reports damage without printing where the damage sits.
307
+ */
308
+ readonly defectsBeforeRun: number;
309
+ readonly defectsInRun: number;
310
+ /** Records in the current unbroken run — the evidence behind any "sound for today" reading. */
311
+ readonly runRecords: number;
277
312
  }
278
313
 
279
314
  export interface InstrumentationHealth {
@@ -394,6 +429,13 @@ function executionMeasurement(
394
429
  if (periodAudits.length === 0) {
395
430
  return LESSON_TO_RULE_FUNNEL_POLICY.unavailable(`guard-audit-not-recorded:${period}`);
396
431
  }
432
+ // Damage that PRECEDES this period's rows says nothing about them; damage that touches them does.
433
+ // A row with no position cannot be placed, and unplaceable is not the same as sound — it refuses.
434
+ const chain = facts.guardAuditChain;
435
+ if (chain !== undefined && chain.defects > 0
436
+ && periodAudits.some((row) => row.chainLine === undefined || row.chainLine < chain.runFrom)) {
437
+ return LESSON_TO_RULE_FUNNEL_POLICY.unavailable(`guard-audit-chain-damaged:${period}`);
438
+ }
397
439
  const audits = periodAudits.filter((row) => row.op === 'publish');
398
440
  if (audits.length === 0) {
399
441
  return LESSON_TO_RULE_FUNNEL_POLICY.unavailable(`guard-publish-not-recorded:${period}`);
@@ -594,6 +636,11 @@ export function assembleCompoundingReport(facts: CompoundingFacts): CompoundingR
594
636
  preChainPrefix: v.preChainPrefix,
595
637
  defects: v.defects.length,
596
638
  defectKinds: [...new Set(v.defects.map((d) => d.kind))],
639
+ ...((age) => ({
640
+ defectsBeforeRun: age.beforeRun.length,
641
+ defectsInRun: age.inRun.length,
642
+ runRecords: age.runRecords,
643
+ }))(classifyChainDefects(v, v.lines)),
597
644
  };
598
645
  });
599
646
 
@@ -623,13 +670,7 @@ export function assembleCompoundingReport(facts: CompoundingFacts): CompoundingR
623
670
  trajectory.length > 0 ? `guard: ${improvedRules}/${trajectory.length} rules recur less in the later half` : 'guard: not enough history',
624
671
  `cold-vs-warm: ${replay.verdict === 'insufficient-data' ? 'INSUFFICIENT DATA (accruing)' : 'READY to measure'}`,
625
672
  instrumentation.applyLegLive ? 'apply leg: live' : 'apply leg: STALE — fix the instrumentation before trusting anything above',
626
- ...(chains.length === 0
627
- ? []
628
- : [
629
- instrumentation.chainsOk
630
- ? 'evidence chain: verified'
631
- : 'evidence chain: CORRUPT — the numbers above are computed from a damaged log',
632
- ]),
673
+ ...(chains.length === 0 ? [] : [chainHeadline(chains)]),
633
674
  ].join(' · ');
634
675
 
635
676
  return { pool, guardTrajectory: trajectory, replay, instrumentation, lessonToRuleFunnel, verdict };
@@ -671,6 +712,68 @@ function renderFunnelPeriodMeasurements(row: LessonToRuleFunnelPeriod): string {
671
712
  return `${renderPromotionMeasurements(row)} · ${renderFunnelMeasurement('executions', row.executions)}`;
672
713
  }
673
714
 
715
+ /**
716
+ * The HEADLINE verdict over every evidence log — three-valued, because two values lied.
717
+ *
718
+ * MEASURED 2026-09-21 on `.dz/guard-audit.jsonl`: 28 defects, the LAST of them dated 2026-09-05,
719
+ * followed by more than a thousand unbroken records. The old headline read
720
+ * "CORRUPT — the numbers above are computed from a damaged log", which is true of the FILE'S
721
+ * HISTORY and false about the numbers it was printed next to. The distinction already existed one
722
+ * function below, in {@link chainVerdictPhrase}; it simply never reached the line a reader sees
723
+ * first. That is the same defect class this report exists to find: a verdict answering a different
724
+ * question than the one it appears to answer.
725
+ */
726
+ /**
727
+ * Whether the report's OWN numbers may be trusted, as a value the caller can turn into an exit code.
728
+ *
729
+ * Backlog 79ce6262 named the defect: the report printed "the numbers above are computed from a
730
+ * damaged log" and exited 0 anyway — a tool announcing its own output untrustworthy and reporting
731
+ * success. That record offered two lawful cures and asked which applies. Both do, on different
732
+ * branches, and only the three-valued verdict lets them coexist: damage BEHIND the current run
733
+ * narrows the WORDING (the numbers stand, exit 0), damage INSIDE it makes the numbers genuinely
734
+ * unreliable and must reach the exit code.
735
+ *
736
+ * `'trusted'` ⇒ 0. `'unreliable'` ⇒ a non-zero the caller chooses — the run succeeded, the verdict
737
+ * cannot be relied on, which is this repository's INCONCLUSIVE shape, not its failure shape.
738
+ */
739
+ export function chainTrust(chains: readonly EvidenceChainHealth[]): 'trusted' | 'unreliable' {
740
+ return chains.some((c) => !c.ok && c.defectsInRun > 0) ? 'unreliable' : 'trusted';
741
+ }
742
+
743
+ export function chainHeadline(chains: readonly EvidenceChainHealth[]): string {
744
+ const broken = chains.filter((c) => !c.ok);
745
+ if (broken.length === 0) return 'evidence chain: verified';
746
+ const live = broken.filter((c) => c.defectsInRun > 0);
747
+ if (live.length === 0) {
748
+ const runs = broken.reduce((n, c) => n + c.runRecords, 0);
749
+ const defects = broken.reduce((n, c) => n + c.defects, 0);
750
+ return `evidence chain: damaged EARLIER — ${defects} defect(s), none inside the current run of `
751
+ + `${runs} unbroken record(s); the numbers above stand, the file's history does not`;
752
+ }
753
+ const inRun = live.reduce((n, c) => n + c.defectsInRun, 0);
754
+ return `evidence chain: CORRUPT — ${inRun} defect(s) INSIDE the current run; the numbers above are `
755
+ + 'computed from a damaged log';
756
+ }
757
+
758
+ /**
759
+ * The verdict phrase for one evidence log, with its WINDOW named. A bare `FAILED` over damage that an
760
+ * unbroken run has already followed is true of the FILE and misleading about TODAY — see
761
+ * {@link EvidenceChainHealth.defectsBeforeRun}.
762
+ */
763
+ export function chainVerdictPhrase(c: EvidenceChainHealth): string {
764
+ if (c.ok) return 'verified';
765
+ const kinds = `[${c.defectKinds.join(', ')}]`;
766
+ if (c.defectsInRun === 0) {
767
+ return `DAMAGED EARLIER — ${c.defects} defect(s) ${kinds}, all BEFORE the current run of `
768
+ + `${c.runRecords} unbroken record(s); numbers over that run stand, the file's history does not`;
769
+ }
770
+ if (c.defectsBeforeRun === 0) {
771
+ return `FAILED — ${c.defects} defect(s) ${kinds} with NO sound records after them`;
772
+ }
773
+ return `FAILED — ${c.defects} defect(s) ${kinds}: ${c.defectsInRun} inside the current run of `
774
+ + `${c.runRecords} record(s), ${c.defectsBeforeRun} before it`;
775
+ }
776
+
674
777
  export function renderCompoundingReport(r: CompoundingReport): string {
675
778
  const out: string[] = [];
676
779
  out.push('dz compounding — does the learning loop pay? (honest report: gates without data say so)');
@@ -695,7 +798,7 @@ export function renderCompoundingReport(r: CompoundingReport): string {
695
798
  );
696
799
  for (const c of r.instrumentation.chains) {
697
800
  out.push(
698
- ` EVIDENCE CHAIN ${c.log}: ${c.ok ? 'verified' : `FAILED — ${c.defects} defect(s) [${c.defectKinds.join(', ')}]`}` +
801
+ ` EVIDENCE CHAIN ${c.log}: ${chainVerdictPhrase(c)}` +
699
802
  ` · ${c.chained} chained · ${c.preChainPrefix} pre-chain (uncovered)`,
700
803
  );
701
804
  }
@@ -0,0 +1,153 @@
1
+ import { isAbsolute, relative, sep } from 'node:path';
2
+
3
+ export type InstrumentFreshness = 'same' | 'stale' | 'unknown';
4
+
5
+ export interface InstrumentCheckInput {
6
+ /** realpath of the running binary, or null when it could not be resolved. */
7
+ readonly binPath: string | null;
8
+ /** version from the package.json that owns binPath, or null. */
9
+ readonly binVersion: string | null;
10
+ /** version from packages/@dzhechkov/harness-cli/package.json, or null outside the monorepo. */
11
+ readonly treeVersion: string | null;
12
+ /** absolute, realpath'd project root. */
13
+ readonly projectRoot: string;
14
+ /**
15
+ * Whether `projectRoot` above really IS realpath'd. The caller resolves it and falls back to a
16
+ * plain resolve when that throws; with a symlinked root that fallback compares a realpath'd
17
+ * binary against a non-realpath'd root, and an IN-TREE binary then looks external. Containment is
18
+ * undecidable in that state, so it is answered `unknown` rather than guessed either way.
19
+ * Named by independent review (Claude Sonnet, 2026-09-20).
20
+ */
21
+ readonly projectRootRealpathed: boolean;
22
+ readonly isMonorepo: boolean;
23
+ }
24
+
25
+ /**
26
+ * `level` is the WHOLE verdict — the caller renders it and never re-derives one of its own. That is
27
+ * deliberate: the first wiring of this module answered `unknown` with `ok: false` while this decider
28
+ * answered `ok`, and two answers to one question is the defect class this repo pays for most often.
29
+ *
30
+ * Three values, because two would lie: `ok` (the instrument is the tree's, or the check does not
31
+ * apply here), `warn` (measured stale — worth saying loudly, never worth failing a health command
32
+ * that gates other people's CI), `unknown` (the evidence could not be gathered — never rendered as
33
+ * a pass, and never as a failure either, since absence of evidence is not a defect).
34
+ */
35
+ export interface InstrumentCheckResult {
36
+ readonly freshness: InstrumentFreshness;
37
+ readonly level: 'ok' | 'warn' | 'unknown';
38
+ readonly detail: string;
39
+ }
40
+
41
+ export interface RankingStateCheckInput {
42
+ readonly flagOn: boolean;
43
+ /** absolute path the state was looked for at. */
44
+ readonly statePath: string;
45
+ readonly stateExists: boolean;
46
+ /** resolved binary path, for the detail — null when unknown. */
47
+ readonly binPath: string | null;
48
+ }
49
+
50
+ /** Ranking state has no freshness concept, so it deliberately has its own result type. */
51
+ export interface RankingStateCheckResult {
52
+ readonly level: 'ok' | 'warn';
53
+ readonly detail: string;
54
+ }
55
+
56
+ /** Strip semver build metadata: `0.8.32+a1b2c3` and `0.8.32` are the same release. */
57
+ function withoutBuildMetadata(version: string): string {
58
+ const plus = version.indexOf('+');
59
+ return plus < 0 ? version : version.slice(0, plus);
60
+ }
61
+
62
+ function isInsideRoot(candidate: string, root: string): boolean {
63
+ const fromRoot = relative(root, candidate);
64
+ return fromRoot === '' || (!isAbsolute(fromRoot) && fromRoot !== '..' && !fromRoot.startsWith(`..${sep}`));
65
+ }
66
+
67
+ /**
68
+ * Decide whether the executable answering `dz doctor` is the workspace's current instrument.
69
+ *
70
+ * LIMITS NAMED BY INDEPENDENT REVIEW (Claude Sonnet, 2026-09-20), none of them hidden behind a
71
+ * passing test:
72
+ * - Containment is a case-SENSITIVE path comparison. On a case-insensitive filesystem, or where the
73
+ * same location is reachable under two path forms, an in-tree binary can read as external. This
74
+ * repo runs on Linux; the cost of being wrong is one extra `warn` line and never an exit code.
75
+ * - The caller attributes a version by walking up from the binary to the NEAREST `package.json`.
76
+ * A shim in package A that loads package B's code is attributed to A, and a broken install with
77
+ * no own manifest is attributed to whatever ancestor has one. The detail always prints the
78
+ * resolved binary path so a reader can see which file was actually measured.
79
+ */
80
+ export function checkInstrumentFreshness(input: InstrumentCheckInput): InstrumentCheckResult {
81
+ const binary = input.binPath ?? '(unresolved)';
82
+ const binaryVersion = input.binVersion ?? 'unknown';
83
+ const treeVersion = input.treeVersion ?? 'unknown';
84
+
85
+ if (!input.isMonorepo) {
86
+ return {
87
+ freshness: 'unknown',
88
+ level: 'ok',
89
+ detail: `not applicable in a consumer project: no packages/@dzhechkov tree version to compare; resolved binary ${binary}; binary version ${binaryVersion}; tree version ${treeVersion}`,
90
+ };
91
+ }
92
+
93
+ if (input.binPath !== null && input.projectRootRealpathed && isInsideRoot(input.binPath, input.projectRoot)) {
94
+ return {
95
+ freshness: 'same',
96
+ level: 'ok',
97
+ detail: `resolved binary ${input.binPath} is inside project root ${input.projectRoot}; binary version ${binaryVersion}; tree version ${treeVersion}; this binary is the tree instrument`,
98
+ };
99
+ }
100
+
101
+ if (!input.projectRootRealpathed) {
102
+ return {
103
+ freshness: 'unknown',
104
+ level: 'unknown',
105
+ detail: `project root ${input.projectRoot} could not be resolved through its symlinks, so it cannot be told whether the answering binary is the tree's own; version could not be determined safely; resolved binary ${binary}; binary version ${binaryVersion}; tree version ${treeVersion}`,
106
+ };
107
+ }
108
+
109
+ if (input.binPath === null || input.binVersion === null || input.treeVersion === null) {
110
+ return {
111
+ freshness: 'unknown',
112
+ level: 'unknown',
113
+ detail: `instrument version could not be determined; resolved binary ${binary}; binary version ${binaryVersion}; tree version ${treeVersion}`,
114
+ };
115
+ }
116
+
117
+ // Semver says build metadata after `+` does not participate in equality, so a pipeline that
118
+ // stamps a commit hash onto the version must not read as a stale instrument.
119
+ if (withoutBuildMetadata(input.binVersion) === withoutBuildMetadata(input.treeVersion)) {
120
+ return {
121
+ freshness: 'same',
122
+ level: 'ok',
123
+ detail: `resolved binary ${input.binPath}; binary version ${input.binVersion}; tree version ${input.treeVersion}; versions match`,
124
+ };
125
+ }
126
+
127
+ return {
128
+ freshness: 'stale',
129
+ level: 'warn',
130
+ detail: `resolved binary ${input.binPath}; binary version ${input.binVersion}; tree version ${input.treeVersion}; the answering instrument is stale`,
131
+ };
132
+ }
133
+
134
+ /** Decide whether enabled bandit re-ranking has the on-disk state needed to operate. */
135
+ export function checkRankingState(input: RankingStateCheckInput): RankingStateCheckResult {
136
+ const binary = input.binPath ?? '(unresolved)';
137
+ if (!input.flagOn) {
138
+ return {
139
+ level: 'ok',
140
+ detail: `bandit re-ranking feature is off; no state is expected at ${input.statePath}; answering binary ${binary}`,
141
+ };
142
+ }
143
+ if (input.stateExists) {
144
+ return {
145
+ level: 'ok',
146
+ detail: `bandit re-ranking is on and state is present at ${input.statePath}; answering binary ${binary}`,
147
+ };
148
+ }
149
+ return {
150
+ level: 'warn',
151
+ detail: `bandit re-ranking is on but state is absent at ${input.statePath}; answering binary ${binary}`,
152
+ };
153
+ }
@@ -1,7 +1,20 @@
1
1
  import type { FeatureTier } from './guard-volume.js';
2
2
 
3
+ /**
4
+ * Начало строки, на котором тир ещё считается ОБЪЯВЛЕННЫМ, а не упомянутым в прозе: необязательный
5
+ * маркер списка (`-`, `*`, `+`, `1.`), необязательный заголовок и необязательное выделение.
6
+ *
7
+ * Маркер списка добавлен 2026-09-21 по измерению: `00_complexity_assessment.md` фичи
8
+ * `amendment-seams` несёт строку `- **Tier: S** — …`, и парсер её НЕ видел. Следствие было не
9
+ * косметическим: `dz contract-check --slug amendment-seams` отвечал NOT-ESTABLISHED с диагнозом
10
+ * «required ADR directory cannot be resolved», хотя в самом приборе есть верная ветка «тиру S
11
+ * каталог ADR не требуется» — она просто никогда не исполнялась, потому что тир читался как
12
+ * неизвестный. То есть вердикт называл не ту причину и отправлял чинить не то (бэклог 5d436aa6).
13
+ */
14
+ const TIER_LINE_START = /^\s*(?:[-*+]\s+|\d+[.)]\s+)?(?:##+\s*)?(?:\*\*)?(?:Tier|Тир)(?=$|[\s:*])/iu;
15
+
3
16
  function tiersAfterMarkers(line: string): FeatureTier[] {
4
- if (!/^\s*(?:##\s*)?(?:\*\*)?(?:Tier|Тир)(?=$|[\s:*])/iu.test(line)) return [];
17
+ if (!TIER_LINE_START.test(line)) return [];
5
18
  const tiers: FeatureTier[] = [];
6
19
  for (const marker of line.matchAll(/Tier|Тир/giu)) {
7
20
  const markerIndex = marker.index ?? 0;
package/src/guard.ts CHANGED
@@ -1338,6 +1338,18 @@ export interface GuardAuditRecord {
1338
1338
  readonly observations?: readonly GuardObservation[];
1339
1339
  /** set when the operator overrode a block with `--force <reason>` — the override is logged, never silent. */
1340
1340
  readonly override?: { readonly forced: true; readonly reason: string };
1341
+ /**
1342
+ * Ids of every rule this run EVALUATED — `checked` plus `notEstablished`, because both mean the
1343
+ * rule was active for the op and got its turn.
1344
+ *
1345
+ * Why the log needs it (backlog 1bee49dd): a guard rule's zero-firing state is its HEALTHY state,
1346
+ * so `violations[]` cannot tell a working safety net from a dead rule. MEASURED 2026-09-21 on
1347
+ * `.dz/guard-audit.jsonl`: 1854 rows, 26 default rules, 6 of which never appear in any violations
1348
+ * array — and 4 of those 6 are not allowlisted, so the moment the report's history floor is met
1349
+ * they would be named dead for doing their job. The evaluation set was already computed in
1350
+ * `GuardResult`; only the record dropped it.
1351
+ */
1352
+ readonly evaluated?: readonly string[];
1341
1353
  }
1342
1354
 
1343
1355
  /** Build the audit record for a guard evaluation (+ an optional forced-override reason). Pure. */
@@ -1350,9 +1362,21 @@ export function auditRecord(result: GuardResult, ts: string, override?: { reason
1350
1362
  ...(Array.isArray(result.notes) && result.notes.length > 0 ? { notes: result.notes } : {}),
1351
1363
  ...(Array.isArray(result.observations) && result.observations.length > 0 ? { observations: result.observations } : {}),
1352
1364
  ...(override && typeof override.reason === 'string' ? { override: { forced: true, reason: override.reason } } : {}),
1365
+ ...(evaluatedRuleIds(result).length > 0 ? { evaluated: evaluatedRuleIds(result) } : {}),
1353
1366
  };
1354
1367
  }
1355
1368
 
1369
+ /**
1370
+ * Every rule that got its turn this run, sorted and de-duplicated. A rule with no input still RAN —
1371
+ * calling that "not evaluated" would reintroduce the very conflation this field exists to remove.
1372
+ */
1373
+ export function evaluatedRuleIds(result: GuardResult): string[] {
1374
+ const ids = new Set<string>();
1375
+ for (const id of Array.isArray(result.checked) ? result.checked : []) if (typeof id === 'string' && id !== '') ids.add(id);
1376
+ for (const id of Array.isArray(result.notEstablished) ? result.notEstablished : []) if (typeof id === 'string' && id !== '') ids.add(id);
1377
+ return [...ids].sort();
1378
+ }
1379
+
1356
1380
  /** The exit-code contract: a block is non-zero unless forced; a warn/pass is zero. */
1357
1381
  export function guardExitCode(result: GuardResult, forced: boolean): number {
1358
1382
  return result.verdict === 'block' && !forced ? 1 : 0;
package/src/index.ts CHANGED
@@ -78,6 +78,8 @@ export {
78
78
  } from './parity.js';
79
79
  export type { RuntimeCapability, FeatureForm, ParityFeature, ParityCell, ParityReportCell, ParityMatrixRow, CapabilityEvidence, UnbackedCapability, ProbedRuntimeVersions } from './parity.js';
80
80
  export * from './operations.js';
81
+ export { checkInstrumentFreshness, checkRankingState } from './doctor-instrument.js';
82
+ export type { InstrumentFreshness, InstrumentCheckInput, InstrumentCheckResult, RankingStateCheckInput, RankingStateCheckResult } from './doctor-instrument.js';
81
83
  // workflows.ts: the ADR-005 templates are RETIRED (feature loop-designer, AM-6) — the module is a
82
84
  // deprecation shim (empty WORKFLOW_NAMES). BREAKING for external harness-core consumers of
83
85
  // WorkflowTemplate/WORKFLOWS/getWorkflow — deliberately channeled through the 0.x MINOR bump and
@@ -137,7 +139,7 @@ export type { RepoBoundaryIo } from './repo-boundary.js';
137
139
  export type { LedgerBackfillPlan, LedgerBackfillRow, RunCostFacts } from './ledger-backfill.js';
138
140
  export type { SweepResult, DriftedSkill, SyncResult, SyncCanonicalOptions } from './skill-drift.js';
139
141
  export { benchmarkSkill, benchmarkSkills, compareSkills } from './benchmark.js';
140
- export { buildRegistry, searchRegistry, filterByCategory, skillPackBaseDirs, discoverSkillPackDirs, discoverSkillCarryingDirs, discoverVerifiablePackDirs } from './registry.js';
142
+ export { buildRegistry, buildShowcaseRegistry, searchRegistry, filterByCategory, skillPackBaseDirs, discoverSkillPackDirs, discoverSkillCarryingDirs, discoverVerifiablePackDirs, packScope, verifiedScopeNote } from './registry.js';
141
143
  export { tokenize, stemToken, stems } from './stem.js';
142
144
  // Package skill-layout resolution (feature dz-install-npx-init) — the ONE seam that knows where an
143
145
  // npm package keeps its skills (flat / templates/.claude/skills / skills). `cmdInstall` calls it;
@@ -782,6 +784,7 @@ export {
782
784
  planReleaseGates,
783
785
  classifyGateExecutions,
784
786
  buildFailureIssue,
787
+ shouldRetryGhWithoutToken,
785
788
  buildReleaseNotes,
786
789
  releaseTagName,
787
790
  firstOutputLine,
@@ -881,7 +884,7 @@ export type {
881
884
  McpSeverity,
882
885
  McpCapability,
883
886
  } from './mcp-scan.js';
884
- export type { RegistryEntry, Registry } from './registry.js';
887
+ export type { RegistryEntry, Registry, ShowcaseRegistry, ShowcaseSkill } from './registry.js';
885
888
  export type { BenchmarkCheck, BenchmarkScore, BenchmarkReport, CompareResult } from './benchmark.js';
886
889
  export {
887
890
  specToOpts,
@@ -494,6 +494,12 @@ export interface MutationObservation {
494
494
  readonly outputError?: string;
495
495
  /** bounded log proving an internal runner failure received at most one retry. */
496
496
  readonly internalAttemptLog?: string;
497
+ /**
498
+ * Whether any suite the entry's `testCommand` selects NAMES the mutated module — see
499
+ * {@link suiteSelectionNamesModule}. Only read on the UNDEFENDED path, and only to add a hint:
500
+ * `false` says "check the command before the tests", never "the property is fine".
501
+ */
502
+ readonly suiteNamesModule?: boolean | 'unknown';
497
503
  }
498
504
 
499
505
  export interface MutationEntryResult {
@@ -1333,7 +1339,13 @@ function classifyMutationOutcomeWithoutAttemptLog(obs: MutationObservation): Mut
1333
1339
  applied: true,
1334
1340
  verdict: 'UNDEFENDED',
1335
1341
  drop: false,
1336
- detail: `suite stayed GREEN with the protection deleted — property UNDEFENDED: "${e.property}" (${e.file}). The suite would not notice this protection regressing (rule 2).`,
1342
+ detail: `suite stayed GREEN with the protection deleted — property UNDEFENDED: "${e.property}" (${e.file}). The suite would not notice this protection regressing (rule 2).`
1343
+ + (obs.suiteNamesModule === false
1344
+ ? ' HINT: no suite in this entry\'s testCommand NAMES this module, so check the COMMAND before the tests'
1345
+ + ' — a suite that never loads the module cannot notice its protection (MEASURED 2026-09-04: two properties'
1346
+ + ' read as UNDEFENDED for exactly this reason, and both tests existed). The hint matches the module stem,'
1347
+ + ' so a transitive import would not be seen by it: it narrows where to look, it does not decide.'
1348
+ : ''),
1337
1349
  };
1338
1350
  }
1339
1351
 
@@ -1408,6 +1420,47 @@ function classifyMutationOutcomeWithoutAttemptLog(obs: MutationObservation): Mut
1408
1420
  };
1409
1421
  }
1410
1422
 
1423
+ /**
1424
+ * Suite paths a registry `testCommand` selects, in order. Tokens that are not suite files (the
1425
+ * runner, its flags) are ignored — the command is a shell line, not a schema, so this reads the
1426
+ * shape it actually has rather than assuming one.
1427
+ */
1428
+ export function parseSuitePaths(testCommand: string): readonly string[] {
1429
+ return String(testCommand ?? '')
1430
+ .split(/\s+/)
1431
+ .filter((token) => /\.(?:test|spec)\.[cm]?[jt]sx?$/.test(token) && !token.startsWith('-'));
1432
+ }
1433
+
1434
+ /**
1435
+ * Does ANY suite the command selects even name the mutated module?
1436
+ *
1437
+ * Why this exists: `UNDEFENDED` reads as "this property has no test", and MEASURED 2026-09-04 that
1438
+ * reading was wrong twice in one run — both tests existed; the registry's `testCommand` simply did
1439
+ * not select the suites that import them (backlog 1f4e4f66). The author filed a finding about two
1440
+ * "unprotected properties" before checking the instrument, which is the failure this hint prevents.
1441
+ *
1442
+ * Deliberately a HINT, never a verdict: it matches the module's STEM in each suite's text, so a
1443
+ * suite that reaches the module through a transitive import is invisible to it. `'unknown'` when no
1444
+ * suite could be read — absence of evidence is not evidence, and a hint that guesses is worse than
1445
+ * no hint.
1446
+ */
1447
+ export function suiteSelectionNamesModule(input: {
1448
+ readonly file: string;
1449
+ readonly suitePaths: readonly string[];
1450
+ readonly readSuite: (path: string) => string | null;
1451
+ }): boolean | 'unknown' {
1452
+ const stem = String(input.file ?? '').split('/').pop()?.replace(/\.[cm]?[jt]sx?$/, '') ?? '';
1453
+ if (stem === '') return 'unknown';
1454
+ let read = 0;
1455
+ for (const suite of input.suitePaths) {
1456
+ const text = input.readSuite(suite);
1457
+ if (text === null) continue;
1458
+ read += 1;
1459
+ if (text.includes(stem)) return true;
1460
+ }
1461
+ return read === 0 ? 'unknown' : false;
1462
+ }
1463
+
1411
1464
  export function classifyMutationOutcome(obs: MutationObservation): MutationEntryResult {
1412
1465
  const result = classifyMutationOutcomeWithoutAttemptLog(obs);
1413
1466
  if (obs.internalAttemptLog === undefined) return result;