peaks-loop 4.0.49 → 4.0.51

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (179) hide show
  1. package/CHANGELOG.md +38 -0
  2. package/README-en.md +1 -1
  3. package/README.md +1 -1
  4. package/dist/cli/commands/audit-commands.js +1 -0
  5. package/dist/cli/commands/baseline-commands.js +163 -25
  6. package/dist/cli/commands/core/skill-command.js +53 -4
  7. package/dist/cli/commands/core/standards-command.d.ts +24 -0
  8. package/dist/cli/commands/core/standards-command.js +74 -0
  9. package/dist/cli/commands/hooks-commands.js +55 -38
  10. package/dist/cli/commands/share-commands.js +113 -20
  11. package/dist/cli/commands/web-commands.js +8 -1
  12. package/dist/cli/commands/workflow-lifecycle-commands.d.ts +6 -0
  13. package/dist/cli/commands/workflow-lifecycle-commands.js +64 -3
  14. package/dist/services/adapter/adapter.d.ts +30 -0
  15. package/dist/services/adapter/auto-adapter.d.ts +13 -0
  16. package/dist/services/adapter/claude-adapter.js +12 -0
  17. package/dist/services/adapter/codex-adapter.d.ts +12 -0
  18. package/dist/services/adapter/codex-adapter.js +12 -0
  19. package/dist/services/adapter/copilot-adapter.d.ts +12 -0
  20. package/dist/services/adapter/copilot-adapter.js +12 -0
  21. package/dist/services/audit/backing-detector.d.ts +25 -7
  22. package/dist/services/audit/backing-detector.js +33 -17
  23. package/dist/services/audit/enforcer-liveness.d.ts +12 -0
  24. package/dist/services/audit/enforcer-liveness.js +100 -0
  25. package/dist/services/audit/enforcers/lint-catalog-governance.d.ts +23 -11
  26. package/dist/services/audit/enforcers/lint-catalog-governance.js +10 -14
  27. package/dist/services/audit/enforcers/lint-rd-handoff-coverage.d.ts +5 -15
  28. package/dist/services/audit/enforcers/lint-rd-handoff-coverage.js +94 -25
  29. package/dist/services/audit/enforcers/lint-style.d.ts +9 -1
  30. package/dist/services/audit/enforcers/lint-style.js +38 -2
  31. package/dist/services/audit/prose-ratio-calculator.d.ts +28 -17
  32. package/dist/services/audit/prose-ratio-calculator.js +25 -18
  33. package/dist/services/audit/red-line-catalog-p2-a.js +1 -1
  34. package/dist/services/audit/red-lines-service.js +51 -7
  35. package/dist/services/capability-audit-service/independent-checker.d.ts +15 -0
  36. package/dist/services/capability-audit-service/independent-checker.js +140 -0
  37. package/dist/services/capability-audit-service/index.d.ts +3 -1
  38. package/dist/services/capability-audit-service/index.js +1 -0
  39. package/dist/services/capability-audit-service/runner.d.ts +17 -13
  40. package/dist/services/capability-audit-service/runner.js +76 -15
  41. package/dist/services/capability-audit-service/types.d.ts +48 -0
  42. package/dist/services/capability-guard-runner/contracts/J01.js +21 -22
  43. package/dist/services/capability-guard-runner/contracts/J02.d.ts +1 -1
  44. package/dist/services/capability-guard-runner/contracts/J02.js +114 -28
  45. package/dist/services/capability-guard-runner/contracts/J03.d.ts +13 -0
  46. package/dist/services/capability-guard-runner/contracts/J03.js +72 -21
  47. package/dist/services/capability-guard-runner/contracts/J04.d.ts +6 -0
  48. package/dist/services/capability-guard-runner/contracts/J04.js +65 -32
  49. package/dist/services/capability-guard-runner/contracts/J05.js +118 -16
  50. package/dist/services/capability-guard-runner/contracts/J06.d.ts +14 -0
  51. package/dist/services/capability-guard-runner/contracts/J06.js +57 -39
  52. package/dist/services/capability-guard-runner/contracts/J07.d.ts +9 -0
  53. package/dist/services/capability-guard-runner/contracts/J07.js +76 -47
  54. package/dist/services/capability-guard-runner/contracts/J08.d.ts +11 -0
  55. package/dist/services/capability-guard-runner/contracts/J08.js +66 -39
  56. package/dist/services/capability-guard-runner/contracts/J09.d.ts +13 -0
  57. package/dist/services/capability-guard-runner/contracts/J09.js +95 -39
  58. package/dist/services/capability-guard-runner/contracts/J10.d.ts +12 -0
  59. package/dist/services/capability-guard-runner/contracts/J10.js +69 -35
  60. package/dist/services/capability-guard-runner/contracts/J11.d.ts +8 -0
  61. package/dist/services/capability-guard-runner/contracts/J11.js +73 -33
  62. package/dist/services/capability-guard-runner/contracts/J12.d.ts +12 -0
  63. package/dist/services/capability-guard-runner/contracts/J12.js +66 -30
  64. package/dist/services/capability-guard-runner/contracts/J13.d.ts +11 -0
  65. package/dist/services/capability-guard-runner/contracts/J13.js +62 -40
  66. package/dist/services/capability-guard-runner/contracts/J14.d.ts +11 -0
  67. package/dist/services/capability-guard-runner/contracts/J14.js +60 -31
  68. package/dist/services/capability-guard-runner/contracts/J15.d.ts +11 -0
  69. package/dist/services/capability-guard-runner/contracts/J15.js +70 -35
  70. package/dist/services/capability-guard-runner/contracts/_shared.d.ts +24 -0
  71. package/dist/services/capability-guard-runner/contracts/_shared.js +67 -0
  72. package/dist/services/capability-guard-runner/registry.d.ts +5 -0
  73. package/dist/services/capability-guard-runner/registry.js +140 -0
  74. package/dist/services/capability-guard-runner/runner.d.ts +26 -0
  75. package/dist/services/capability-guard-runner/runner.js +63 -6
  76. package/dist/services/code/auto-compact-modes.d.ts +13 -2
  77. package/dist/services/code/auto-compact-modes.js +20 -4
  78. package/dist/services/code/post-compact-detector.js +20 -11
  79. package/dist/services/code/step-08-gate.js +21 -6
  80. package/dist/services/config/config-safety.js +11 -9
  81. package/dist/services/dispatch/sub-agent-dispatcher.d.ts +11 -30
  82. package/dist/services/dispatch/sub-agent-dispatcher.js +5 -48
  83. package/dist/services/final-review/pre-post-diff.js +10 -2
  84. package/dist/services/ide/adapters/claude-code-adapter.js +0 -1
  85. package/dist/services/ide/adapters/codex-adapter.js +1 -2
  86. package/dist/services/ide/adapters/cursor-adapter.js +1 -2
  87. package/dist/services/ide/adapters/hermes-adapter.js +1 -2
  88. package/dist/services/ide/adapters/openclaw-adapter.js +1 -2
  89. package/dist/services/ide/adapters/qoder-adapter.js +1 -2
  90. package/dist/services/ide/adapters/tongyi-lingma-adapter.js +1 -2
  91. package/dist/services/ide/adapters/trae-adapter.js +1 -2
  92. package/dist/services/ide/adapters/zcode-adapter.js +0 -1
  93. package/dist/services/ide/ide-types.d.ts +0 -2
  94. package/dist/services/observability/observability-service.d.ts +1 -1
  95. package/dist/services/scan/api-diff-types.js +20 -2
  96. package/dist/services/security/safe-settings-path.js +19 -1
  97. package/dist/services/skill/skill-search-service.d.ts +3 -3
  98. package/dist/services/standards/loop-engineering-lint.d.ts +1 -1
  99. package/dist/services/standards/loop-engineering-lint.js +6 -0
  100. package/dist/services/web/daemon-registry.js +27 -2
  101. package/dist/services/workspace/claude-settings-template.d.ts +53 -37
  102. package/dist/services/workspace/claude-settings-template.js +105 -83
  103. package/dist/services/workspace/generated-artifacts-stamp.d.ts +119 -0
  104. package/dist/services/workspace/generated-artifacts-stamp.js +167 -0
  105. package/dist/services/workspace/workspace-claude-settings-materializer.d.ts +8 -0
  106. package/dist/services/workspace/workspace-claude-settings-materializer.js +38 -3
  107. package/dist/services/workspace/workspace-service.js +11 -1
  108. package/dist/shared/fs-utils.d.ts +26 -0
  109. package/dist/shared/fs-utils.js +35 -0
  110. package/package.json +9 -7
  111. package/scripts/copy-templates.mjs +0 -12
  112. package/scripts/install-skills.mjs +154 -53
  113. package/skills/bee/peaks-perf-audit/SKILL.md +2 -2
  114. package/skills/bee/peaks-perf-audit/references/audit-protocol.md +1 -1
  115. package/skills/bee/peaks-prd/SKILL.md +3 -3
  116. package/skills/bee/peaks-prd/references/prd-for-multi-pass.md +1 -1
  117. package/skills/bee/peaks-prd/references/workflow.md +1 -1
  118. package/skills/bee/peaks-qa/SKILL.md +6 -7
  119. package/skills/bee/peaks-qa/references/external-capability-guidance.md +1 -1
  120. package/skills/bee/peaks-qa/references/qa-fanout-contract.md +1 -1
  121. package/skills/bee/peaks-qa/references/reading-handoff-frontmatter.md +2 -2
  122. package/skills/bee/peaks-rd/SKILL.md +2 -3
  123. package/skills/bee/peaks-rd/references/code-reviewer-4dim-hint.md +1 -1
  124. package/skills/bee/peaks-rd/references/external-references.md +1 -1
  125. package/skills/bee/peaks-rd/references/mandatory-perf-baseline.md +1 -1
  126. package/skills/bee/peaks-rd/references/ocr-multilang-1.8.md +2 -2
  127. package/skills/bee/peaks-rd/references/parallel-review-fanout.md +2 -2
  128. package/skills/bee/peaks-rd/references/rd-fanout-contracts.md +11 -8
  129. package/skills/bee/peaks-rd/references/rd-runbook.md +1 -1
  130. package/skills/bee/peaks-rd/references/rd-sub-agent-dispatch.md +7 -7
  131. package/skills/bee/peaks-rd/references/rd-transition-gates.md +1 -1
  132. package/skills/bee/peaks-rd/references/reading-v2-slice-results.md +1 -1
  133. package/skills/bee/peaks-rd/references/v2-12-fanout-collapse.md +7 -5
  134. package/skills/bee/peaks-rd/references/writing-handoff-frontmatter.md +3 -3
  135. package/skills/bee/peaks-reviewer/SKILL.md +1 -1
  136. package/skills/bee/peaks-security-audit/SKILL.md +3 -3
  137. package/skills/bee/peaks-security-audit/references/audit-protocol.md +1 -1
  138. package/skills/bee/peaks-txt/references/context-capsule.md +1 -1
  139. package/skills/bee/peaks-ui/SKILL.md +1 -1
  140. package/skills/peaks-audit/SKILL.md +1 -1
  141. package/skills/peaks-code/SKILL.md +20 -18
  142. package/skills/peaks-code/references/context-governance.md +1 -1
  143. package/skills/peaks-code/references/dag-orchestrator.md +3 -4
  144. package/skills/peaks-code/references/external-references.md +1 -1
  145. package/skills/peaks-code/references/external-skill-invocation.md +2 -2
  146. package/skills/peaks-code/references/fanout-mandatory.md +3 -3
  147. package/skills/peaks-code/references/frontend-only-mode.md +2 -2
  148. package/skills/peaks-code/references/gstack-integration.md +1 -1
  149. package/skills/peaks-code/references/micro-cycle.md +1 -1
  150. package/skills/peaks-code/references/periodic-checkpoint.md +4 -4
  151. package/skills/peaks-code/references/project-scan-checklist.md +1 -1
  152. package/skills/peaks-code/references/resume-detection.md +1 -1
  153. package/skills/peaks-code/references/runbook.md +6 -3
  154. package/skills/peaks-code/references/session-overload-signal-index.md +6 -4
  155. package/skills/peaks-code/references/startup-sequence.md +17 -17
  156. package/skills/peaks-code/references/step-0-8-gate.md +1 -1
  157. package/skills/peaks-code/references/step-11-memory-sediment.md +2 -2
  158. package/skills/peaks-code/references/sub-agent-dispatch.md +26 -25
  159. package/skills/peaks-code/references/swarm-dispatch-contract.md +1 -1
  160. package/skills/peaks-code/references/workflow-gates-and-types.md +3 -3
  161. package/skills/peaks-code/references/worktree-governance.md +1 -1
  162. package/skills/peaks-final-review/SKILL.md +3 -3
  163. package/skills/peaks-ide/references/audit-log-helper.md +5 -4
  164. package/skills/peaks-resume/SKILL.md +1 -1
  165. package/skills/peaks-slice-decompose/SKILL.md +4 -4
  166. package/skills/peaks-slice-decompose/references/cross-pass-edge-interpretation.md +1 -1
  167. package/skills/peaks-slice-decompose/references/granularity-decision.md +1 -1
  168. package/skills/peaks-slice-decompose/references/v2-schema.md +2 -2
  169. package/skills/peaks-solo/SKILL.md +1 -2
  170. package/dist/cli/commands/context-builder-commands.d.ts +0 -11
  171. package/dist/cli/commands/context-builder-commands.js +0 -85
  172. package/dist/services/hooks/write-gate.js +0 -111
  173. package/skills/bee/peaks-prd/references/command-migration.md +0 -3
  174. package/skills/bee/peaks-qa/references/command-migration.md +0 -3
  175. package/skills/bee/peaks-rd/references/command-migration.md +0 -3
  176. package/skills/bee/peaks-sc/references/command-migration.md +0 -3
  177. package/skills/bee/peaks-txt/references/command-migration.md +0 -3
  178. package/skills/bee/peaks-ui/references/command-migration.md +0 -3
  179. package/skills/peaks-code/references/command-migration.md +0 -3
@@ -1,21 +1,25 @@
1
1
  /**
2
- * Prose-only ratio calculator — Slice C Group G3 (v2.14.0).
2
+ * Prose-only ratio calculator — Slice C Group G3 (v2.14.0),
3
+ * corrected in S3 of the 2026-09-15 diagnosis-remediation job.
3
4
  *
4
- * Computes the prose-only ratio for a set of red-line entries. Per
5
- * spec §10.2 + the v2.12.1 reform (`.peaks/memory/2026-06-27-
6
- * prose-only-catalog-followup.md`), an entry counts as prose-only
7
- * only when BOTH:
8
- * 1. `backing === 'prose-only'`
9
- * 2. `informational !== true`
5
+ * An entry counts as prose-only when `backing === 'prose-only'`.
6
+ * Full stop. There is no second condition.
10
7
  *
11
- * The 80 discovered advisory SKILL.md phrases (auto-marked
12
- * `informational=true` by `classifier.ts:141`) are excluded from
13
- * the ratio so the gate (≤ 5% per slice C AC A3.1) reflects the
14
- * actionable backlog.
8
+ * The pre-S3 version also required `informational !== true`, on the
9
+ * reasoning that auto-discovered advisory SKILL.md phrases are "not
10
+ * actionable red lines". Whatever the merits of that reading, the effect
11
+ * was to move 44 of 152 rows — 29% of the catalog — out of the
12
+ * denominator, so the gate reported `proseOnly: 0` while the same JSON
13
+ * carried 44 rows with `"backing": "prose-only"`. A metric whose
14
+ * denominator can be redefined by the code it measures is not a metric.
15
15
  *
16
- * Karpathy §2 simplicity: one exported function plus a thin
17
- * calculator interface; no I/O. The pure form makes the ≥8
18
- * test cases in prose-ratio-calculator.test.ts trivial.
16
+ * `informational` survives as a triage label — `discoveredProseOnly`
17
+ * below counts those rows — but it no longer moves any number that the
18
+ * ratio is computed from. Expect the ratio to look much worse than it
19
+ * did; that is this correction working.
20
+ *
21
+ * Karpathy §2 simplicity: one exported function plus a thin calculator
22
+ * interface; no I/O.
19
23
  */
20
24
  /** Default target: 5% (per A3.1). */
21
25
  export const DEFAULT_PROSE_RATIO_TARGET = 0.05;
@@ -24,18 +28,20 @@ export function computeProseRatio(entries, options = {}) {
24
28
  let cliBacked = 0;
25
29
  let partial = 0;
26
30
  let proseOnly = 0;
31
+ let discoveredProseOnly = 0;
27
32
  let informational = 0;
28
33
  for (const entry of entries) {
29
- if (entry.informational === true) {
34
+ if (entry.informational === true)
30
35
  informational += 1;
31
- continue;
32
- }
33
36
  if (entry.backing === 'cli-backed')
34
37
  cliBacked += 1;
35
38
  else if (entry.backing === 'partial')
36
39
  partial += 1;
37
- else if (entry.backing === 'prose-only')
40
+ else if (entry.backing === 'prose-only') {
38
41
  proseOnly += 1;
42
+ if (entry.informational === true)
43
+ discoveredProseOnly += 1;
44
+ }
39
45
  }
40
46
  const totalRedLines = entries.length;
41
47
  const ratio = totalRedLines === 0 ? 0 : proseOnly / totalRedLines;
@@ -44,6 +50,7 @@ export function computeProseRatio(entries, options = {}) {
44
50
  cliBacked,
45
51
  partial,
46
52
  proseOnly,
53
+ discoveredProseOnly,
47
54
  informational,
48
55
  ratio,
49
56
  target,
@@ -216,7 +216,7 @@ const PRD_ARTIFACT_HANDOFF = {
216
216
  };
217
217
  const RD_HANDOFF_CONTRACT = {
218
218
  id: 'rl-rd-handoff-contract-001',
219
- rule: 'peaks-rd SKILL.md must declare the QA-handoff BLOCKING contract (tech-doc + perf-baseline)',
219
+ rule: 'peaks-rd must not hand off to QA without a non-empty RD artifact under rd/requests/',
220
220
  markers: ['BLOCKING'],
221
221
  phrases: ['do not hand off to qa without', 'tech-doc', 'perf-baseline'],
222
222
  enforcerRef: 'src/services/audit/enforcers/lint-rd-handoff-coverage.ts',
@@ -16,6 +16,8 @@ import { existsSync, readdirSync, readFileSync } from 'node:fs';
16
16
  import { join } from 'node:path';
17
17
  import { classifyFiles } from './classifier.js';
18
18
  import { classifyBackingBatch } from './backing-detector.js';
19
+ import { computeLiveEnforcers } from './enforcer-liveness.js';
20
+ import { RED_LINE_CATALOG } from './red-line-catalog.js';
19
21
  import { scanSkillsTree } from './scanners/skills-tree-scanner.js';
20
22
  import { scanRulesTree } from './scanners/rules-tree-scanner.js';
21
23
  import { scanOpenSpecTree } from './scanners/openspec-scanner.js';
@@ -28,6 +30,7 @@ import { readSkillFiles, lintSectionShape, lintSectionOrder, lintFrontmatterShap
28
30
  import { lintRefPathResolves, lintNoBrokenMkdir, lintNoPwdSymlinkJumps, lintNoRelativeArchivePaths, } from './enforcers/lint-reference-integrity.js';
29
31
  import { lintCliBackMandatorText, lintCliBackNoOrphanBlocking, lintCliBackNoOrphanMustNot, } from './enforcers/lint-cli-back.js';
30
32
  import { lintNoFluff, lintNoClosingPrompt, lintStatusHeader, } from './enforcers/lint-output-style.js';
33
+ import { lintRdHandoffContract, lintRdCoverageDiscipline, } from './enforcers/lint-rd-handoff-coverage.js';
31
34
  import { lintOpenSpecAcceptanceBullets, lintOpenSpecSpecReference,
32
35
  // (Removed in v2.11.0 Group A: `lintTechDocPresenceShape`)
33
36
  lintPeaksDoctorAcknowledged, } from './enforcers/lint-workflow-shape.js';
@@ -59,15 +62,16 @@ function tally(entries) {
59
62
  let partial = 0;
60
63
  let proseOnly = 0;
61
64
  for (const entry of entries) {
62
- // v2.12.1 catalog governance: informational entries (auto-discovered
63
- // prose phrases without a catalog template) are counted in the
64
- // total but excluded from `proseOnly` so the ratio (per spec §10.2
65
- // ≤ 5%) reflects the actionable backlog only.
65
+ // S3 of the 2026-09-15 diagnosis-remediation job: `informational` is a
66
+ // triage label, not a ratio input. The pre-S3 version skipped
67
+ // informational rows here too, which is how the audit came to report
68
+ // `proseOnly: 0` while carrying 44 `"backing": "prose-only"` rows in
69
+ // the same envelope. Every prose-only row is now counted.
66
70
  if (entry.backing === 'cli-backed')
67
71
  cliBacked++;
68
72
  else if (entry.backing === 'partial')
69
73
  partial++;
70
- else if (!entry.informational)
74
+ else if (entry.backing === 'prose-only')
71
75
  proseOnly++;
72
76
  }
73
77
  return {
@@ -87,7 +91,12 @@ export function runRedLinesAudit(input) {
87
91
  const openspec = scanOpenSpecTree({ projectRoot: input.projectRoot });
88
92
  const fileInputs = buildFileInputs(skills, rules, openspec);
89
93
  const classified = classifyFiles(fileInputs);
90
- const backed = classifyBackingBatch(classified.entries, input.projectRoot);
94
+ // A9: `cli-backed` requires a call site, not just a file on disk. The
95
+ // live set is computed once for the whole catalog; `null` means the
96
+ // project has no `src/` tree and liveness is undecidable, in which case
97
+ // the detector falls back to the pre-A9 "file exists" rule.
98
+ const liveness = computeLiveEnforcers(input.projectRoot, RED_LINE_CATALOG.map((entry) => entry.enforcerRef).filter((ref) => ref !== null));
99
+ const backed = classifyBackingBatch(classified.entries, input.projectRoot, liveness.live);
91
100
  // Sub-agent-sid enforcer (Task 2): dogfoods Slice 0.5 sid-naming-guard.
92
101
  const subAgentSids = findInvalidSubAgentSids(input.projectRoot);
93
102
  const runtimeSids = findInvalidRuntimeSids(input.projectRoot);
@@ -97,7 +106,22 @@ export function runRedLinesAudit(input) {
97
106
  ...openspec.warnings,
98
107
  ...classified.warnings.map((message) => ({ file: '(classifier)', message })),
99
108
  ...backed.warnings.map((message) => ({ file: '(backing-detector)', message })),
109
+ ...liveness.warnings.map((message) => ({ file: '(enforcer-liveness)', message })),
100
110
  ];
111
+ // A9: name every enforcer that was downgraded, so the drop in
112
+ // `cliBacked` is attributable rather than mysterious.
113
+ if (liveness.unknown) {
114
+ warnings.push({
115
+ file: '(enforcer-liveness)',
116
+ message: 'no src/ tree under the project root; enforcer liveness is undecidable, so `cli-backed` falls back to the pre-A9 "the enforcer file exists" rule',
117
+ });
118
+ }
119
+ for (const ref of backed.deadEnforcers) {
120
+ warnings.push({
121
+ file: ref,
122
+ message: 'enforcer file exists but nothing outside src/services/audit/enforcers/ imports it; red lines backed by it are counted as prose-only',
123
+ });
124
+ }
101
125
  if (subAgentSids.scanned && subAgentSids.invalid.length > 0) {
102
126
  for (const sid of subAgentSids.invalid) {
103
127
  warnings.push({
@@ -261,15 +285,31 @@ export function runRedLinesAudit(input) {
261
285
  const skillsRoot = join(input.projectRoot, 'skills');
262
286
  if (existsSync(skillsRoot)) {
263
287
  const skillNames = [];
288
+ const skippedRoots = [];
264
289
  for (const entry of readdirSync(skillsRoot, { withFileTypes: true })) {
265
290
  if (!entry.isDirectory())
266
291
  continue;
267
292
  if (entry.name.startsWith('.'))
268
293
  continue;
269
- if (!existsSync(join(skillsRoot, entry.name, 'SKILL.md')))
294
+ if (!existsSync(join(skillsRoot, entry.name, 'SKILL.md'))) {
295
+ skippedRoots.push(entry.name);
270
296
  continue;
297
+ }
271
298
  skillNames.push(entry.name);
272
299
  }
300
+ // A11 scope limit, declared rather than left implicit: only skills
301
+ // with a top-level `SKILL.md` are linted. Nested roots such as
302
+ // `skills/bee/` hold real skills (`peaks-perf-audit`, `peaks-rd`, …)
303
+ // that the classifier DOES see (their markers appear as
304
+ // `rl-discovered-skills-bee-*` rows) but that every Theme A–G
305
+ // lint-style enforcer skips. A clean Theme A result therefore means
306
+ // "clean among the linted skills", not "clean everywhere".
307
+ for (const root of skippedRoots) {
308
+ warnings.push({
309
+ file: `skills/${root}`,
310
+ message: 'no top-level SKILL.md — this directory and every skill nested under it are skipped by all lint-style enforcers (including rl-section-hard-contracts-001); their red-line rows are unverified',
311
+ });
312
+ }
273
313
  const skillFiles = readSkillFiles(skillsRoot, skillNames);
274
314
  for (const skill of skillFiles) {
275
315
  const refsDir = join(skillsRoot, skill.name, 'references');
@@ -291,6 +331,10 @@ export function runRedLinesAudit(input) {
291
331
  ...lintNoFluff(skill),
292
332
  ...lintNoClosingPrompt(skill),
293
333
  ...lintPeaksDoctorAcknowledged(skill),
334
+ // A10: the RD handoff gate reads the artifact, not the SKILL.md
335
+ // sentence that promises one. See lint-rd-handoff-coverage.ts.
336
+ ...lintRdHandoffContract(skill, input.projectRoot),
337
+ ...lintRdCoverageDiscipline(skill),
294
338
  ];
295
339
  for (const hit of lintHits) {
296
340
  enforcerFindings.push({
@@ -0,0 +1,15 @@
1
+ import { type CapabilityBaselineRow } from '../capability-baseline/types.js';
2
+ import type { GuardContract, GuardRunResult } from '../capability-guard-runner/types.js';
3
+ import type { IndependentCheckResult } from './types.js';
4
+ export interface IndependentCheckInput {
5
+ readonly projectRoot: string;
6
+ readonly baselineRows: ReadonlyArray<CapabilityBaselineRow>;
7
+ readonly contracts: ReadonlyArray<GuardContract>;
8
+ readonly guardResults: ReadonlyArray<GuardRunResult>;
9
+ }
10
+ /**
11
+ * Run the credential-free independent evaluation. The verdict is `drifted`
12
+ * whenever a concrete deviation is observed — this function has no path that
13
+ * returns `consistent` without having checked.
14
+ */
15
+ export declare function runIndependentCheck(input: IndependentCheckInput): IndependentCheckResult;
@@ -0,0 +1,140 @@
1
+ // src/services/capability-audit-service/independent-checker.ts
2
+ //
3
+ // The live, credential-free audit scorer.
4
+ //
5
+ // WHY A DETERMINISTIC CHECKER AND NOT AN LLM CALL
6
+ // ----------------------------------------------
7
+ // `publish.yml` is a secretless OIDC trusted-publishing workflow: `id-token:
8
+ // write`, no npm token, and no LLM credential of any kind in the environment.
9
+ // An LLM scorer therefore cannot run in the gate that decides whether a
10
+ // release happens. Adding a long-lived API secret to a secretless pipeline to
11
+ // serve that gate would be a threat-model regression, and an LLM verdict is
12
+ // non-deterministic — the same commit could flip between runs. So the live
13
+ // scorer must be credential-free.
14
+ //
15
+ // WHY IT IS STILL "INDEPENDENT"
16
+ // -----------------------------
17
+ // Independence is a property of the information channel, not of the substrate
18
+ // (RL-5 constrains what the scorer READS: `scorer.reads: evaluation_package_only`
19
+ // — it never requires a model). The scorer this replaces was handed
20
+ // `{baselineJourneyId, guard}` and asked to re-state it; an LLM given that same
21
+ // payload would be exactly as much a rubber stamp. The disease was the payload.
22
+ //
23
+ // This checker answers a question no guard contract can answer, from inputs no
24
+ // guard reads:
25
+ // - a guard sees only ITSELF, so it cannot report that the observation set was
26
+ // silently narrowed, or that a frozen row is armed by no contract at all;
27
+ // - this checker sees the whole frozen claim set AND the whole registry.
28
+ // It reads only the evaluation package — no author reasoning, no session id, no
29
+ // self-praise framing — so RL-5's exclusions hold by construction.
30
+ //
31
+ // WHAT IT DOES NOT COVER (stated, not hidden)
32
+ // -------------------------------------------
33
+ // It does not read `forbiddenChanges` prose, and it cannot judge behaviour
34
+ // beyond what the 15 guard contracts already exercise. Its claim is narrower
35
+ // than "the 15 journeys are intact"; `coverage` in the result discloses exactly
36
+ // how narrow, so `consistent` is never read as more than it is.
37
+ import { existsSync } from 'node:fs';
38
+ import { join } from 'node:path';
39
+ import { P0_JOURNEY_IDS } from '../capability-baseline/types.js';
40
+ function finding(code, journeyId, detail) {
41
+ return { code, journeyId, detail };
42
+ }
43
+ /**
44
+ * The observed journey set must be exactly the frozen P0 set. `runAllGuards`
45
+ * aggregates whatever contracts it was handed, so a registry that lost a
46
+ * journey reports a clean `pass: 14 / fail: 0` — a narrower check that looks
47
+ * exactly like a green one. Nothing in the guard results can say so; only the
48
+ * frozen set can.
49
+ */
50
+ function checkObservationSet(frozen, observed) {
51
+ const out = [];
52
+ const counts = new Map();
53
+ for (const j of observed)
54
+ counts.set(j, (counts.get(j) ?? 0) + 1);
55
+ for (const j of frozen) {
56
+ const n = counts.get(j) ?? 0;
57
+ if (n === 0)
58
+ out.push(finding('OBSERVATION_INCOMPLETE', j, `${j} is in the frozen baseline but no guard result was observed for it`));
59
+ else if (n > 1)
60
+ out.push(finding('OBSERVATION_INCOMPLETE', j, `${j} produced ${String(n)} guard results; the frozen baseline declares it once`));
61
+ }
62
+ for (const j of counts.keys()) {
63
+ if (!frozen.includes(j))
64
+ out.push(finding('OBSERVATION_INCOMPLETE', j, `${j} was observed but is not a frozen P0 journey`));
65
+ }
66
+ return out;
67
+ }
68
+ /** The frozen claim set itself must be the P0 set, with no duplicate rows. */
69
+ function checkFrozenRows(rows) {
70
+ const out = [];
71
+ const seen = new Map();
72
+ for (const r of rows)
73
+ seen.set(r.journeyId, (seen.get(r.journeyId) ?? 0) + 1);
74
+ for (const j of P0_JOURNEY_IDS) {
75
+ const n = seen.get(j) ?? 0;
76
+ if (n === 0)
77
+ out.push(finding('BASELINE_ROW_SET_INVALID', j, `frozen baseline has no row for ${j}`));
78
+ else if (n > 1)
79
+ out.push(finding('BASELINE_ROW_SET_INVALID', j, `frozen baseline declares ${j} ${String(n)} times`));
80
+ }
81
+ for (const j of seen.keys()) {
82
+ if (!P0_JOURNEY_IDS.includes(j))
83
+ out.push(finding('BASELINE_ROW_SET_INVALID', j, `frozen baseline declares ${j}, which is not a P0 journey`));
84
+ }
85
+ return out;
86
+ }
87
+ /**
88
+ * Every `sourceFiles` entry of every frozen row must still exist. The guard
89
+ * contracts check this too, but only through their own contract — so a
90
+ * contract rewritten to drop that probe takes the check with it. Reading the
91
+ * frozen text directly means the binding survives such a rewrite.
92
+ */
93
+ function checkSourceBindings(projectRoot, rows) {
94
+ const out = [];
95
+ for (const row of rows) {
96
+ for (const f of row.sourceFiles) {
97
+ if (!existsSync(join(projectRoot, f))) {
98
+ out.push(finding('SOURCE_FILE_MISSING', row.journeyId, `frozen sourceFiles entry "${f}" is not on disk`));
99
+ }
100
+ }
101
+ }
102
+ return out;
103
+ }
104
+ function countArmed(rows, contracts) {
105
+ let armed = 0;
106
+ for (const row of rows) {
107
+ for (const inv of row.invariants) {
108
+ if (contracts.some((c) => c.source.baselineRow === row.journeyId && c.source.invariant === inv))
109
+ armed += 1;
110
+ }
111
+ }
112
+ return armed;
113
+ }
114
+ /**
115
+ * Run the credential-free independent evaluation. The verdict is `drifted`
116
+ * whenever a concrete deviation is observed — this function has no path that
117
+ * returns `consistent` without having checked.
118
+ */
119
+ export function runIndependentCheck(input) {
120
+ const observed = input.guardResults.map((r) => r.journeyId);
121
+ const findings = [
122
+ ...checkFrozenRows(input.baselineRows),
123
+ ...checkObservationSet(input.baselineRows.map((r) => r.journeyId), observed),
124
+ ...checkSourceBindings(input.projectRoot, input.baselineRows)
125
+ ];
126
+ const coverage = {
127
+ observations: observed.length,
128
+ observationsExpected: P0_JOURNEY_IDS.length,
129
+ invariantsFrozen: input.baselineRows.reduce((n, r) => n + r.invariants.length, 0),
130
+ invariantsArmed: countArmed(input.baselineRows, input.contracts),
131
+ // Disclosed, not checked: free-text prohibitions cannot be judged
132
+ // deterministically without turning a keyword scan into a fake verdict.
133
+ forbiddenChangesUnverified: input.baselineRows.reduce((n, r) => n + r.forbiddenChanges.length, 0)
134
+ };
135
+ return {
136
+ verdict: findings.length === 0 ? 'consistent' : 'drifted',
137
+ findings,
138
+ coverage
139
+ };
140
+ }
@@ -1,3 +1,5 @@
1
1
  export { crossCheck } from './cross-check.js';
2
+ export { runIndependentCheck } from './independent-checker.js';
3
+ export type { IndependentCheckInput } from './independent-checker.js';
2
4
  export { isStale } from './staleness.js';
3
- export type { AuditVerdict, AuditEvidenceKind, AuditDimension, CrossCheck, CapabilityAuditResult } from './types.js';
5
+ export type { AuditVerdict, AuditEvidenceKind, AuditDimension, AuditFinding, AuditFindingCode, AuditCoverage, IndependentCheckResult, CrossCheck, CapabilityAuditResult } from './types.js';
@@ -1,2 +1,3 @@
1
1
  export { crossCheck } from './cross-check.js';
2
+ export { runIndependentCheck } from './independent-checker.js';
2
3
  export { isStale } from './staleness.js';
@@ -1,21 +1,25 @@
1
1
  import type { CapabilityAuditResult } from './types.js';
2
- import type { JourneyId } from '../capability-baseline/types.js';
3
- import type { GuardRunResult } from '../capability-guard-runner/types.js';
2
+ import type { CapabilityBaselineRow, JourneyId } from '../capability-baseline/types.js';
3
+ import type { GuardContract, GuardRunResult } from '../capability-guard-runner/types.js';
4
+ /**
5
+ * `stub` means the "independent" verdict came from a hard-coded response, not
6
+ * from a separate context. A stub is not an evaluation, so an audit that used
7
+ * one is marked `degraded` and can never report `consistent`.
8
+ *
9
+ * `live` runs the deterministic independent checker: a real separate-context
10
+ * evaluation that needs no credentials, which is why it is the only kind that
11
+ * can run inside the secretless OIDC publish gate.
12
+ */
13
+ export type AuditScorerMode = 'stub' | 'live';
4
14
  export interface RunAuditInput {
5
15
  readonly projectRoot: string;
6
16
  readonly sessionId: string;
7
17
  readonly journeyId: JourneyId;
8
- readonly llmRunner: {
9
- call(system: string, user: string, opts: {
10
- maxTokens: number;
11
- }): Promise<{
12
- output: string;
13
- tokens: {
14
- input: number;
15
- output: number;
16
- };
17
- }>;
18
- };
18
+ readonly scorerMode: AuditScorerMode;
19
+ /** The frozen claim set under audit. */
20
+ readonly baselineRows: ReadonlyArray<CapabilityBaselineRow>;
21
+ /** The arming witness: which frozen invariants some contract enforces. */
22
+ readonly contracts: ReadonlyArray<GuardContract>;
19
23
  readonly guardSummary: {
20
24
  readonly pass: number;
21
25
  readonly fail: number;
@@ -1,29 +1,87 @@
1
1
  import { mkdirSync, writeFileSync } from 'node:fs';
2
2
  import { join } from 'node:path';
3
3
  import { crossCheck } from './cross-check.js';
4
- const SYSTEM = 'You are an INDEPENDENT audit scorer. Compare the supplied capability baseline to the supplied current behavior summary. Output a single JSON object: {"verdict":"consistent" | "drifted" | "inconclusive"}. No prose.';
4
+ import { runIndependentCheck } from './independent-checker.js';
5
+ function scoreFor(status) {
6
+ return status === 'pass' ? 1 : status === 'fail' ? 0 : 0.5;
7
+ }
5
8
  export async function runAudit(input) {
6
- const userPayload = JSON.stringify({ baselineJourneyId: input.journeyId, guard: input.guardSummary });
7
- const r = await input.llmRunner.call(SYSTEM, userPayload, { maxTokens: 200 });
8
- const { verdict: independentVerdict } = JSON.parse(r.output);
9
+ const degraded = input.scorerMode === 'stub';
10
+ // A stub run performs no evaluation, so the checker is not run either — its
11
+ // result would be misread as an evaluation that happened.
12
+ const check = degraded
13
+ ? null
14
+ : runIndependentCheck({
15
+ projectRoot: input.projectRoot,
16
+ baselineRows: input.baselineRows,
17
+ contracts: input.contracts,
18
+ guardResults: input.guardSummary.results
19
+ });
9
20
  const xc = crossCheck({
10
21
  guardPass: input.guardSummary.pass,
11
22
  guardFail: input.guardSummary.fail,
12
- independentPass: independentVerdict === 'consistent' ? 1 : 0,
13
- independentFail: independentVerdict === 'drifted' ? 1 : 0,
23
+ // A degraded run has no independent verdict to compare; 0/0 keeps the
24
+ // cross-check shape without inventing one.
25
+ independentPass: check?.verdict === 'consistent' ? 1 : 0,
26
+ independentFail: check?.verdict === 'drifted' ? 1 : 0,
14
27
  karpathy: 'skipped'
15
28
  });
16
- let verdict = independentVerdict;
17
- if (xc.guardVsAudit === 'diverge')
29
+ // S1's rule is unchanged and load-bearing: a run that performed no separate
30
+ // evaluation can never be `consistent`. S11 adds the live branch. Every
31
+ // concrete deviation — a failed guard contract, or a finding from the
32
+ // independent checker — reports `drifted` instead of hiding behind
33
+ // `inconclusive`. So `inconclusive` is now reachable only when no evaluation
34
+ // ran at all, which is what it should mean.
35
+ let verdict = 'consistent';
36
+ if (degraded)
18
37
  verdict = 'inconclusive';
19
- const dimensions = [{
38
+ else if (input.guardSummary.fail > 0)
39
+ verdict = 'drifted';
40
+ else if ((check?.findings.length ?? 0) > 0)
41
+ verdict = 'drifted';
42
+ // One dimension per journey actually run, scored from the guard result —
43
+ // previously this was a single row whose score was derived from the stub.
44
+ const dimensions = input.guardSummary.results.map((g) => {
45
+ // When the contract fails, include the diff detail in the evidence summary
46
+ // so the gate step log (and any artifact) carries a real diagnostic
47
+ // instead of just "workflow-trace → fail". The summary is bounded so a
48
+ // runaway diff can't bloat every dimension; the contract itself is the
49
+ // authoritative source.
50
+ const detail = g.status === 'fail' && g.diff
51
+ ? ` | ${g.diff.reason}: ${g.diff.after}`.slice(0, 4000)
52
+ : '';
53
+ return {
54
+ journeyId: g.journeyId,
55
+ consistencyScore: scoreFor(g.status),
56
+ evidence: [{
57
+ kind: 'guard-run',
58
+ ref: `capability-guard-runner:${g.journeyId}`,
59
+ summary: `${g.contract} → ${g.status}${detail}`
60
+ }]
61
+ };
62
+ });
63
+ if (dimensions.length === 0) {
64
+ dimensions.push({
20
65
  journeyId: input.journeyId,
21
66
  consistencyScore: verdict === 'consistent' ? 1 : verdict === 'drifted' ? 0 : 0.5,
22
- evidence: [
23
- { kind: 'guard-run', ref: `capability-guard-runner:${input.guardSummary.total}`, summary: `${input.guardSummary.pass} pass / ${input.guardSummary.fail} fail` },
24
- { kind: 'independent-eval', ref: 'audit-llm-context', summary: `independent verdict: ${independentVerdict}` }
25
- ]
26
- }];
67
+ evidence: [{ kind: 'guard-run', ref: 'capability-guard-runner:0', summary: 'no contract results were supplied' }]
68
+ });
69
+ }
70
+ const independentRef = degraded ? 'audit-independent-checker:stub' : 'audit-independent-checker:deterministic';
71
+ const first = dimensions[0];
72
+ dimensions[0] = {
73
+ ...first,
74
+ evidence: [
75
+ ...first.evidence,
76
+ {
77
+ kind: 'independent-eval',
78
+ ref: independentRef,
79
+ summary: check === null
80
+ ? 'degraded: stub scorer (no independent context ran); the verdict was not derived from an evaluation'
81
+ : `independent verdict: ${check.verdict}; observations ${String(check.coverage.observations)}/${String(check.coverage.observationsExpected)}; invariants armed ${String(check.coverage.invariantsArmed)}/${String(check.coverage.invariantsFrozen)}; findings: ${check.findings.length === 0 ? 'none' : check.findings.map((f) => `${f.code}(${f.journeyId})`).join(',')}`
82
+ }
83
+ ]
84
+ };
27
85
  const auditId = `audit-${Date.now()}-${Math.random().toString(16).slice(2, 8)}`;
28
86
  const out = {
29
87
  auditId,
@@ -31,7 +89,10 @@ export async function runAudit(input) {
31
89
  verdict,
32
90
  dimensions,
33
91
  crossCheck: xc,
34
- requiresUserDecision: verdict === 'inconclusive'
92
+ requiresUserDecision: verdict === 'inconclusive',
93
+ degraded,
94
+ findings: check === null ? null : check.findings,
95
+ coverage: check === null ? null : check.coverage
35
96
  };
36
97
  const dir = join(input.projectRoot, '.peaks', '_runtime', input.sessionId, 'capability-audit');
37
98
  mkdirSync(dir, { recursive: true });
@@ -14,6 +14,41 @@ export interface CrossCheck {
14
14
  readonly guardVsAudit: 'agree' | 'diverge' | 'partial';
15
15
  readonly karpathyVsAudit: 'agree' | 'diverge' | 'partial';
16
16
  }
17
+ /**
18
+ * Why an independent verdict came out `drifted`. Each code names a concrete,
19
+ * inspectable deviation rather than a summary judgement.
20
+ */
21
+ export type AuditFindingCode =
22
+ /** The observed journey set is not the frozen P0 set. */
23
+ 'OBSERVATION_INCOMPLETE'
24
+ /** The frozen baseline's own row set is not the P0 set. */
25
+ | 'BASELINE_ROW_SET_INVALID'
26
+ /** A frozen `sourceFiles` entry no longer exists on disk. */
27
+ | 'SOURCE_FILE_MISSING';
28
+ export interface AuditFinding {
29
+ readonly code: AuditFindingCode;
30
+ readonly journeyId: JourneyId;
31
+ readonly detail: string;
32
+ }
33
+ /**
34
+ * How wide the audit's claim actually is. Reported alongside the verdict so
35
+ * `consistent` is never read as broader than it is: the check verifies the
36
+ * frozen row set, the observation set and the file bindings — it does not
37
+ * evaluate `forbiddenChanges` prose, and it judges no behaviour beyond what
38
+ * the guard contracts already exercise.
39
+ */
40
+ export interface AuditCoverage {
41
+ readonly observations: number;
42
+ readonly observationsExpected: number;
43
+ readonly invariantsFrozen: number;
44
+ readonly invariantsArmed: number;
45
+ readonly forbiddenChangesUnverified: number;
46
+ }
47
+ export interface IndependentCheckResult {
48
+ readonly verdict: 'consistent' | 'drifted';
49
+ readonly findings: ReadonlyArray<AuditFinding>;
50
+ readonly coverage: AuditCoverage;
51
+ }
17
52
  export interface CapabilityAuditResult {
18
53
  readonly auditId: string;
19
54
  readonly auditedAt: string;
@@ -21,4 +56,17 @@ export interface CapabilityAuditResult {
21
56
  readonly dimensions: ReadonlyArray<AuditDimension>;
22
57
  readonly crossCheck: CrossCheck;
23
58
  readonly requiresUserDecision: boolean;
59
+ /**
60
+ * True when no separate-context evaluation ran at all — i.e. the scorer was
61
+ * the stub, not the deterministic independent checker. A degraded audit can
62
+ * never be `consistent`.
63
+ */
64
+ readonly degraded: boolean;
65
+ /**
66
+ * The independent checker's findings, in the order it produced them. Empty
67
+ * on a `consistent` live run; `null` on a degraded run, where no check ran.
68
+ */
69
+ readonly findings: ReadonlyArray<AuditFinding> | null;
70
+ /** How wide this audit's claim is; `null` on a degraded run. */
71
+ readonly coverage: AuditCoverage | null;
24
72
  }