peaks-loop 4.0.48 → 4.0.50

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (156) hide show
  1. package/CHANGELOG.md +44 -0
  2. package/README-en.md +1 -1
  3. package/README.md +1 -1
  4. package/dist/cli/commands/audit-commands.js +1 -0
  5. package/dist/cli/commands/baseline-commands.js +163 -25
  6. package/dist/cli/commands/compact-command.js +1 -3
  7. package/dist/cli/commands/core/skill-command.js +53 -4
  8. package/dist/cli/commands/core/standards-command.d.ts +24 -0
  9. package/dist/cli/commands/core/standards-command.js +74 -0
  10. package/dist/cli/commands/feedback-commands.d.ts +11 -7
  11. package/dist/cli/commands/feedback-commands.js +49 -17
  12. package/dist/cli/commands/final-review-commands.js +12 -0
  13. package/dist/cli/commands/hooks-commands.js +55 -38
  14. package/dist/cli/commands/loop-eval-commands.js +22 -6
  15. package/dist/cli/commands/share-commands.js +37 -11
  16. package/dist/cli/commands/slice-integrate-commands.js +17 -0
  17. package/dist/cli/commands/web-commands.js +8 -1
  18. package/dist/cli/commands/workflow-lifecycle-commands.d.ts +6 -0
  19. package/dist/cli/commands/workflow-lifecycle-commands.js +64 -3
  20. package/dist/services/adapter/adapter.d.ts +30 -0
  21. package/dist/services/adapter/auto-adapter.d.ts +13 -0
  22. package/dist/services/adapter/claude-adapter.js +12 -0
  23. package/dist/services/adapter/codex-adapter.d.ts +12 -0
  24. package/dist/services/adapter/codex-adapter.js +12 -0
  25. package/dist/services/adapter/copilot-adapter.d.ts +12 -0
  26. package/dist/services/adapter/copilot-adapter.js +12 -0
  27. package/dist/services/artifacts/artifact-prerequisites.js +10 -0
  28. package/dist/services/artifacts/request-artifact-service.js +59 -38
  29. package/dist/services/audit/backing-detector.d.ts +25 -7
  30. package/dist/services/audit/backing-detector.js +33 -17
  31. package/dist/services/audit/enforcer-liveness.d.ts +12 -0
  32. package/dist/services/audit/enforcer-liveness.js +100 -0
  33. package/dist/services/audit/enforcers/active-skill-resolver.js +14 -1
  34. package/dist/services/audit/enforcers/lint-catalog-governance.d.ts +23 -11
  35. package/dist/services/audit/enforcers/lint-catalog-governance.js +10 -14
  36. package/dist/services/audit/enforcers/lint-rd-handoff-coverage.d.ts +5 -15
  37. package/dist/services/audit/enforcers/lint-rd-handoff-coverage.js +94 -25
  38. package/dist/services/audit/enforcers/lint-style.d.ts +9 -1
  39. package/dist/services/audit/enforcers/lint-style.js +38 -2
  40. package/dist/services/audit/prose-ratio-calculator.d.ts +28 -17
  41. package/dist/services/audit/prose-ratio-calculator.js +25 -18
  42. package/dist/services/audit/red-line-catalog-p2-a.js +1 -1
  43. package/dist/services/audit/red-lines-service.js +51 -7
  44. package/dist/services/capability-audit-service/independent-checker.d.ts +15 -0
  45. package/dist/services/capability-audit-service/independent-checker.js +140 -0
  46. package/dist/services/capability-audit-service/index.d.ts +3 -1
  47. package/dist/services/capability-audit-service/index.js +1 -0
  48. package/dist/services/capability-audit-service/runner.d.ts +17 -13
  49. package/dist/services/capability-audit-service/runner.js +76 -15
  50. package/dist/services/capability-audit-service/types.d.ts +48 -0
  51. package/dist/services/capability-guard-runner/contracts/J01.js +21 -22
  52. package/dist/services/capability-guard-runner/contracts/J02.d.ts +1 -1
  53. package/dist/services/capability-guard-runner/contracts/J02.js +114 -28
  54. package/dist/services/capability-guard-runner/contracts/J03.d.ts +13 -0
  55. package/dist/services/capability-guard-runner/contracts/J03.js +72 -21
  56. package/dist/services/capability-guard-runner/contracts/J04.d.ts +6 -0
  57. package/dist/services/capability-guard-runner/contracts/J04.js +65 -32
  58. package/dist/services/capability-guard-runner/contracts/J05.js +118 -16
  59. package/dist/services/capability-guard-runner/contracts/J06.d.ts +14 -0
  60. package/dist/services/capability-guard-runner/contracts/J06.js +57 -39
  61. package/dist/services/capability-guard-runner/contracts/J07.d.ts +9 -0
  62. package/dist/services/capability-guard-runner/contracts/J07.js +76 -47
  63. package/dist/services/capability-guard-runner/contracts/J08.d.ts +11 -0
  64. package/dist/services/capability-guard-runner/contracts/J08.js +66 -39
  65. package/dist/services/capability-guard-runner/contracts/J09.d.ts +13 -0
  66. package/dist/services/capability-guard-runner/contracts/J09.js +95 -39
  67. package/dist/services/capability-guard-runner/contracts/J10.d.ts +12 -0
  68. package/dist/services/capability-guard-runner/contracts/J10.js +69 -35
  69. package/dist/services/capability-guard-runner/contracts/J11.d.ts +8 -0
  70. package/dist/services/capability-guard-runner/contracts/J11.js +73 -33
  71. package/dist/services/capability-guard-runner/contracts/J12.d.ts +12 -0
  72. package/dist/services/capability-guard-runner/contracts/J12.js +66 -30
  73. package/dist/services/capability-guard-runner/contracts/J13.d.ts +11 -0
  74. package/dist/services/capability-guard-runner/contracts/J13.js +62 -40
  75. package/dist/services/capability-guard-runner/contracts/J14.d.ts +11 -0
  76. package/dist/services/capability-guard-runner/contracts/J14.js +60 -31
  77. package/dist/services/capability-guard-runner/contracts/J15.d.ts +11 -0
  78. package/dist/services/capability-guard-runner/contracts/J15.js +70 -35
  79. package/dist/services/capability-guard-runner/contracts/_shared.d.ts +24 -0
  80. package/dist/services/capability-guard-runner/contracts/_shared.js +67 -0
  81. package/dist/services/capability-guard-runner/registry.d.ts +5 -0
  82. package/dist/services/capability-guard-runner/registry.js +140 -0
  83. package/dist/services/capability-guard-runner/runner.d.ts +26 -0
  84. package/dist/services/capability-guard-runner/runner.js +63 -6
  85. package/dist/services/code/auto-compact-lifecycle.d.ts +75 -0
  86. package/dist/services/code/auto-compact-lifecycle.js +65 -16
  87. package/dist/services/code/auto-compact-modes.d.ts +13 -2
  88. package/dist/services/code/auto-compact-modes.js +20 -4
  89. package/dist/services/code/auto-compact-orchestrator.js +119 -19
  90. package/dist/services/code/compact-event-settle.d.ts +20 -8
  91. package/dist/services/code/compact-event-settle.js +21 -0
  92. package/dist/services/code/post-compact-detector.js +20 -11
  93. package/dist/services/code/step-08-gate.js +21 -6
  94. package/dist/services/compact-statusline/compact-statusline-service.js +56 -22
  95. package/dist/services/config/config-safety.js +11 -9
  96. package/dist/services/context/auto-compact-types.d.ts +20 -2
  97. package/dist/services/feedback/feedback-promotion-service.d.ts +137 -14
  98. package/dist/services/feedback/feedback-promotion-service.js +341 -20
  99. package/dist/services/feedback/promotion-artifact-evidence.d.ts +69 -0
  100. package/dist/services/feedback/promotion-artifact-evidence.js +332 -0
  101. package/dist/services/final-review/pre-post-diff.js +10 -2
  102. package/dist/services/job/job-progress-store.js +18 -3
  103. package/dist/services/observability/jsonl-store.d.ts +19 -0
  104. package/dist/services/observability/jsonl-store.js +27 -2
  105. package/dist/services/observability/observability-service.d.ts +11 -4
  106. package/dist/services/observability/observability-service.js +16 -3
  107. package/dist/services/prd/handoff-service.js +43 -0
  108. package/dist/services/qa/qa-business-review-state.js +19 -5
  109. package/dist/services/sc/sc-service.d.ts +8 -0
  110. package/dist/services/sc/sc-service.js +8 -1
  111. package/dist/services/scan/api-diff-types.js +20 -2
  112. package/dist/services/security/safe-settings-path.js +19 -1
  113. package/dist/services/session/getSessionDir.d.ts +33 -0
  114. package/dist/services/session/getSessionDir.js +60 -0
  115. package/dist/services/skill/skill-search-service.d.ts +3 -3
  116. package/dist/services/slice/slice-review-state.js +19 -4
  117. package/dist/services/standards/loop-engineering-lint.d.ts +1 -1
  118. package/dist/services/standards/loop-engineering-lint.js +6 -0
  119. package/dist/services/web/daemon-registry.js +27 -2
  120. package/dist/services/workflow/pipeline-verify-gate-support.js +10 -11
  121. package/dist/services/workflow/pipeline-verify-service.d.ts +1 -1
  122. package/dist/services/workflow/pipeline-verify-service.js +23 -10
  123. package/dist/services/workflow/pipeline-verify-types.d.ts +5 -3
  124. package/dist/services/workspace/claude-settings-template.d.ts +53 -37
  125. package/dist/services/workspace/claude-settings-template.js +105 -83
  126. package/dist/services/workspace/generated-artifacts-stamp.d.ts +119 -0
  127. package/dist/services/workspace/generated-artifacts-stamp.js +167 -0
  128. package/dist/services/workspace/workspace-claude-settings-materializer.d.ts +8 -0
  129. package/dist/services/workspace/workspace-claude-settings-materializer.js +38 -3
  130. package/dist/services/workspace/workspace-service.js +11 -1
  131. package/dist/shared/fs-utils.d.ts +26 -0
  132. package/dist/shared/fs-utils.js +35 -0
  133. package/dist/shared/runtime-root.d.ts +73 -0
  134. package/dist/shared/runtime-root.js +77 -0
  135. package/package.json +9 -7
  136. package/scripts/copy-templates.mjs +0 -12
  137. package/scripts/install-skills.mjs +154 -53
  138. package/skills/bee/peaks-qa/SKILL.md +0 -1
  139. package/skills/bee/peaks-rd/SKILL.md +0 -1
  140. package/skills/peaks-code/SKILL.md +12 -10
  141. package/skills/peaks-code/references/periodic-checkpoint.md +2 -2
  142. package/skills/peaks-code/references/runbook.md +3 -0
  143. package/skills/peaks-code/references/session-overload-signal-index.md +4 -2
  144. package/skills/peaks-code/references/startup-sequence.md +2 -2
  145. package/skills/peaks-code/references/step-0-8-gate.md +1 -1
  146. package/skills/peaks-code/references/sub-agent-dispatch.md +19 -19
  147. package/dist/cli/commands/context-builder-commands.d.ts +0 -11
  148. package/dist/cli/commands/context-builder-commands.js +0 -85
  149. package/dist/services/hooks/write-gate.js +0 -111
  150. package/skills/bee/peaks-prd/references/command-migration.md +0 -3
  151. package/skills/bee/peaks-qa/references/command-migration.md +0 -3
  152. package/skills/bee/peaks-rd/references/command-migration.md +0 -3
  153. package/skills/bee/peaks-sc/references/command-migration.md +0 -3
  154. package/skills/bee/peaks-txt/references/command-migration.md +0 -3
  155. package/skills/bee/peaks-ui/references/command-migration.md +0 -3
  156. package/skills/peaks-code/references/command-migration.md +0 -3
@@ -0,0 +1,140 @@
1
+ // src/services/capability-audit-service/independent-checker.ts
2
+ //
3
+ // The live, credential-free audit scorer.
4
+ //
5
+ // WHY A DETERMINISTIC CHECKER AND NOT AN LLM CALL
6
+ // ----------------------------------------------
7
+ // `publish.yml` is a secretless OIDC trusted-publishing workflow: `id-token:
8
+ // write`, no npm token, and no LLM credential of any kind in the environment.
9
+ // An LLM scorer therefore cannot run in the gate that decides whether a
10
+ // release happens. Adding a long-lived API secret to a secretless pipeline to
11
+ // serve that gate would be a threat-model regression, and an LLM verdict is
12
+ // non-deterministic — the same commit could flip between runs. So the live
13
+ // scorer must be credential-free.
14
+ //
15
+ // WHY IT IS STILL "INDEPENDENT"
16
+ // -----------------------------
17
+ // Independence is a property of the information channel, not of the substrate
18
+ // (RL-5 constrains what the scorer READS: `scorer.reads: evaluation_package_only`
19
+ // — it never requires a model). The scorer this replaces was handed
20
+ // `{baselineJourneyId, guard}` and asked to re-state it; an LLM given that same
21
+ // payload would be exactly as much a rubber stamp. The disease was the payload.
22
+ //
23
+ // This checker answers a question no guard contract can answer, from inputs no
24
+ // guard reads:
25
+ // - a guard sees only ITSELF, so it cannot report that the observation set was
26
+ // silently narrowed, or that a frozen row is armed by no contract at all;
27
+ // - this checker sees the whole frozen claim set AND the whole registry.
28
+ // It reads only the evaluation package — no author reasoning, no session id, no
29
+ // self-praise framing — so RL-5's exclusions hold by construction.
30
+ //
31
+ // WHAT IT DOES NOT COVER (stated, not hidden)
32
+ // -------------------------------------------
33
+ // It does not read `forbiddenChanges` prose, and it cannot judge behaviour
34
+ // beyond what the 15 guard contracts already exercise. Its claim is narrower
35
+ // than "the 15 journeys are intact"; `coverage` in the result discloses exactly
36
+ // how narrow, so `consistent` is never read as more than it is.
37
+ import { existsSync } from 'node:fs';
38
+ import { join } from 'node:path';
39
+ import { P0_JOURNEY_IDS } from '../capability-baseline/types.js';
40
+ function finding(code, journeyId, detail) {
41
+ return { code, journeyId, detail };
42
+ }
43
+ /**
44
+ * The observed journey set must be exactly the frozen P0 set. `runAllGuards`
45
+ * aggregates whatever contracts it was handed, so a registry that lost a
46
+ * journey reports a clean `pass: 14 / fail: 0` — a narrower check that looks
47
+ * exactly like a green one. Nothing in the guard results can say so; only the
48
+ * frozen set can.
49
+ */
50
+ function checkObservationSet(frozen, observed) {
51
+ const out = [];
52
+ const counts = new Map();
53
+ for (const j of observed)
54
+ counts.set(j, (counts.get(j) ?? 0) + 1);
55
+ for (const j of frozen) {
56
+ const n = counts.get(j) ?? 0;
57
+ if (n === 0)
58
+ out.push(finding('OBSERVATION_INCOMPLETE', j, `${j} is in the frozen baseline but no guard result was observed for it`));
59
+ else if (n > 1)
60
+ out.push(finding('OBSERVATION_INCOMPLETE', j, `${j} produced ${String(n)} guard results; the frozen baseline declares it once`));
61
+ }
62
+ for (const j of counts.keys()) {
63
+ if (!frozen.includes(j))
64
+ out.push(finding('OBSERVATION_INCOMPLETE', j, `${j} was observed but is not a frozen P0 journey`));
65
+ }
66
+ return out;
67
+ }
68
+ /** The frozen claim set itself must be the P0 set, with no duplicate rows. */
69
+ function checkFrozenRows(rows) {
70
+ const out = [];
71
+ const seen = new Map();
72
+ for (const r of rows)
73
+ seen.set(r.journeyId, (seen.get(r.journeyId) ?? 0) + 1);
74
+ for (const j of P0_JOURNEY_IDS) {
75
+ const n = seen.get(j) ?? 0;
76
+ if (n === 0)
77
+ out.push(finding('BASELINE_ROW_SET_INVALID', j, `frozen baseline has no row for ${j}`));
78
+ else if (n > 1)
79
+ out.push(finding('BASELINE_ROW_SET_INVALID', j, `frozen baseline declares ${j} ${String(n)} times`));
80
+ }
81
+ for (const j of seen.keys()) {
82
+ if (!P0_JOURNEY_IDS.includes(j))
83
+ out.push(finding('BASELINE_ROW_SET_INVALID', j, `frozen baseline declares ${j}, which is not a P0 journey`));
84
+ }
85
+ return out;
86
+ }
87
+ /**
88
+ * Every `sourceFiles` entry of every frozen row must still exist. The guard
89
+ * contracts check this too, but only through their own contract — so a
90
+ * contract rewritten to drop that probe takes the check with it. Reading the
91
+ * frozen text directly means the binding survives such a rewrite.
92
+ */
93
+ function checkSourceBindings(projectRoot, rows) {
94
+ const out = [];
95
+ for (const row of rows) {
96
+ for (const f of row.sourceFiles) {
97
+ if (!existsSync(join(projectRoot, f))) {
98
+ out.push(finding('SOURCE_FILE_MISSING', row.journeyId, `frozen sourceFiles entry "${f}" is not on disk`));
99
+ }
100
+ }
101
+ }
102
+ return out;
103
+ }
104
+ function countArmed(rows, contracts) {
105
+ let armed = 0;
106
+ for (const row of rows) {
107
+ for (const inv of row.invariants) {
108
+ if (contracts.some((c) => c.source.baselineRow === row.journeyId && c.source.invariant === inv))
109
+ armed += 1;
110
+ }
111
+ }
112
+ return armed;
113
+ }
114
+ /**
115
+ * Run the credential-free independent evaluation. The verdict is `drifted`
116
+ * whenever a concrete deviation is observed — this function has no path that
117
+ * returns `consistent` without having checked.
118
+ */
119
+ export function runIndependentCheck(input) {
120
+ const observed = input.guardResults.map((r) => r.journeyId);
121
+ const findings = [
122
+ ...checkFrozenRows(input.baselineRows),
123
+ ...checkObservationSet(input.baselineRows.map((r) => r.journeyId), observed),
124
+ ...checkSourceBindings(input.projectRoot, input.baselineRows)
125
+ ];
126
+ const coverage = {
127
+ observations: observed.length,
128
+ observationsExpected: P0_JOURNEY_IDS.length,
129
+ invariantsFrozen: input.baselineRows.reduce((n, r) => n + r.invariants.length, 0),
130
+ invariantsArmed: countArmed(input.baselineRows, input.contracts),
131
+ // Disclosed, not checked: free-text prohibitions cannot be judged
132
+ // deterministically without turning a keyword scan into a fake verdict.
133
+ forbiddenChangesUnverified: input.baselineRows.reduce((n, r) => n + r.forbiddenChanges.length, 0)
134
+ };
135
+ return {
136
+ verdict: findings.length === 0 ? 'consistent' : 'drifted',
137
+ findings,
138
+ coverage
139
+ };
140
+ }
@@ -1,3 +1,5 @@
1
1
  export { crossCheck } from './cross-check.js';
2
+ export { runIndependentCheck } from './independent-checker.js';
3
+ export type { IndependentCheckInput } from './independent-checker.js';
2
4
  export { isStale } from './staleness.js';
3
- export type { AuditVerdict, AuditEvidenceKind, AuditDimension, CrossCheck, CapabilityAuditResult } from './types.js';
5
+ export type { AuditVerdict, AuditEvidenceKind, AuditDimension, AuditFinding, AuditFindingCode, AuditCoverage, IndependentCheckResult, CrossCheck, CapabilityAuditResult } from './types.js';
@@ -1,2 +1,3 @@
1
1
  export { crossCheck } from './cross-check.js';
2
+ export { runIndependentCheck } from './independent-checker.js';
2
3
  export { isStale } from './staleness.js';
@@ -1,21 +1,25 @@
1
1
  import type { CapabilityAuditResult } from './types.js';
2
- import type { JourneyId } from '../capability-baseline/types.js';
3
- import type { GuardRunResult } from '../capability-guard-runner/types.js';
2
+ import type { CapabilityBaselineRow, JourneyId } from '../capability-baseline/types.js';
3
+ import type { GuardContract, GuardRunResult } from '../capability-guard-runner/types.js';
4
+ /**
5
+ * `stub` means the "independent" verdict came from a hard-coded response, not
6
+ * from a separate context. A stub is not an evaluation, so an audit that used
7
+ * one is marked `degraded` and can never report `consistent`.
8
+ *
9
+ * `live` runs the deterministic independent checker: a real separate-context
10
+ * evaluation that needs no credentials, which is why it is the only kind that
11
+ * can run inside the secretless OIDC publish gate.
12
+ */
13
+ export type AuditScorerMode = 'stub' | 'live';
4
14
  export interface RunAuditInput {
5
15
  readonly projectRoot: string;
6
16
  readonly sessionId: string;
7
17
  readonly journeyId: JourneyId;
8
- readonly llmRunner: {
9
- call(system: string, user: string, opts: {
10
- maxTokens: number;
11
- }): Promise<{
12
- output: string;
13
- tokens: {
14
- input: number;
15
- output: number;
16
- };
17
- }>;
18
- };
18
+ readonly scorerMode: AuditScorerMode;
19
+ /** The frozen claim set under audit. */
20
+ readonly baselineRows: ReadonlyArray<CapabilityBaselineRow>;
21
+ /** The arming witness: which frozen invariants some contract enforces. */
22
+ readonly contracts: ReadonlyArray<GuardContract>;
19
23
  readonly guardSummary: {
20
24
  readonly pass: number;
21
25
  readonly fail: number;
@@ -1,29 +1,87 @@
1
1
  import { mkdirSync, writeFileSync } from 'node:fs';
2
2
  import { join } from 'node:path';
3
3
  import { crossCheck } from './cross-check.js';
4
- const SYSTEM = 'You are an INDEPENDENT audit scorer. Compare the supplied capability baseline to the supplied current behavior summary. Output a single JSON object: {"verdict":"consistent" | "drifted" | "inconclusive"}. No prose.';
4
+ import { runIndependentCheck } from './independent-checker.js';
5
+ function scoreFor(status) {
6
+ return status === 'pass' ? 1 : status === 'fail' ? 0 : 0.5;
7
+ }
5
8
  export async function runAudit(input) {
6
- const userPayload = JSON.stringify({ baselineJourneyId: input.journeyId, guard: input.guardSummary });
7
- const r = await input.llmRunner.call(SYSTEM, userPayload, { maxTokens: 200 });
8
- const { verdict: independentVerdict } = JSON.parse(r.output);
9
+ const degraded = input.scorerMode === 'stub';
10
+ // A stub run performs no evaluation, so the checker is not run either — its
11
+ // result would be misread as an evaluation that happened.
12
+ const check = degraded
13
+ ? null
14
+ : runIndependentCheck({
15
+ projectRoot: input.projectRoot,
16
+ baselineRows: input.baselineRows,
17
+ contracts: input.contracts,
18
+ guardResults: input.guardSummary.results
19
+ });
9
20
  const xc = crossCheck({
10
21
  guardPass: input.guardSummary.pass,
11
22
  guardFail: input.guardSummary.fail,
12
- independentPass: independentVerdict === 'consistent' ? 1 : 0,
13
- independentFail: independentVerdict === 'drifted' ? 1 : 0,
23
+ // A degraded run has no independent verdict to compare; 0/0 keeps the
24
+ // cross-check shape without inventing one.
25
+ independentPass: check?.verdict === 'consistent' ? 1 : 0,
26
+ independentFail: check?.verdict === 'drifted' ? 1 : 0,
14
27
  karpathy: 'skipped'
15
28
  });
16
- let verdict = independentVerdict;
17
- if (xc.guardVsAudit === 'diverge')
29
+ // S1's rule is unchanged and load-bearing: a run that performed no separate
30
+ // evaluation can never be `consistent`. S11 adds the live branch. Every
31
+ // concrete deviation — a failed guard contract, or a finding from the
32
+ // independent checker — reports `drifted` instead of hiding behind
33
+ // `inconclusive`. So `inconclusive` is now reachable only when no evaluation
34
+ // ran at all, which is what it should mean.
35
+ let verdict = 'consistent';
36
+ if (degraded)
18
37
  verdict = 'inconclusive';
19
- const dimensions = [{
38
+ else if (input.guardSummary.fail > 0)
39
+ verdict = 'drifted';
40
+ else if ((check?.findings.length ?? 0) > 0)
41
+ verdict = 'drifted';
42
+ // One dimension per journey actually run, scored from the guard result —
43
+ // previously this was a single row whose score was derived from the stub.
44
+ const dimensions = input.guardSummary.results.map((g) => {
45
+ // When the contract fails, include the diff detail in the evidence summary
46
+ // so the gate step log (and any artifact) carries a real diagnostic
47
+ // instead of just "workflow-trace → fail". The summary is bounded so a
48
+ // runaway diff can't bloat every dimension; the contract itself is the
49
+ // authoritative source.
50
+ const detail = g.status === 'fail' && g.diff
51
+ ? ` | ${g.diff.reason}: ${g.diff.after}`.slice(0, 4000)
52
+ : '';
53
+ return {
54
+ journeyId: g.journeyId,
55
+ consistencyScore: scoreFor(g.status),
56
+ evidence: [{
57
+ kind: 'guard-run',
58
+ ref: `capability-guard-runner:${g.journeyId}`,
59
+ summary: `${g.contract} → ${g.status}${detail}`
60
+ }]
61
+ };
62
+ });
63
+ if (dimensions.length === 0) {
64
+ dimensions.push({
20
65
  journeyId: input.journeyId,
21
66
  consistencyScore: verdict === 'consistent' ? 1 : verdict === 'drifted' ? 0 : 0.5,
22
- evidence: [
23
- { kind: 'guard-run', ref: `capability-guard-runner:${input.guardSummary.total}`, summary: `${input.guardSummary.pass} pass / ${input.guardSummary.fail} fail` },
24
- { kind: 'independent-eval', ref: 'audit-llm-context', summary: `independent verdict: ${independentVerdict}` }
25
- ]
26
- }];
67
+ evidence: [{ kind: 'guard-run', ref: 'capability-guard-runner:0', summary: 'no contract results were supplied' }]
68
+ });
69
+ }
70
+ const independentRef = degraded ? 'audit-independent-checker:stub' : 'audit-independent-checker:deterministic';
71
+ const first = dimensions[0];
72
+ dimensions[0] = {
73
+ ...first,
74
+ evidence: [
75
+ ...first.evidence,
76
+ {
77
+ kind: 'independent-eval',
78
+ ref: independentRef,
79
+ summary: check === null
80
+ ? 'degraded: stub scorer (no independent context ran); the verdict was not derived from an evaluation'
81
+ : `independent verdict: ${check.verdict}; observations ${String(check.coverage.observations)}/${String(check.coverage.observationsExpected)}; invariants armed ${String(check.coverage.invariantsArmed)}/${String(check.coverage.invariantsFrozen)}; findings: ${check.findings.length === 0 ? 'none' : check.findings.map((f) => `${f.code}(${f.journeyId})`).join(',')}`
82
+ }
83
+ ]
84
+ };
27
85
  const auditId = `audit-${Date.now()}-${Math.random().toString(16).slice(2, 8)}`;
28
86
  const out = {
29
87
  auditId,
@@ -31,7 +89,10 @@ export async function runAudit(input) {
31
89
  verdict,
32
90
  dimensions,
33
91
  crossCheck: xc,
34
- requiresUserDecision: verdict === 'inconclusive'
92
+ requiresUserDecision: verdict === 'inconclusive',
93
+ degraded,
94
+ findings: check === null ? null : check.findings,
95
+ coverage: check === null ? null : check.coverage
35
96
  };
36
97
  const dir = join(input.projectRoot, '.peaks', '_runtime', input.sessionId, 'capability-audit');
37
98
  mkdirSync(dir, { recursive: true });
@@ -14,6 +14,41 @@ export interface CrossCheck {
14
14
  readonly guardVsAudit: 'agree' | 'diverge' | 'partial';
15
15
  readonly karpathyVsAudit: 'agree' | 'diverge' | 'partial';
16
16
  }
17
+ /**
18
+ * Why an independent verdict came out `drifted`. Each code names a concrete,
19
+ * inspectable deviation rather than a summary judgement.
20
+ */
21
+ export type AuditFindingCode =
22
+ /** The observed journey set is not the frozen P0 set. */
23
+ 'OBSERVATION_INCOMPLETE'
24
+ /** The frozen baseline's own row set is not the P0 set. */
25
+ | 'BASELINE_ROW_SET_INVALID'
26
+ /** A frozen `sourceFiles` entry no longer exists on disk. */
27
+ | 'SOURCE_FILE_MISSING';
28
+ export interface AuditFinding {
29
+ readonly code: AuditFindingCode;
30
+ readonly journeyId: JourneyId;
31
+ readonly detail: string;
32
+ }
33
+ /**
34
+ * How wide the audit's claim actually is. Reported alongside the verdict so
35
+ * `consistent` is never read as broader than it is: the check verifies the
36
+ * frozen row set, the observation set and the file bindings — it does not
37
+ * evaluate `forbiddenChanges` prose, and it judges no behaviour beyond what
38
+ * the guard contracts already exercise.
39
+ */
40
+ export interface AuditCoverage {
41
+ readonly observations: number;
42
+ readonly observationsExpected: number;
43
+ readonly invariantsFrozen: number;
44
+ readonly invariantsArmed: number;
45
+ readonly forbiddenChangesUnverified: number;
46
+ }
47
+ export interface IndependentCheckResult {
48
+ readonly verdict: 'consistent' | 'drifted';
49
+ readonly findings: ReadonlyArray<AuditFinding>;
50
+ readonly coverage: AuditCoverage;
51
+ }
17
52
  export interface CapabilityAuditResult {
18
53
  readonly auditId: string;
19
54
  readonly auditedAt: string;
@@ -21,4 +56,17 @@ export interface CapabilityAuditResult {
21
56
  readonly dimensions: ReadonlyArray<AuditDimension>;
22
57
  readonly crossCheck: CrossCheck;
23
58
  readonly requiresUserDecision: boolean;
59
+ /**
60
+ * True when no separate-context evaluation ran at all — i.e. the scorer was
61
+ * the stub, not the deterministic independent checker. A degraded audit can
62
+ * never be `consistent`.
63
+ */
64
+ readonly degraded: boolean;
65
+ /**
66
+ * The independent checker's findings, in the order it produced them. Empty
67
+ * on a `consistent` live run; `null` on a degraded run, where no check ran.
68
+ */
69
+ readonly findings: ReadonlyArray<AuditFinding> | null;
70
+ /** How wide this audit's claim is; `null` on a degraded run. */
71
+ readonly coverage: AuditCoverage | null;
24
72
  }
@@ -1,5 +1,6 @@
1
1
  import { execFileSync } from 'node:child_process';
2
2
  import { join } from 'node:path';
3
+ import { combineProbes, fail, missingSourceFiles, pass, probe, requireBaselineRow } from './_shared.js';
3
4
  const FIXTURES = [
4
5
  ['make', 'implement a CLI parser'],
5
6
  ['make', 'refactor the service'],
@@ -9,9 +10,11 @@ const FIXTURES = [
9
10
  ['run', 'execute a workflow']
10
11
  ];
11
12
  export async function runJ01Contract(ctx) {
13
+ const row = requireBaselineRow(ctx);
14
+ const missing = missingSourceFiles(ctx, row);
12
15
  const bin = process.env.PEAKS_BIN_OVERRIDE ?? join(ctx.projectRoot, 'bin', 'peaks.js');
13
- let allOk = true;
14
- let firstFailure = '';
16
+ const failures = [];
17
+ let routed = 0;
15
18
  for (const [command, input] of FIXTURES) {
16
19
  try {
17
20
  const stdout = execFileSync('node', [bin, command, input], {
@@ -21,29 +24,25 @@ export async function runJ01Contract(ctx) {
21
24
  }).toString('utf8');
22
25
  const env = JSON.parse(stdout);
23
26
  if (!env.ok) {
24
- allOk = false;
25
- firstFailure = `${command} ${input}`;
26
- break;
27
+ failures.push(`${command} ${input}: ok=false`);
28
+ }
29
+ else if (typeof env.data?.routedSkill === 'string' && env.data.routedSkill.length > 0) {
30
+ routed += 1;
27
31
  }
28
32
  }
29
33
  catch (e) {
30
- allOk = false;
31
- firstFailure = `${command} ${input}: ${e.message}`;
32
- break;
34
+ failures.push(`${command} ${input}: ${e.message.slice(0, 120)}`);
33
35
  }
34
36
  }
35
- return allOk
36
- ? {
37
- journeyId: 'J01',
38
- contract: 'envelope-arg-shapes',
39
- status: 'pass',
40
- artifactPath: 'tests/integration/super-command-routing.test.ts'
41
- }
42
- : {
43
- journeyId: 'J01',
44
- contract: 'envelope-arg-shapes',
45
- status: 'fail',
46
- diff: { before: 'all 6 routing cases ok', after: firstFailure, reason: 'J01#1 broken: super-command routing NL path deviates from frozen baseline' },
47
- artifactPath: 'tests/integration/super-command-routing.test.ts'
48
- };
37
+ const result = combineProbes([
38
+ probe(missing.length === 0, `baseline sourceFiles present (${row.sourceFiles.length})`),
39
+ probe(failures.length === 0, `all ${String(FIXTURES.length)} NL routing cases return ok (failures: ${failures.join('; ') || 'none'})`),
40
+ // The invariant is that the SYSTEM picks the skill: a bare ok envelope is
41
+ // not enough, the answer must name the skill it chose.
42
+ probe(routed > 0, `the envelope names the routed skill (${String(routed)}/${String(FIXTURES.length)})`)
43
+ ]);
44
+ const artifact = row.sourceFiles[2] ?? 'tests/integration/super-command-routing.test.ts';
45
+ if (result.ok)
46
+ return pass(ctx, artifact);
47
+ return fail(ctx, artifact, 'every NL fixture routes through the super-command surface and the envelope names the chosen skill', result.detail, 'J01 invariant broken: super-command routing deviates from the frozen baseline');
49
48
  }
@@ -1,2 +1,2 @@
1
1
  import type { GuardContext, GuardRunResult } from '../types.js';
2
- export declare function runJ02Contract(ctx: GuardContext, projectRoot?: string): Promise<GuardRunResult>;
2
+ export declare function runJ02Contract(ctx: GuardContext): Promise<GuardRunResult>;
@@ -1,34 +1,120 @@
1
1
  import { execFileSync } from 'node:child_process';
2
- import { mkdtempSync } from 'node:fs';
2
+ import { mkdtempSync, rmSync } from 'node:fs';
3
3
  import { tmpdir } from 'node:os';
4
- import { join } from 'node:path';
4
+ import { join, resolve } from 'node:path';
5
+ import { combineProbes, fail, missingSourceFiles, pass, probe, requireBaselineRow } from './_shared.js';
5
6
  const STATES = ['spec-locked', 'implemented', 'qa-handoff', 'handed-off'];
6
- export async function runJ02Contract(ctx, projectRoot = ctx.projectRoot) {
7
- const bin = join(ctx.projectRoot, 'bin', 'peaks.js');
7
+ // Per-child timeout. J02 runs six peaks CLIs in a fresh cwd; on cold CI
8
+ // runners the first invocation pays tsx startup before the resolved file is
9
+ // hot, and any single child that hangs would block the whole guard run until
10
+ // the default node timeout (forever). 60s is comfortable on warm hosts and
11
+ // tight enough that a genuine hang surfaces in the gate step within the
12
+ // publish workflow's per-step budget.
13
+ const CHILD_TIMEOUT_MS = 60_000;
14
+ export async function runJ02Contract(ctx) {
15
+ const row = requireBaselineRow(ctx);
16
+ const missing = missingSourceFiles(ctx, row);
17
+ // Must be ABSOLUTE: the child runs with `cwd: tmp`, so a relative
18
+ // `bin/peaks.js` would be resolved against the temp workspace.
19
+ const bin = resolve(ctx.projectRoot, 'bin', 'peaks.js');
8
20
  const tmp = mkdtempSync(join(tmpdir(), 'cbl-J02-'));
9
- const ws = execFileSync('node', [bin, 'workspace', 'init', '--project', tmp, '--json'], { cwd: tmp, windowsHide: true }).toString('utf8');
10
- const { data: { sessionId } } = JSON.parse(ws);
11
- const rid = '2026-08-03-j02-fixture';
12
- const initOut = execFileSync('node', [bin, 'request', 'init', '--role', 'rd', '--id', rid, '--project', tmp, '--session-id', sessionId, '--apply', '--json'], { cwd: tmp, windowsHide: true }).toString('utf8');
13
- const initEnv = JSON.parse(initOut);
14
- // `request init` writes the file as `NNN-<id-slug>.md`. The transition CLI accepts
15
- // the file's basename (without .md) as the requestId. Derive it from data.path.
16
- const baseName = initEnv.data.path.split(/[\\/]/).pop() ?? '';
17
- const requestId = baseName.replace(/\.md$/i, '');
18
- let last = '';
19
- for (const s of STATES) {
20
- const out = execFileSync('node', [bin, 'request', 'transition', requestId, '--role', 'rd', '--state', s,
21
- '--project', tmp, '--session-id', sessionId, '--confirm', '--allow-incomplete',
22
- '--reason', 'J02 contract fixture', '--json'], { cwd: tmp, windowsHide: true }).toString('utf8');
23
- const env = JSON.parse(out);
24
- last = env.data.state;
25
- }
26
- const ok = last === 'handed-off';
27
- return {
28
- journeyId: 'J02',
29
- contract: 'workflow-trace',
30
- status: ok ? 'pass' : 'fail',
31
- ...(ok ? {} : { diff: { before: 'handed-off', after: last, reason: 'J02#1 broken: RD state machine no longer reaches handed-off' } }),
32
- artifactPath: 'tests/integration/business-capability-e2e.test.ts'
21
+ // Each peaks-CLI invocation in this contract is wrapped so a child that
22
+ // hangs (or a child that produces no stdout) cannot silently pin the gate
23
+ // step until CI's job-level timeout fires. `execFileSync` already throws on
24
+ // non-zero exit; we additionally enforce timeout and surface the actual
25
+ // stderr in the error so a future failure can be diagnosed without re-running
26
+ // with a debugger.
27
+ const run = (args) => {
28
+ try {
29
+ const stdout = execFileSync('node', [bin, ...args], {
30
+ cwd: tmp,
31
+ windowsHide: true,
32
+ timeout: CHILD_TIMEOUT_MS,
33
+ encoding: 'utf8',
34
+ // The contract spawns peaks CLIs in a temp workspace; on a CI runner
35
+ // there is no IDE context, so peaks would refuse every `request.*`
36
+ // command with CALLER_ID_INVALID. Provide a deterministic caller id
37
+ // tied to the contract name. The contract is the only producer of
38
+ // this artifact, so the synthetic id cannot collide with anything
39
+ // real a developer is working on.
40
+ env: { ...process.env, PEAKS_CALLER_ID: `guard-J02-${ctx.sessionId}` }
41
+ });
42
+ return { stdout, stderr: '' };
43
+ }
44
+ catch (e) {
45
+ const err = e;
46
+ const stderr = typeof err.stderr === 'string'
47
+ ? err.stderr
48
+ : Buffer.isBuffer(err.stderr) ? err.stderr.toString('utf8') : '';
49
+ const stdout = typeof err.stdout === 'string'
50
+ ? err.stdout
51
+ : Buffer.isBuffer(err.stdout) ? err.stdout.toString('utf8') : '';
52
+ // Preserve the original error type/message but attach stderr so the
53
+ // outer try-catch's `e.message.slice(0, 160)` sees something useful.
54
+ // Truncate stdout aggressively to keep the audit envelope bounded, but
55
+ // pick the HEAD and TAIL of the buffer so the leading envelope header
56
+ // and the trailing error are both visible.
57
+ const stdoutHead = stdout.slice(0, 600);
58
+ const stdoutTail = stdout.length > 1200 ? stdout.slice(-400) : '';
59
+ const stdoutPart = stdoutTail
60
+ ? `${stdoutHead}...<truncated ${stdout.length - 1000}B>...${stdoutTail}`
61
+ : stdoutHead;
62
+ const wrapped = new Error(`${err.message} | stderr=${stderr.slice(0, 400)} | stdout=${stdoutPart}`);
63
+ wrapped.stdout = stdout;
64
+ throw wrapped;
65
+ }
33
66
  };
67
+ try {
68
+ const ws = run(['workspace', 'init', '--project', tmp, '--json']);
69
+ const { data: { sessionId } } = JSON.parse(ws.stdout);
70
+ const rid = '2026-08-03-j02-fixture';
71
+ const initOut = run(['request', 'init', '--role', 'rd', '--id', rid, '--project', tmp, '--session-id', sessionId, '--apply', '--json']);
72
+ const initEnv = JSON.parse(initOut.stdout);
73
+ // `request init` writes the file as `NNN-<id-slug>.md`. The transition CLI accepts
74
+ // the file's basename (without .md) as the requestId. Derive it from data.path.
75
+ const baseName = initEnv.data.path.split(/[\\/]/).pop() ?? '';
76
+ const requestId = baseName.replace(/\.md$/i, '');
77
+ const transition = (state, extra) => run([
78
+ 'request', 'transition', requestId, '--role', 'rd', '--state', state,
79
+ '--project', tmp, '--session-id', sessionId, '--confirm',
80
+ '--reason', 'J02 contract fixture', ...extra, '--json'
81
+ ]).stdout;
82
+ // Hard-gate probe, run FIRST: from the freshly initialised state, jumping
83
+ // straight to the terminal state skips every intermediate gate and must not
84
+ // be accepted without the explicit incomplete-work escape hatch. Doing this
85
+ // before the legal walk is what makes it a skip (from `qa-handoff` the move
86
+ // to `handed-off` is legal and proves nothing).
87
+ let gateSkipRefused = false;
88
+ let skipError = 'not attempted';
89
+ try {
90
+ transition('handed-off', []);
91
+ skipError = 'transition was accepted';
92
+ }
93
+ catch (e) {
94
+ gateSkipRefused = true;
95
+ skipError = e.message.slice(0, 160);
96
+ }
97
+ let last = '';
98
+ try {
99
+ for (const s of STATES) {
100
+ const env = JSON.parse(transition(s, ['--allow-incomplete']));
101
+ last = env.data.state;
102
+ }
103
+ }
104
+ catch (e) {
105
+ last = `error: ${e.message.slice(0, 160)}`;
106
+ }
107
+ const result = combineProbes([
108
+ probe(missing.length === 0, `baseline sourceFiles present (${row.sourceFiles.length})`),
109
+ probe(gateSkipRefused, `an incomplete jump to handed-off is refused (${skipError})`),
110
+ probe(last === 'handed-off', `the RD state machine reaches handed-off (saw ${last})`)
111
+ ]);
112
+ const artifact = row.sourceFiles[3] ?? 'tests/integration/job-e2e.test.ts';
113
+ if (result.ok)
114
+ return pass(ctx, artifact);
115
+ return fail(ctx, artifact, 'every hard gate is enforced and the RD state machine still reaches handed-off', result.detail, 'J02 invariant broken: the RD state machine or its hard-gate enforcement changed');
116
+ }
117
+ finally {
118
+ rmSync(tmp, { recursive: true, force: true });
119
+ }
34
120
  }
@@ -1,2 +1,15 @@
1
1
  import type { GuardContext, GuardRunResult } from '../types.js';
2
+ /**
3
+ * Behavioural probe of the "no silent-catch / fake-green reintroduced"
4
+ * invariant.
5
+ *
6
+ * The previous version read `final-review-types.ts` and asserted it contained
7
+ * four dimension strings — a file whose own name is `final-review`, so the
8
+ * check restated its filename.
9
+ *
10
+ * Here the repository's own AST guard (`scripts/lint/silent-warning-detector.mjs`)
11
+ * is executed and its per-rule violation counts are compared against the frozen
12
+ * ceilings. A newly swallowed `catch` in `src/**` raises a count and reddens the
13
+ * journey.
14
+ */
2
15
  export declare function runJ03Contract(ctx: GuardContext): Promise<GuardRunResult>;