peaks-loop 4.0.48 → 4.0.50
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +44 -0
- package/README-en.md +1 -1
- package/README.md +1 -1
- package/dist/cli/commands/audit-commands.js +1 -0
- package/dist/cli/commands/baseline-commands.js +163 -25
- package/dist/cli/commands/compact-command.js +1 -3
- package/dist/cli/commands/core/skill-command.js +53 -4
- package/dist/cli/commands/core/standards-command.d.ts +24 -0
- package/dist/cli/commands/core/standards-command.js +74 -0
- package/dist/cli/commands/feedback-commands.d.ts +11 -7
- package/dist/cli/commands/feedback-commands.js +49 -17
- package/dist/cli/commands/final-review-commands.js +12 -0
- package/dist/cli/commands/hooks-commands.js +55 -38
- package/dist/cli/commands/loop-eval-commands.js +22 -6
- package/dist/cli/commands/share-commands.js +37 -11
- package/dist/cli/commands/slice-integrate-commands.js +17 -0
- package/dist/cli/commands/web-commands.js +8 -1
- package/dist/cli/commands/workflow-lifecycle-commands.d.ts +6 -0
- package/dist/cli/commands/workflow-lifecycle-commands.js +64 -3
- package/dist/services/adapter/adapter.d.ts +30 -0
- package/dist/services/adapter/auto-adapter.d.ts +13 -0
- package/dist/services/adapter/claude-adapter.js +12 -0
- package/dist/services/adapter/codex-adapter.d.ts +12 -0
- package/dist/services/adapter/codex-adapter.js +12 -0
- package/dist/services/adapter/copilot-adapter.d.ts +12 -0
- package/dist/services/adapter/copilot-adapter.js +12 -0
- package/dist/services/artifacts/artifact-prerequisites.js +10 -0
- package/dist/services/artifacts/request-artifact-service.js +59 -38
- package/dist/services/audit/backing-detector.d.ts +25 -7
- package/dist/services/audit/backing-detector.js +33 -17
- package/dist/services/audit/enforcer-liveness.d.ts +12 -0
- package/dist/services/audit/enforcer-liveness.js +100 -0
- package/dist/services/audit/enforcers/active-skill-resolver.js +14 -1
- package/dist/services/audit/enforcers/lint-catalog-governance.d.ts +23 -11
- package/dist/services/audit/enforcers/lint-catalog-governance.js +10 -14
- package/dist/services/audit/enforcers/lint-rd-handoff-coverage.d.ts +5 -15
- package/dist/services/audit/enforcers/lint-rd-handoff-coverage.js +94 -25
- package/dist/services/audit/enforcers/lint-style.d.ts +9 -1
- package/dist/services/audit/enforcers/lint-style.js +38 -2
- package/dist/services/audit/prose-ratio-calculator.d.ts +28 -17
- package/dist/services/audit/prose-ratio-calculator.js +25 -18
- package/dist/services/audit/red-line-catalog-p2-a.js +1 -1
- package/dist/services/audit/red-lines-service.js +51 -7
- package/dist/services/capability-audit-service/independent-checker.d.ts +15 -0
- package/dist/services/capability-audit-service/independent-checker.js +140 -0
- package/dist/services/capability-audit-service/index.d.ts +3 -1
- package/dist/services/capability-audit-service/index.js +1 -0
- package/dist/services/capability-audit-service/runner.d.ts +17 -13
- package/dist/services/capability-audit-service/runner.js +76 -15
- package/dist/services/capability-audit-service/types.d.ts +48 -0
- package/dist/services/capability-guard-runner/contracts/J01.js +21 -22
- package/dist/services/capability-guard-runner/contracts/J02.d.ts +1 -1
- package/dist/services/capability-guard-runner/contracts/J02.js +114 -28
- package/dist/services/capability-guard-runner/contracts/J03.d.ts +13 -0
- package/dist/services/capability-guard-runner/contracts/J03.js +72 -21
- package/dist/services/capability-guard-runner/contracts/J04.d.ts +6 -0
- package/dist/services/capability-guard-runner/contracts/J04.js +65 -32
- package/dist/services/capability-guard-runner/contracts/J05.js +118 -16
- package/dist/services/capability-guard-runner/contracts/J06.d.ts +14 -0
- package/dist/services/capability-guard-runner/contracts/J06.js +57 -39
- package/dist/services/capability-guard-runner/contracts/J07.d.ts +9 -0
- package/dist/services/capability-guard-runner/contracts/J07.js +76 -47
- package/dist/services/capability-guard-runner/contracts/J08.d.ts +11 -0
- package/dist/services/capability-guard-runner/contracts/J08.js +66 -39
- package/dist/services/capability-guard-runner/contracts/J09.d.ts +13 -0
- package/dist/services/capability-guard-runner/contracts/J09.js +95 -39
- package/dist/services/capability-guard-runner/contracts/J10.d.ts +12 -0
- package/dist/services/capability-guard-runner/contracts/J10.js +69 -35
- package/dist/services/capability-guard-runner/contracts/J11.d.ts +8 -0
- package/dist/services/capability-guard-runner/contracts/J11.js +73 -33
- package/dist/services/capability-guard-runner/contracts/J12.d.ts +12 -0
- package/dist/services/capability-guard-runner/contracts/J12.js +66 -30
- package/dist/services/capability-guard-runner/contracts/J13.d.ts +11 -0
- package/dist/services/capability-guard-runner/contracts/J13.js +62 -40
- package/dist/services/capability-guard-runner/contracts/J14.d.ts +11 -0
- package/dist/services/capability-guard-runner/contracts/J14.js +60 -31
- package/dist/services/capability-guard-runner/contracts/J15.d.ts +11 -0
- package/dist/services/capability-guard-runner/contracts/J15.js +70 -35
- package/dist/services/capability-guard-runner/contracts/_shared.d.ts +24 -0
- package/dist/services/capability-guard-runner/contracts/_shared.js +67 -0
- package/dist/services/capability-guard-runner/registry.d.ts +5 -0
- package/dist/services/capability-guard-runner/registry.js +140 -0
- package/dist/services/capability-guard-runner/runner.d.ts +26 -0
- package/dist/services/capability-guard-runner/runner.js +63 -6
- package/dist/services/code/auto-compact-lifecycle.d.ts +75 -0
- package/dist/services/code/auto-compact-lifecycle.js +65 -16
- package/dist/services/code/auto-compact-modes.d.ts +13 -2
- package/dist/services/code/auto-compact-modes.js +20 -4
- package/dist/services/code/auto-compact-orchestrator.js +119 -19
- package/dist/services/code/compact-event-settle.d.ts +20 -8
- package/dist/services/code/compact-event-settle.js +21 -0
- package/dist/services/code/post-compact-detector.js +20 -11
- package/dist/services/code/step-08-gate.js +21 -6
- package/dist/services/compact-statusline/compact-statusline-service.js +56 -22
- package/dist/services/config/config-safety.js +11 -9
- package/dist/services/context/auto-compact-types.d.ts +20 -2
- package/dist/services/feedback/feedback-promotion-service.d.ts +137 -14
- package/dist/services/feedback/feedback-promotion-service.js +341 -20
- package/dist/services/feedback/promotion-artifact-evidence.d.ts +69 -0
- package/dist/services/feedback/promotion-artifact-evidence.js +332 -0
- package/dist/services/final-review/pre-post-diff.js +10 -2
- package/dist/services/job/job-progress-store.js +18 -3
- package/dist/services/observability/jsonl-store.d.ts +19 -0
- package/dist/services/observability/jsonl-store.js +27 -2
- package/dist/services/observability/observability-service.d.ts +11 -4
- package/dist/services/observability/observability-service.js +16 -3
- package/dist/services/prd/handoff-service.js +43 -0
- package/dist/services/qa/qa-business-review-state.js +19 -5
- package/dist/services/sc/sc-service.d.ts +8 -0
- package/dist/services/sc/sc-service.js +8 -1
- package/dist/services/scan/api-diff-types.js +20 -2
- package/dist/services/security/safe-settings-path.js +19 -1
- package/dist/services/session/getSessionDir.d.ts +33 -0
- package/dist/services/session/getSessionDir.js +60 -0
- package/dist/services/skill/skill-search-service.d.ts +3 -3
- package/dist/services/slice/slice-review-state.js +19 -4
- package/dist/services/standards/loop-engineering-lint.d.ts +1 -1
- package/dist/services/standards/loop-engineering-lint.js +6 -0
- package/dist/services/web/daemon-registry.js +27 -2
- package/dist/services/workflow/pipeline-verify-gate-support.js +10 -11
- package/dist/services/workflow/pipeline-verify-service.d.ts +1 -1
- package/dist/services/workflow/pipeline-verify-service.js +23 -10
- package/dist/services/workflow/pipeline-verify-types.d.ts +5 -3
- package/dist/services/workspace/claude-settings-template.d.ts +53 -37
- package/dist/services/workspace/claude-settings-template.js +105 -83
- package/dist/services/workspace/generated-artifacts-stamp.d.ts +119 -0
- package/dist/services/workspace/generated-artifacts-stamp.js +167 -0
- package/dist/services/workspace/workspace-claude-settings-materializer.d.ts +8 -0
- package/dist/services/workspace/workspace-claude-settings-materializer.js +38 -3
- package/dist/services/workspace/workspace-service.js +11 -1
- package/dist/shared/fs-utils.d.ts +26 -0
- package/dist/shared/fs-utils.js +35 -0
- package/dist/shared/runtime-root.d.ts +73 -0
- package/dist/shared/runtime-root.js +77 -0
- package/package.json +9 -7
- package/scripts/copy-templates.mjs +0 -12
- package/scripts/install-skills.mjs +154 -53
- package/skills/bee/peaks-qa/SKILL.md +0 -1
- package/skills/bee/peaks-rd/SKILL.md +0 -1
- package/skills/peaks-code/SKILL.md +12 -10
- package/skills/peaks-code/references/periodic-checkpoint.md +2 -2
- package/skills/peaks-code/references/runbook.md +3 -0
- package/skills/peaks-code/references/session-overload-signal-index.md +4 -2
- package/skills/peaks-code/references/startup-sequence.md +2 -2
- package/skills/peaks-code/references/step-0-8-gate.md +1 -1
- package/skills/peaks-code/references/sub-agent-dispatch.md +19 -19
- package/dist/cli/commands/context-builder-commands.d.ts +0 -11
- package/dist/cli/commands/context-builder-commands.js +0 -85
- package/dist/services/hooks/write-gate.js +0 -111
- package/skills/bee/peaks-prd/references/command-migration.md +0 -3
- package/skills/bee/peaks-qa/references/command-migration.md +0 -3
- package/skills/bee/peaks-rd/references/command-migration.md +0 -3
- package/skills/bee/peaks-sc/references/command-migration.md +0 -3
- package/skills/bee/peaks-txt/references/command-migration.md +0 -3
- package/skills/bee/peaks-ui/references/command-migration.md +0 -3
- package/skills/peaks-code/references/command-migration.md +0 -3
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
// src/services/capability-audit-service/independent-checker.ts
|
|
2
|
+
//
|
|
3
|
+
// The live, credential-free audit scorer.
|
|
4
|
+
//
|
|
5
|
+
// WHY A DETERMINISTIC CHECKER AND NOT AN LLM CALL
|
|
6
|
+
// ----------------------------------------------
|
|
7
|
+
// `publish.yml` is a secretless OIDC trusted-publishing workflow: `id-token:
|
|
8
|
+
// write`, no npm token, and no LLM credential of any kind in the environment.
|
|
9
|
+
// An LLM scorer therefore cannot run in the gate that decides whether a
|
|
10
|
+
// release happens. Adding a long-lived API secret to a secretless pipeline to
|
|
11
|
+
// serve that gate would be a threat-model regression, and an LLM verdict is
|
|
12
|
+
// non-deterministic — the same commit could flip between runs. So the live
|
|
13
|
+
// scorer must be credential-free.
|
|
14
|
+
//
|
|
15
|
+
// WHY IT IS STILL "INDEPENDENT"
|
|
16
|
+
// -----------------------------
|
|
17
|
+
// Independence is a property of the information channel, not of the substrate
|
|
18
|
+
// (RL-5 constrains what the scorer READS: `scorer.reads: evaluation_package_only`
|
|
19
|
+
// — it never requires a model). The scorer this replaces was handed
|
|
20
|
+
// `{baselineJourneyId, guard}` and asked to re-state it; an LLM given that same
|
|
21
|
+
// payload would be exactly as much a rubber stamp. The disease was the payload.
|
|
22
|
+
//
|
|
23
|
+
// This checker answers a question no guard contract can answer, from inputs no
|
|
24
|
+
// guard reads:
|
|
25
|
+
// - a guard sees only ITSELF, so it cannot report that the observation set was
|
|
26
|
+
// silently narrowed, or that a frozen row is armed by no contract at all;
|
|
27
|
+
// - this checker sees the whole frozen claim set AND the whole registry.
|
|
28
|
+
// It reads only the evaluation package — no author reasoning, no session id, no
|
|
29
|
+
// self-praise framing — so RL-5's exclusions hold by construction.
|
|
30
|
+
//
|
|
31
|
+
// WHAT IT DOES NOT COVER (stated, not hidden)
|
|
32
|
+
// -------------------------------------------
|
|
33
|
+
// It does not read `forbiddenChanges` prose, and it cannot judge behaviour
|
|
34
|
+
// beyond what the 15 guard contracts already exercise. Its claim is narrower
|
|
35
|
+
// than "the 15 journeys are intact"; `coverage` in the result discloses exactly
|
|
36
|
+
// how narrow, so `consistent` is never read as more than it is.
|
|
37
|
+
import { existsSync } from 'node:fs';
|
|
38
|
+
import { join } from 'node:path';
|
|
39
|
+
import { P0_JOURNEY_IDS } from '../capability-baseline/types.js';
|
|
40
|
+
function finding(code, journeyId, detail) {
|
|
41
|
+
return { code, journeyId, detail };
|
|
42
|
+
}
|
|
43
|
+
/**
|
|
44
|
+
* The observed journey set must be exactly the frozen P0 set. `runAllGuards`
|
|
45
|
+
* aggregates whatever contracts it was handed, so a registry that lost a
|
|
46
|
+
* journey reports a clean `pass: 14 / fail: 0` — a narrower check that looks
|
|
47
|
+
* exactly like a green one. Nothing in the guard results can say so; only the
|
|
48
|
+
* frozen set can.
|
|
49
|
+
*/
|
|
50
|
+
function checkObservationSet(frozen, observed) {
|
|
51
|
+
const out = [];
|
|
52
|
+
const counts = new Map();
|
|
53
|
+
for (const j of observed)
|
|
54
|
+
counts.set(j, (counts.get(j) ?? 0) + 1);
|
|
55
|
+
for (const j of frozen) {
|
|
56
|
+
const n = counts.get(j) ?? 0;
|
|
57
|
+
if (n === 0)
|
|
58
|
+
out.push(finding('OBSERVATION_INCOMPLETE', j, `${j} is in the frozen baseline but no guard result was observed for it`));
|
|
59
|
+
else if (n > 1)
|
|
60
|
+
out.push(finding('OBSERVATION_INCOMPLETE', j, `${j} produced ${String(n)} guard results; the frozen baseline declares it once`));
|
|
61
|
+
}
|
|
62
|
+
for (const j of counts.keys()) {
|
|
63
|
+
if (!frozen.includes(j))
|
|
64
|
+
out.push(finding('OBSERVATION_INCOMPLETE', j, `${j} was observed but is not a frozen P0 journey`));
|
|
65
|
+
}
|
|
66
|
+
return out;
|
|
67
|
+
}
|
|
68
|
+
/** The frozen claim set itself must be the P0 set, with no duplicate rows. */
|
|
69
|
+
function checkFrozenRows(rows) {
|
|
70
|
+
const out = [];
|
|
71
|
+
const seen = new Map();
|
|
72
|
+
for (const r of rows)
|
|
73
|
+
seen.set(r.journeyId, (seen.get(r.journeyId) ?? 0) + 1);
|
|
74
|
+
for (const j of P0_JOURNEY_IDS) {
|
|
75
|
+
const n = seen.get(j) ?? 0;
|
|
76
|
+
if (n === 0)
|
|
77
|
+
out.push(finding('BASELINE_ROW_SET_INVALID', j, `frozen baseline has no row for ${j}`));
|
|
78
|
+
else if (n > 1)
|
|
79
|
+
out.push(finding('BASELINE_ROW_SET_INVALID', j, `frozen baseline declares ${j} ${String(n)} times`));
|
|
80
|
+
}
|
|
81
|
+
for (const j of seen.keys()) {
|
|
82
|
+
if (!P0_JOURNEY_IDS.includes(j))
|
|
83
|
+
out.push(finding('BASELINE_ROW_SET_INVALID', j, `frozen baseline declares ${j}, which is not a P0 journey`));
|
|
84
|
+
}
|
|
85
|
+
return out;
|
|
86
|
+
}
|
|
87
|
+
/**
|
|
88
|
+
* Every `sourceFiles` entry of every frozen row must still exist. The guard
|
|
89
|
+
* contracts check this too, but only through their own contract — so a
|
|
90
|
+
* contract rewritten to drop that probe takes the check with it. Reading the
|
|
91
|
+
* frozen text directly means the binding survives such a rewrite.
|
|
92
|
+
*/
|
|
93
|
+
function checkSourceBindings(projectRoot, rows) {
|
|
94
|
+
const out = [];
|
|
95
|
+
for (const row of rows) {
|
|
96
|
+
for (const f of row.sourceFiles) {
|
|
97
|
+
if (!existsSync(join(projectRoot, f))) {
|
|
98
|
+
out.push(finding('SOURCE_FILE_MISSING', row.journeyId, `frozen sourceFiles entry "${f}" is not on disk`));
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
}
|
|
102
|
+
return out;
|
|
103
|
+
}
|
|
104
|
+
function countArmed(rows, contracts) {
|
|
105
|
+
let armed = 0;
|
|
106
|
+
for (const row of rows) {
|
|
107
|
+
for (const inv of row.invariants) {
|
|
108
|
+
if (contracts.some((c) => c.source.baselineRow === row.journeyId && c.source.invariant === inv))
|
|
109
|
+
armed += 1;
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
return armed;
|
|
113
|
+
}
|
|
114
|
+
/**
|
|
115
|
+
* Run the credential-free independent evaluation. The verdict is `drifted`
|
|
116
|
+
* whenever a concrete deviation is observed — this function has no path that
|
|
117
|
+
* returns `consistent` without having checked.
|
|
118
|
+
*/
|
|
119
|
+
export function runIndependentCheck(input) {
|
|
120
|
+
const observed = input.guardResults.map((r) => r.journeyId);
|
|
121
|
+
const findings = [
|
|
122
|
+
...checkFrozenRows(input.baselineRows),
|
|
123
|
+
...checkObservationSet(input.baselineRows.map((r) => r.journeyId), observed),
|
|
124
|
+
...checkSourceBindings(input.projectRoot, input.baselineRows)
|
|
125
|
+
];
|
|
126
|
+
const coverage = {
|
|
127
|
+
observations: observed.length,
|
|
128
|
+
observationsExpected: P0_JOURNEY_IDS.length,
|
|
129
|
+
invariantsFrozen: input.baselineRows.reduce((n, r) => n + r.invariants.length, 0),
|
|
130
|
+
invariantsArmed: countArmed(input.baselineRows, input.contracts),
|
|
131
|
+
// Disclosed, not checked: free-text prohibitions cannot be judged
|
|
132
|
+
// deterministically without turning a keyword scan into a fake verdict.
|
|
133
|
+
forbiddenChangesUnverified: input.baselineRows.reduce((n, r) => n + r.forbiddenChanges.length, 0)
|
|
134
|
+
};
|
|
135
|
+
return {
|
|
136
|
+
verdict: findings.length === 0 ? 'consistent' : 'drifted',
|
|
137
|
+
findings,
|
|
138
|
+
coverage
|
|
139
|
+
};
|
|
140
|
+
}
|
|
@@ -1,3 +1,5 @@
|
|
|
1
1
|
export { crossCheck } from './cross-check.js';
|
|
2
|
+
export { runIndependentCheck } from './independent-checker.js';
|
|
3
|
+
export type { IndependentCheckInput } from './independent-checker.js';
|
|
2
4
|
export { isStale } from './staleness.js';
|
|
3
|
-
export type { AuditVerdict, AuditEvidenceKind, AuditDimension, CrossCheck, CapabilityAuditResult } from './types.js';
|
|
5
|
+
export type { AuditVerdict, AuditEvidenceKind, AuditDimension, AuditFinding, AuditFindingCode, AuditCoverage, IndependentCheckResult, CrossCheck, CapabilityAuditResult } from './types.js';
|
|
@@ -1,21 +1,25 @@
|
|
|
1
1
|
import type { CapabilityAuditResult } from './types.js';
|
|
2
|
-
import type { JourneyId } from '../capability-baseline/types.js';
|
|
3
|
-
import type { GuardRunResult } from '../capability-guard-runner/types.js';
|
|
2
|
+
import type { CapabilityBaselineRow, JourneyId } from '../capability-baseline/types.js';
|
|
3
|
+
import type { GuardContract, GuardRunResult } from '../capability-guard-runner/types.js';
|
|
4
|
+
/**
|
|
5
|
+
* `stub` means the "independent" verdict came from a hard-coded response, not
|
|
6
|
+
* from a separate context. A stub is not an evaluation, so an audit that used
|
|
7
|
+
* one is marked `degraded` and can never report `consistent`.
|
|
8
|
+
*
|
|
9
|
+
* `live` runs the deterministic independent checker: a real separate-context
|
|
10
|
+
* evaluation that needs no credentials, which is why it is the only kind that
|
|
11
|
+
* can run inside the secretless OIDC publish gate.
|
|
12
|
+
*/
|
|
13
|
+
export type AuditScorerMode = 'stub' | 'live';
|
|
4
14
|
export interface RunAuditInput {
|
|
5
15
|
readonly projectRoot: string;
|
|
6
16
|
readonly sessionId: string;
|
|
7
17
|
readonly journeyId: JourneyId;
|
|
8
|
-
readonly
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
tokens: {
|
|
14
|
-
input: number;
|
|
15
|
-
output: number;
|
|
16
|
-
};
|
|
17
|
-
}>;
|
|
18
|
-
};
|
|
18
|
+
readonly scorerMode: AuditScorerMode;
|
|
19
|
+
/** The frozen claim set under audit. */
|
|
20
|
+
readonly baselineRows: ReadonlyArray<CapabilityBaselineRow>;
|
|
21
|
+
/** The arming witness: which frozen invariants some contract enforces. */
|
|
22
|
+
readonly contracts: ReadonlyArray<GuardContract>;
|
|
19
23
|
readonly guardSummary: {
|
|
20
24
|
readonly pass: number;
|
|
21
25
|
readonly fail: number;
|
|
@@ -1,29 +1,87 @@
|
|
|
1
1
|
import { mkdirSync, writeFileSync } from 'node:fs';
|
|
2
2
|
import { join } from 'node:path';
|
|
3
3
|
import { crossCheck } from './cross-check.js';
|
|
4
|
-
|
|
4
|
+
import { runIndependentCheck } from './independent-checker.js';
|
|
5
|
+
function scoreFor(status) {
|
|
6
|
+
return status === 'pass' ? 1 : status === 'fail' ? 0 : 0.5;
|
|
7
|
+
}
|
|
5
8
|
export async function runAudit(input) {
|
|
6
|
-
const
|
|
7
|
-
|
|
8
|
-
|
|
9
|
+
const degraded = input.scorerMode === 'stub';
|
|
10
|
+
// A stub run performs no evaluation, so the checker is not run either — its
|
|
11
|
+
// result would be misread as an evaluation that happened.
|
|
12
|
+
const check = degraded
|
|
13
|
+
? null
|
|
14
|
+
: runIndependentCheck({
|
|
15
|
+
projectRoot: input.projectRoot,
|
|
16
|
+
baselineRows: input.baselineRows,
|
|
17
|
+
contracts: input.contracts,
|
|
18
|
+
guardResults: input.guardSummary.results
|
|
19
|
+
});
|
|
9
20
|
const xc = crossCheck({
|
|
10
21
|
guardPass: input.guardSummary.pass,
|
|
11
22
|
guardFail: input.guardSummary.fail,
|
|
12
|
-
|
|
13
|
-
|
|
23
|
+
// A degraded run has no independent verdict to compare; 0/0 keeps the
|
|
24
|
+
// cross-check shape without inventing one.
|
|
25
|
+
independentPass: check?.verdict === 'consistent' ? 1 : 0,
|
|
26
|
+
independentFail: check?.verdict === 'drifted' ? 1 : 0,
|
|
14
27
|
karpathy: 'skipped'
|
|
15
28
|
});
|
|
16
|
-
|
|
17
|
-
|
|
29
|
+
// S1's rule is unchanged and load-bearing: a run that performed no separate
|
|
30
|
+
// evaluation can never be `consistent`. S11 adds the live branch. Every
|
|
31
|
+
// concrete deviation — a failed guard contract, or a finding from the
|
|
32
|
+
// independent checker — reports `drifted` instead of hiding behind
|
|
33
|
+
// `inconclusive`. So `inconclusive` is now reachable only when no evaluation
|
|
34
|
+
// ran at all, which is what it should mean.
|
|
35
|
+
let verdict = 'consistent';
|
|
36
|
+
if (degraded)
|
|
18
37
|
verdict = 'inconclusive';
|
|
19
|
-
|
|
38
|
+
else if (input.guardSummary.fail > 0)
|
|
39
|
+
verdict = 'drifted';
|
|
40
|
+
else if ((check?.findings.length ?? 0) > 0)
|
|
41
|
+
verdict = 'drifted';
|
|
42
|
+
// One dimension per journey actually run, scored from the guard result —
|
|
43
|
+
// previously this was a single row whose score was derived from the stub.
|
|
44
|
+
const dimensions = input.guardSummary.results.map((g) => {
|
|
45
|
+
// When the contract fails, include the diff detail in the evidence summary
|
|
46
|
+
// so the gate step log (and any artifact) carries a real diagnostic
|
|
47
|
+
// instead of just "workflow-trace → fail". The summary is bounded so a
|
|
48
|
+
// runaway diff can't bloat every dimension; the contract itself is the
|
|
49
|
+
// authoritative source.
|
|
50
|
+
const detail = g.status === 'fail' && g.diff
|
|
51
|
+
? ` | ${g.diff.reason}: ${g.diff.after}`.slice(0, 4000)
|
|
52
|
+
: '';
|
|
53
|
+
return {
|
|
54
|
+
journeyId: g.journeyId,
|
|
55
|
+
consistencyScore: scoreFor(g.status),
|
|
56
|
+
evidence: [{
|
|
57
|
+
kind: 'guard-run',
|
|
58
|
+
ref: `capability-guard-runner:${g.journeyId}`,
|
|
59
|
+
summary: `${g.contract} → ${g.status}${detail}`
|
|
60
|
+
}]
|
|
61
|
+
};
|
|
62
|
+
});
|
|
63
|
+
if (dimensions.length === 0) {
|
|
64
|
+
dimensions.push({
|
|
20
65
|
journeyId: input.journeyId,
|
|
21
66
|
consistencyScore: verdict === 'consistent' ? 1 : verdict === 'drifted' ? 0 : 0.5,
|
|
22
|
-
evidence: [
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
67
|
+
evidence: [{ kind: 'guard-run', ref: 'capability-guard-runner:0', summary: 'no contract results were supplied' }]
|
|
68
|
+
});
|
|
69
|
+
}
|
|
70
|
+
const independentRef = degraded ? 'audit-independent-checker:stub' : 'audit-independent-checker:deterministic';
|
|
71
|
+
const first = dimensions[0];
|
|
72
|
+
dimensions[0] = {
|
|
73
|
+
...first,
|
|
74
|
+
evidence: [
|
|
75
|
+
...first.evidence,
|
|
76
|
+
{
|
|
77
|
+
kind: 'independent-eval',
|
|
78
|
+
ref: independentRef,
|
|
79
|
+
summary: check === null
|
|
80
|
+
? 'degraded: stub scorer (no independent context ran); the verdict was not derived from an evaluation'
|
|
81
|
+
: `independent verdict: ${check.verdict}; observations ${String(check.coverage.observations)}/${String(check.coverage.observationsExpected)}; invariants armed ${String(check.coverage.invariantsArmed)}/${String(check.coverage.invariantsFrozen)}; findings: ${check.findings.length === 0 ? 'none' : check.findings.map((f) => `${f.code}(${f.journeyId})`).join(',')}`
|
|
82
|
+
}
|
|
83
|
+
]
|
|
84
|
+
};
|
|
27
85
|
const auditId = `audit-${Date.now()}-${Math.random().toString(16).slice(2, 8)}`;
|
|
28
86
|
const out = {
|
|
29
87
|
auditId,
|
|
@@ -31,7 +89,10 @@ export async function runAudit(input) {
|
|
|
31
89
|
verdict,
|
|
32
90
|
dimensions,
|
|
33
91
|
crossCheck: xc,
|
|
34
|
-
requiresUserDecision: verdict === 'inconclusive'
|
|
92
|
+
requiresUserDecision: verdict === 'inconclusive',
|
|
93
|
+
degraded,
|
|
94
|
+
findings: check === null ? null : check.findings,
|
|
95
|
+
coverage: check === null ? null : check.coverage
|
|
35
96
|
};
|
|
36
97
|
const dir = join(input.projectRoot, '.peaks', '_runtime', input.sessionId, 'capability-audit');
|
|
37
98
|
mkdirSync(dir, { recursive: true });
|
|
@@ -14,6 +14,41 @@ export interface CrossCheck {
|
|
|
14
14
|
readonly guardVsAudit: 'agree' | 'diverge' | 'partial';
|
|
15
15
|
readonly karpathyVsAudit: 'agree' | 'diverge' | 'partial';
|
|
16
16
|
}
|
|
17
|
+
/**
|
|
18
|
+
* Why an independent verdict came out `drifted`. Each code names a concrete,
|
|
19
|
+
* inspectable deviation rather than a summary judgement.
|
|
20
|
+
*/
|
|
21
|
+
export type AuditFindingCode =
|
|
22
|
+
/** The observed journey set is not the frozen P0 set. */
|
|
23
|
+
'OBSERVATION_INCOMPLETE'
|
|
24
|
+
/** The frozen baseline's own row set is not the P0 set. */
|
|
25
|
+
| 'BASELINE_ROW_SET_INVALID'
|
|
26
|
+
/** A frozen `sourceFiles` entry no longer exists on disk. */
|
|
27
|
+
| 'SOURCE_FILE_MISSING';
|
|
28
|
+
export interface AuditFinding {
|
|
29
|
+
readonly code: AuditFindingCode;
|
|
30
|
+
readonly journeyId: JourneyId;
|
|
31
|
+
readonly detail: string;
|
|
32
|
+
}
|
|
33
|
+
/**
|
|
34
|
+
* How wide the audit's claim actually is. Reported alongside the verdict so
|
|
35
|
+
* `consistent` is never read as broader than it is: the check verifies the
|
|
36
|
+
* frozen row set, the observation set and the file bindings — it does not
|
|
37
|
+
* evaluate `forbiddenChanges` prose, and it judges no behaviour beyond what
|
|
38
|
+
* the guard contracts already exercise.
|
|
39
|
+
*/
|
|
40
|
+
export interface AuditCoverage {
|
|
41
|
+
readonly observations: number;
|
|
42
|
+
readonly observationsExpected: number;
|
|
43
|
+
readonly invariantsFrozen: number;
|
|
44
|
+
readonly invariantsArmed: number;
|
|
45
|
+
readonly forbiddenChangesUnverified: number;
|
|
46
|
+
}
|
|
47
|
+
export interface IndependentCheckResult {
|
|
48
|
+
readonly verdict: 'consistent' | 'drifted';
|
|
49
|
+
readonly findings: ReadonlyArray<AuditFinding>;
|
|
50
|
+
readonly coverage: AuditCoverage;
|
|
51
|
+
}
|
|
17
52
|
export interface CapabilityAuditResult {
|
|
18
53
|
readonly auditId: string;
|
|
19
54
|
readonly auditedAt: string;
|
|
@@ -21,4 +56,17 @@ export interface CapabilityAuditResult {
|
|
|
21
56
|
readonly dimensions: ReadonlyArray<AuditDimension>;
|
|
22
57
|
readonly crossCheck: CrossCheck;
|
|
23
58
|
readonly requiresUserDecision: boolean;
|
|
59
|
+
/**
|
|
60
|
+
* True when no separate-context evaluation ran at all — i.e. the scorer was
|
|
61
|
+
* the stub, not the deterministic independent checker. A degraded audit can
|
|
62
|
+
* never be `consistent`.
|
|
63
|
+
*/
|
|
64
|
+
readonly degraded: boolean;
|
|
65
|
+
/**
|
|
66
|
+
* The independent checker's findings, in the order it produced them. Empty
|
|
67
|
+
* on a `consistent` live run; `null` on a degraded run, where no check ran.
|
|
68
|
+
*/
|
|
69
|
+
readonly findings: ReadonlyArray<AuditFinding> | null;
|
|
70
|
+
/** How wide this audit's claim is; `null` on a degraded run. */
|
|
71
|
+
readonly coverage: AuditCoverage | null;
|
|
24
72
|
}
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { execFileSync } from 'node:child_process';
|
|
2
2
|
import { join } from 'node:path';
|
|
3
|
+
import { combineProbes, fail, missingSourceFiles, pass, probe, requireBaselineRow } from './_shared.js';
|
|
3
4
|
const FIXTURES = [
|
|
4
5
|
['make', 'implement a CLI parser'],
|
|
5
6
|
['make', 'refactor the service'],
|
|
@@ -9,9 +10,11 @@ const FIXTURES = [
|
|
|
9
10
|
['run', 'execute a workflow']
|
|
10
11
|
];
|
|
11
12
|
export async function runJ01Contract(ctx) {
|
|
13
|
+
const row = requireBaselineRow(ctx);
|
|
14
|
+
const missing = missingSourceFiles(ctx, row);
|
|
12
15
|
const bin = process.env.PEAKS_BIN_OVERRIDE ?? join(ctx.projectRoot, 'bin', 'peaks.js');
|
|
13
|
-
|
|
14
|
-
let
|
|
16
|
+
const failures = [];
|
|
17
|
+
let routed = 0;
|
|
15
18
|
for (const [command, input] of FIXTURES) {
|
|
16
19
|
try {
|
|
17
20
|
const stdout = execFileSync('node', [bin, command, input], {
|
|
@@ -21,29 +24,25 @@ export async function runJ01Contract(ctx) {
|
|
|
21
24
|
}).toString('utf8');
|
|
22
25
|
const env = JSON.parse(stdout);
|
|
23
26
|
if (!env.ok) {
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
+
failures.push(`${command} ${input}: ok=false`);
|
|
28
|
+
}
|
|
29
|
+
else if (typeof env.data?.routedSkill === 'string' && env.data.routedSkill.length > 0) {
|
|
30
|
+
routed += 1;
|
|
27
31
|
}
|
|
28
32
|
}
|
|
29
33
|
catch (e) {
|
|
30
|
-
|
|
31
|
-
firstFailure = `${command} ${input}: ${e.message}`;
|
|
32
|
-
break;
|
|
34
|
+
failures.push(`${command} ${input}: ${e.message.slice(0, 120)}`);
|
|
33
35
|
}
|
|
34
36
|
}
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
diff: { before: 'all 6 routing cases ok', after: firstFailure, reason: 'J01#1 broken: super-command routing NL path deviates from frozen baseline' },
|
|
47
|
-
artifactPath: 'tests/integration/super-command-routing.test.ts'
|
|
48
|
-
};
|
|
37
|
+
const result = combineProbes([
|
|
38
|
+
probe(missing.length === 0, `baseline sourceFiles present (${row.sourceFiles.length})`),
|
|
39
|
+
probe(failures.length === 0, `all ${String(FIXTURES.length)} NL routing cases return ok (failures: ${failures.join('; ') || 'none'})`),
|
|
40
|
+
// The invariant is that the SYSTEM picks the skill: a bare ok envelope is
|
|
41
|
+
// not enough, the answer must name the skill it chose.
|
|
42
|
+
probe(routed > 0, `the envelope names the routed skill (${String(routed)}/${String(FIXTURES.length)})`)
|
|
43
|
+
]);
|
|
44
|
+
const artifact = row.sourceFiles[2] ?? 'tests/integration/super-command-routing.test.ts';
|
|
45
|
+
if (result.ok)
|
|
46
|
+
return pass(ctx, artifact);
|
|
47
|
+
return fail(ctx, artifact, 'every NL fixture routes through the super-command surface and the envelope names the chosen skill', result.detail, 'J01 invariant broken: super-command routing deviates from the frozen baseline');
|
|
49
48
|
}
|
|
@@ -1,2 +1,2 @@
|
|
|
1
1
|
import type { GuardContext, GuardRunResult } from '../types.js';
|
|
2
|
-
export declare function runJ02Contract(ctx: GuardContext
|
|
2
|
+
export declare function runJ02Contract(ctx: GuardContext): Promise<GuardRunResult>;
|
|
@@ -1,34 +1,120 @@
|
|
|
1
1
|
import { execFileSync } from 'node:child_process';
|
|
2
|
-
import { mkdtempSync } from 'node:fs';
|
|
2
|
+
import { mkdtempSync, rmSync } from 'node:fs';
|
|
3
3
|
import { tmpdir } from 'node:os';
|
|
4
|
-
import { join } from 'node:path';
|
|
4
|
+
import { join, resolve } from 'node:path';
|
|
5
|
+
import { combineProbes, fail, missingSourceFiles, pass, probe, requireBaselineRow } from './_shared.js';
|
|
5
6
|
const STATES = ['spec-locked', 'implemented', 'qa-handoff', 'handed-off'];
|
|
6
|
-
|
|
7
|
-
|
|
7
|
+
// Per-child timeout. J02 runs six peaks CLIs in a fresh cwd; on cold CI
|
|
8
|
+
// runners the first invocation pays tsx startup before the resolved file is
|
|
9
|
+
// hot, and any single child that hangs would block the whole guard run until
|
|
10
|
+
// the default node timeout (forever). 60s is comfortable on warm hosts and
|
|
11
|
+
// tight enough that a genuine hang surfaces in the gate step within the
|
|
12
|
+
// publish workflow's per-step budget.
|
|
13
|
+
const CHILD_TIMEOUT_MS = 60_000;
|
|
14
|
+
export async function runJ02Contract(ctx) {
|
|
15
|
+
const row = requireBaselineRow(ctx);
|
|
16
|
+
const missing = missingSourceFiles(ctx, row);
|
|
17
|
+
// Must be ABSOLUTE: the child runs with `cwd: tmp`, so a relative
|
|
18
|
+
// `bin/peaks.js` would be resolved against the temp workspace.
|
|
19
|
+
const bin = resolve(ctx.projectRoot, 'bin', 'peaks.js');
|
|
8
20
|
const tmp = mkdtempSync(join(tmpdir(), 'cbl-J02-'));
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
//
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
21
|
+
// Each peaks-CLI invocation in this contract is wrapped so a child that
|
|
22
|
+
// hangs (or a child that produces no stdout) cannot silently pin the gate
|
|
23
|
+
// step until CI's job-level timeout fires. `execFileSync` already throws on
|
|
24
|
+
// non-zero exit; we additionally enforce timeout and surface the actual
|
|
25
|
+
// stderr in the error so a future failure can be diagnosed without re-running
|
|
26
|
+
// with a debugger.
|
|
27
|
+
const run = (args) => {
|
|
28
|
+
try {
|
|
29
|
+
const stdout = execFileSync('node', [bin, ...args], {
|
|
30
|
+
cwd: tmp,
|
|
31
|
+
windowsHide: true,
|
|
32
|
+
timeout: CHILD_TIMEOUT_MS,
|
|
33
|
+
encoding: 'utf8',
|
|
34
|
+
// The contract spawns peaks CLIs in a temp workspace; on a CI runner
|
|
35
|
+
// there is no IDE context, so peaks would refuse every `request.*`
|
|
36
|
+
// command with CALLER_ID_INVALID. Provide a deterministic caller id
|
|
37
|
+
// tied to the contract name. The contract is the only producer of
|
|
38
|
+
// this artifact, so the synthetic id cannot collide with anything
|
|
39
|
+
// real a developer is working on.
|
|
40
|
+
env: { ...process.env, PEAKS_CALLER_ID: `guard-J02-${ctx.sessionId}` }
|
|
41
|
+
});
|
|
42
|
+
return { stdout, stderr: '' };
|
|
43
|
+
}
|
|
44
|
+
catch (e) {
|
|
45
|
+
const err = e;
|
|
46
|
+
const stderr = typeof err.stderr === 'string'
|
|
47
|
+
? err.stderr
|
|
48
|
+
: Buffer.isBuffer(err.stderr) ? err.stderr.toString('utf8') : '';
|
|
49
|
+
const stdout = typeof err.stdout === 'string'
|
|
50
|
+
? err.stdout
|
|
51
|
+
: Buffer.isBuffer(err.stdout) ? err.stdout.toString('utf8') : '';
|
|
52
|
+
// Preserve the original error type/message but attach stderr so the
|
|
53
|
+
// outer try-catch's `e.message.slice(0, 160)` sees something useful.
|
|
54
|
+
// Truncate stdout aggressively to keep the audit envelope bounded, but
|
|
55
|
+
// pick the HEAD and TAIL of the buffer so the leading envelope header
|
|
56
|
+
// and the trailing error are both visible.
|
|
57
|
+
const stdoutHead = stdout.slice(0, 600);
|
|
58
|
+
const stdoutTail = stdout.length > 1200 ? stdout.slice(-400) : '';
|
|
59
|
+
const stdoutPart = stdoutTail
|
|
60
|
+
? `${stdoutHead}...<truncated ${stdout.length - 1000}B>...${stdoutTail}`
|
|
61
|
+
: stdoutHead;
|
|
62
|
+
const wrapped = new Error(`${err.message} | stderr=${stderr.slice(0, 400)} | stdout=${stdoutPart}`);
|
|
63
|
+
wrapped.stdout = stdout;
|
|
64
|
+
throw wrapped;
|
|
65
|
+
}
|
|
33
66
|
};
|
|
67
|
+
try {
|
|
68
|
+
const ws = run(['workspace', 'init', '--project', tmp, '--json']);
|
|
69
|
+
const { data: { sessionId } } = JSON.parse(ws.stdout);
|
|
70
|
+
const rid = '2026-08-03-j02-fixture';
|
|
71
|
+
const initOut = run(['request', 'init', '--role', 'rd', '--id', rid, '--project', tmp, '--session-id', sessionId, '--apply', '--json']);
|
|
72
|
+
const initEnv = JSON.parse(initOut.stdout);
|
|
73
|
+
// `request init` writes the file as `NNN-<id-slug>.md`. The transition CLI accepts
|
|
74
|
+
// the file's basename (without .md) as the requestId. Derive it from data.path.
|
|
75
|
+
const baseName = initEnv.data.path.split(/[\\/]/).pop() ?? '';
|
|
76
|
+
const requestId = baseName.replace(/\.md$/i, '');
|
|
77
|
+
const transition = (state, extra) => run([
|
|
78
|
+
'request', 'transition', requestId, '--role', 'rd', '--state', state,
|
|
79
|
+
'--project', tmp, '--session-id', sessionId, '--confirm',
|
|
80
|
+
'--reason', 'J02 contract fixture', ...extra, '--json'
|
|
81
|
+
]).stdout;
|
|
82
|
+
// Hard-gate probe, run FIRST: from the freshly initialised state, jumping
|
|
83
|
+
// straight to the terminal state skips every intermediate gate and must not
|
|
84
|
+
// be accepted without the explicit incomplete-work escape hatch. Doing this
|
|
85
|
+
// before the legal walk is what makes it a skip (from `qa-handoff` the move
|
|
86
|
+
// to `handed-off` is legal and proves nothing).
|
|
87
|
+
let gateSkipRefused = false;
|
|
88
|
+
let skipError = 'not attempted';
|
|
89
|
+
try {
|
|
90
|
+
transition('handed-off', []);
|
|
91
|
+
skipError = 'transition was accepted';
|
|
92
|
+
}
|
|
93
|
+
catch (e) {
|
|
94
|
+
gateSkipRefused = true;
|
|
95
|
+
skipError = e.message.slice(0, 160);
|
|
96
|
+
}
|
|
97
|
+
let last = '';
|
|
98
|
+
try {
|
|
99
|
+
for (const s of STATES) {
|
|
100
|
+
const env = JSON.parse(transition(s, ['--allow-incomplete']));
|
|
101
|
+
last = env.data.state;
|
|
102
|
+
}
|
|
103
|
+
}
|
|
104
|
+
catch (e) {
|
|
105
|
+
last = `error: ${e.message.slice(0, 160)}`;
|
|
106
|
+
}
|
|
107
|
+
const result = combineProbes([
|
|
108
|
+
probe(missing.length === 0, `baseline sourceFiles present (${row.sourceFiles.length})`),
|
|
109
|
+
probe(gateSkipRefused, `an incomplete jump to handed-off is refused (${skipError})`),
|
|
110
|
+
probe(last === 'handed-off', `the RD state machine reaches handed-off (saw ${last})`)
|
|
111
|
+
]);
|
|
112
|
+
const artifact = row.sourceFiles[3] ?? 'tests/integration/job-e2e.test.ts';
|
|
113
|
+
if (result.ok)
|
|
114
|
+
return pass(ctx, artifact);
|
|
115
|
+
return fail(ctx, artifact, 'every hard gate is enforced and the RD state machine still reaches handed-off', result.detail, 'J02 invariant broken: the RD state machine or its hard-gate enforcement changed');
|
|
116
|
+
}
|
|
117
|
+
finally {
|
|
118
|
+
rmSync(tmp, { recursive: true, force: true });
|
|
119
|
+
}
|
|
34
120
|
}
|
|
@@ -1,2 +1,15 @@
|
|
|
1
1
|
import type { GuardContext, GuardRunResult } from '../types.js';
|
|
2
|
+
/**
|
|
3
|
+
* Behavioural probe of the "no silent-catch / fake-green reintroduced"
|
|
4
|
+
* invariant.
|
|
5
|
+
*
|
|
6
|
+
* The previous version read `final-review-types.ts` and asserted it contained
|
|
7
|
+
* four dimension strings — a file whose own name is `final-review`, so the
|
|
8
|
+
* check restated its filename.
|
|
9
|
+
*
|
|
10
|
+
* Here the repository's own AST guard (`scripts/lint/silent-warning-detector.mjs`)
|
|
11
|
+
* is executed and its per-rule violation counts are compared against the frozen
|
|
12
|
+
* ceilings. A newly swallowed `catch` in `src/**` raises a count and reddens the
|
|
13
|
+
* journey.
|
|
14
|
+
*/
|
|
2
15
|
export declare function runJ03Contract(ctx: GuardContext): Promise<GuardRunResult>;
|