peaks-loop 4.0.10 → 4.0.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +57 -0
- package/config/eslint/.peaks-rules.cjs +123 -0
- package/dist/cli/commands/code-commands.js +2 -0
- package/dist/cli/commands/code-orchestrator-can-do.d.ts +17 -0
- package/dist/cli/commands/code-orchestrator-can-do.js +75 -0
- package/dist/cli/commands/workspace/reconcile-command.js +5 -2
- package/dist/reporters/bdd-reporter.d.ts +36 -0
- package/dist/reporters/bdd-reporter.js +159 -0
- package/dist/services/audit/enforcers/active-skill-resolver.js +4 -1
- package/dist/services/code/orchestrator-can-do.d.ts +114 -0
- package/dist/services/code/orchestrator-can-do.js +205 -0
- package/dist/services/doctor/doctor-service/checks/skill-presence.js +1 -1
- package/dist/services/doctor/doctor-service/checks/workspace-layout.d.ts +5 -2
- package/dist/services/doctor/doctor-service/checks/workspace-layout.js +70 -9
- package/dist/services/doctor/doctor-service/types.d.ts +9 -0
- package/dist/services/hooks/presence-marker-detector.d.ts +7 -5
- package/dist/services/hooks/presence-marker-detector.js +45 -85
- package/dist/services/migration/v2-10-to-v2-11-service.js +1 -3
- package/dist/services/qa/bdd-test-style-verifier.d.ts +88 -0
- package/dist/services/qa/bdd-test-style-verifier.js +268 -0
- package/dist/services/sc/sc-service.d.ts +11 -7
- package/dist/services/sc/sc-service.js +29 -26
- package/dist/services/session/binding-store.d.ts +6 -6
- package/dist/services/skills/hooks-settings-service.d.ts +1 -1
- package/dist/services/skills/hooks-settings-service.js +1 -1
- package/dist/services/skills/presence-lease-service.js +1 -0
- package/dist/services/skills/skill-presence-service.d.ts +0 -1
- package/dist/services/skills/skill-presence-service.js +112 -102
- package/dist/services/skills/skill-statusline-service.js +73 -39
- package/docs/test-style-contract.md +135 -0
- package/package.json +6 -4
- package/skills/bee/peaks-qa/references/qa-sub-agent-dispatch.md +17 -1
- package/skills/bee/peaks-rd/references/rd-sub-agent-dispatch.md +21 -1
- package/skills/peaks-code/SKILL.md +6 -0
- package/skills/peaks-code/references/session-overload-signal-index.md +60 -0
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,62 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 4.0.12 — 2026-08-05 (5-slice optimization bundle)
|
|
4
|
+
|
|
5
|
+
**publish.yml strict tag gate** (slice 1, commit `f60f7597`):
|
|
6
|
+
- New `gate-strict-tag-format` step inserted after `Checkout` (publish.yml lines 104-130). Runs `git describe --tags --exact-match HEAD` and validates against `^v[0-9]+\.[0-9]+\.[0-9]+$`; rejects `v1` / `v4.0` / `v4.0.11-rc1` / `v4.0.11+sha` shapes with `::error title=strict-tag-format::`.
|
|
7
|
+
- Drift guard: `tests/unit/publish-tag-strict.test.ts` (4 cases).
|
|
8
|
+
|
|
9
|
+
**PreToolUse:Bash hook JSON output validation** (slice 2, commit `eb13e44c`):
|
|
10
|
+
- Regression fix for the 2026-07-27 root cause. `.claude/settings.json` line 14 + `src/services/skills/hooks-settings-service.ts:86` both now append ` --json` to the `peaks gate enforce` hook command.
|
|
11
|
+
- Drift guard: `tests/unit/hooks/gate-enforce-template-json-flag.test.ts` (3 cases).
|
|
12
|
+
- PRD mis-reference correction noted: the actual hook template owner is `hooks-settings-service.ts`, not `claude-settings-template.ts` (the latter carries `peaks code gate-step-08`, a different hook path).
|
|
13
|
+
|
|
14
|
+
**Session-overload signal index** (slice 3, commit `6ef12bde`):
|
|
15
|
+
- New `skills/peaks-code/references/session-overload-signal-index.md` (60 lines): 7-signal lookup table (prompt size / auto-compact zone / sub-agent dispatch guard / statusline compact bar / in-flight batch deferral / compact stalled / sub-agent heartbeat) with thresholds + files + LLM actions + decision flowchart.
|
|
16
|
+
- SKILL.md Step N+2 section gets a 1-line "See also" pointer (+141 bytes; under 30000 cap).
|
|
17
|
+
- Codifies red line: LLM MUST NOT re-ask user about cost/length/context (cite `references/job-loop.md:59` red line #2).
|
|
18
|
+
|
|
19
|
+
**active-skill.json → sid-scoped lease projection** (slice 4, 3 sub-slices):
|
|
20
|
+
- Slice 4-A (`dc350c2c`): `skill-presence-service.ts` write path removed; `presence-marker-detector.ts` deprecated-constant comments added.
|
|
21
|
+
- Slice 4-B (`7699f70f`): `skill-statusline-service.ts` `readPresenceReadOnly` rewritten to canonical-only (drops `legacyPresence: true`, enumerates `listPresenceLeases`, deletes `PRESENCE_FILE` / `PRESENCE_FILE_LEGACY`). Fixes the "new session statusline empty" issue (root cause: outerSessionId drift between session.json and the global single-slot file).
|
|
22
|
+
- Slice 4-C (`1b08a62c`): doctor / sc / migration / hooks / skills / reconcile-command all migrated to canonical lease reads. New `STALE_SINGLE_SLOT_FILES` workspace-layout check surfaces orphan `.peaks/_runtime/active-skill.json` as a "stale single-slot presence" diagnostic (not an error). Drift guard: `tests/unit/workspace/active-skill-json-cleanup.test.ts` (3 cases).
|
|
23
|
+
- Multi-session on one project is now safe (each session writes its own `presence-<callerId>.json` under `.peaks/_runtime/<sid>/presence-index/`; the deprecated single-slot global file no longer races).
|
|
24
|
+
|
|
25
|
+
**Orchestrator-can-do probe CLI** (slice 5, commit `52736c82`):
|
|
26
|
+
- New `peaks code orchestrator-can-do --slice-spec <text> --json` command (the 2026-08-05 lesson converted into a downstream-inheritable CLI). Evaluates 4 boundary questions (source code? sub-agent available? requires user decision? context sustainable?) and returns `{ canDoInSession, blockers, warnings, suggestions, contextRatio, subAgentAvailable }`.
|
|
27
|
+
- SKILL.md gains a Step 0.51 paragraph pointing at it (LLM must run this probe before deciding to push a slice to the next session).
|
|
28
|
+
- 17 unit tests cover all 4 boundary branches.
|
|
29
|
+
- Anti-fake-green pattern: any future orchestrator capability judgment MUST call this CLI rather than re-derive the 4-question matrix from intuition.
|
|
30
|
+
|
|
31
|
+
**Lockstep bump.** peaks-loop-shared `0.0.41 → 0.0.42` (CLI_VERSION re-stamped to 4.0.12); peaks-loop-mut `0.1.14 → 0.1.15`; peaks-loop-shared-channel `0.0.18 → 0.0.19`.
|
|
32
|
+
|
|
33
|
+
## 4.0.11 — 2026-08-05 (BDD test-style + statusline bugs)
|
|
34
|
+
|
|
35
|
+
**BDD given-when-then test style** (rid-2026-08-05-bdd-test-style, 5 slice / 19 commit):
|
|
36
|
+
- `scripts/migrate-to-bdd.mjs` — TS Compiler API-based AST migrator that rewrites every `it()` / `test()` / `describe()` to the given-when-then contract (it description with `when X` or `should Y` + 3-line `// given:` / `// when:` / `// then:` body comment). Idempotent.
|
|
37
|
+
- `src/services/qa/bdd-test-style-verifier.ts` — peaks-qa verification-time verifier that scans `git diff HEAD~1 -- '*.test.ts'` and rejects non-BDD slices (LLM-only enforcement, since callerId from `process.env.CLAUDE_CODE_SESSION_ID` cannot distinguish LLM vs human in Claude Code).
|
|
38
|
+
- `src/reporters/bdd-reporter.ts` — vitest custom reporter (flag-enabled via `--reporter ./src/reporters/bdd-reporter.ts`) emitting `Feature: <file>` / `Scenario: <describe>` / `Given|When|Then` document view.
|
|
39
|
+
- `skills/bee/peaks-rd/references/rd-sub-agent-dispatch.md` + `peaks-qa/references/qa-sub-agent-dispatch.md` — `## BDD Test Style Contract` / `## BDD Test Style Verification` soft-constraint sections added.
|
|
40
|
+
- `docs/test-style-contract.md` — LLM test-style guide included in npm `files` array (downstream opt-in).
|
|
41
|
+
- 26 test files migrated across 11 commits (one per top-level directory); 12 files already BDD-form (idempotent migrator skipped silently); 49 unit-test files total now in BDD shape.
|
|
42
|
+
- 49/49 unit tests behaviour-preserved (488 passed / 25 skipped / 0 introduced failures).
|
|
43
|
+
|
|
44
|
+
**Statusline bug fixes** (3 rid this release):
|
|
45
|
+
- `skill-statusline-service.ts` `readActiveLeaf`: stale `queued` dispatch entries no longer pollute statusline as in-flight leaves. `terminalStatuses` set extended.
|
|
46
|
+
- `presence-lease-service.ts` `setPresenceLease`: lease object now persists `mode` (was being silently dropped — regression from 4.0.8 Presence Lease Graph introduction). `[full-auto]` / `[assisted]` / `[swarm]` / `[strict]` tags now render.
|
|
47
|
+
- `audit/enforcers/active-skill-resolver.ts` legacy fall-back: per-caller `active-skill-*.json` legacy walk now reads and propagates `mode` (was hard-coded `mode: null`).
|
|
48
|
+
|
|
49
|
+
**Build-chain repair** (silences `npx tsc -p tsconfig.build.json` regression):
|
|
50
|
+
- `src/reporters/bdd-reporter.ts` no longer imports the un-exported `TestModule` / `TestCase` from `vitest/reporters`. Local minimal interfaces (`BddTestModuleLike` / `BddTestCaseLike`) match the runtime shape vitest passes to reporter hooks.
|
|
51
|
+
- Build now succeeds end-to-end; the `dist/` artifacts (which `bin/peaks.js` actually loads) reflect the source-level fixes.
|
|
52
|
+
|
|
53
|
+
**Cleanup tail** (5 rid carried over from b1 sweep):
|
|
54
|
+
- 63 request artifacts re-staged to terminal state (handed-off / verdict-issued / complete / sc-handoff → done).
|
|
55
|
+
- 8 OpenSpec proposal drift detected and routed through `peaks request transition`.
|
|
56
|
+
- One envelope-test-output log dropped from project root (was orphan inside `.gitignore:5 *.log` but never deleted).
|
|
57
|
+
|
|
58
|
+
**Lockstep bump.** peaks-loop-shared `0.0.40 → 0.0.41` (CLI_VERSION re-stamped to 4.0.11).
|
|
59
|
+
|
|
3
60
|
## 4.0.10 — 2026-08-04 (path-canonicalize + statusline-read-isolation)
|
|
4
61
|
|
|
5
62
|
**Windows statusline fixed.** `peaks-loop@4.0.9` always rendered `peaks empty` on Windows Git Bash. Root cause: the session-binding reader used strict `===` to compare `projectRoot`; the binding had been written with backslashes (`C:\Users\...`) but `peaks skill presence:set --project C:/Users/...` arrived with forward slashes, and Node treats them as distinct strings. The 4.0.8 fail-closed `PEAKS_SESSION_NOT_BOUND` gate then blocked the presence marker write, and the statusline never had a real skill to display.
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* peaks-loop ESLint rules bundle (npm-package exports)
|
|
3
|
+
*
|
|
4
|
+
* rid-2026-08-05-jsts-lint-bundle — LLM auto-fix loop trigger.
|
|
5
|
+
*
|
|
6
|
+
* This file is a JSON-shape glue config. It does NOT define custom
|
|
7
|
+
* rules. It composes upstream packages only:
|
|
8
|
+
*
|
|
9
|
+
* - eslint:recommended
|
|
10
|
+
* - plugin:@typescript-eslint/recommended-type-checked
|
|
11
|
+
* - plugin:import/recommended
|
|
12
|
+
* - plugin:import/typescript
|
|
13
|
+
*
|
|
14
|
+
* Framework-specific rules (eslint-plugin-react, eslint-plugin-vue,
|
|
15
|
+
* eslint-plugin-svelte, eslint-plugin-nestjs, etc.) are LAYER 3 and
|
|
16
|
+
* loaded dynamically by `peaks code lint` via `npx --package <pkg>
|
|
17
|
+
* -- eslint`. They are NOT installed in this package's devDependencies
|
|
18
|
+
* (sediment §二 G-lint-1 turn-5 red line).
|
|
19
|
+
*
|
|
20
|
+
* --fix / --write / prettier are FORBIDDEN at the peaks code lint
|
|
21
|
+
* wrapper entry; the wrapper is a read-only verifier, not a formatter
|
|
22
|
+
* (sediment §二 G-lint-2). The thresholds below are intentionally
|
|
23
|
+
* permissive (warn, not error) so peaks-loop 4.0.10 baseline can adopt
|
|
24
|
+
* the bundle without auto-failing.
|
|
25
|
+
*/
|
|
26
|
+
'use strict';
|
|
27
|
+
|
|
28
|
+
/** @type {import('eslint').Linter.Config} */
|
|
29
|
+
module.exports = {
|
|
30
|
+
root: false,
|
|
31
|
+
parser: '@typescript-eslint/parser',
|
|
32
|
+
parserOptions: {
|
|
33
|
+
ecmaVersion: 2022,
|
|
34
|
+
sourceType: 'module',
|
|
35
|
+
project: ['./tsconfig.json', './tsconfig.build.json'],
|
|
36
|
+
tsconfigRootDir: __dirname + '/..'
|
|
37
|
+
},
|
|
38
|
+
env: {
|
|
39
|
+
node: true,
|
|
40
|
+
es2022: true
|
|
41
|
+
},
|
|
42
|
+
plugins: ['@typescript-eslint', 'import'],
|
|
43
|
+
extends: [
|
|
44
|
+
'eslint:recommended',
|
|
45
|
+
'plugin:@typescript-eslint/recommended-type-checked',
|
|
46
|
+
'plugin:import/recommended',
|
|
47
|
+
'plugin:import/typescript'
|
|
48
|
+
],
|
|
49
|
+
settings: {
|
|
50
|
+
'import/resolver': {
|
|
51
|
+
typescript: {
|
|
52
|
+
alwaysTryTypes: true,
|
|
53
|
+
project: ['./tsconfig.json', './tsconfig.build.json']
|
|
54
|
+
},
|
|
55
|
+
node: {
|
|
56
|
+
extensions: ['.js', '.ts', '.tsx', '.jsx']
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
},
|
|
60
|
+
ignorePatterns: [
|
|
61
|
+
'node_modules/',
|
|
62
|
+
'dist/',
|
|
63
|
+
'coverage/',
|
|
64
|
+
'output-styles/',
|
|
65
|
+
'skills/',
|
|
66
|
+
'agents/',
|
|
67
|
+
'bin/',
|
|
68
|
+
'scratch/',
|
|
69
|
+
'examples/'
|
|
70
|
+
],
|
|
71
|
+
rules: {
|
|
72
|
+
// L1 (eslint built-in) — always on, no plugin package required.
|
|
73
|
+
complexity: ['warn', { max: 10 }],
|
|
74
|
+
'max-lines-per-function': [
|
|
75
|
+
'warn',
|
|
76
|
+
{ max: 50, skipComments: true, skipBlankLines: true }
|
|
77
|
+
],
|
|
78
|
+
'max-params': ['warn', { max: 4 }],
|
|
79
|
+
'no-magic-numbers': [
|
|
80
|
+
'warn',
|
|
81
|
+
{ ignore: [0, 1, -1, 100, 1000] }
|
|
82
|
+
],
|
|
83
|
+
'no-explicit-any': 'warn',
|
|
84
|
+
'prefer-const': 'warn',
|
|
85
|
+
'no-var': 'error',
|
|
86
|
+
eqeqeq: ['warn', 'always', { null: 'ignore' }],
|
|
87
|
+
|
|
88
|
+
// L2 (@typescript-eslint) — type-aware; requires the
|
|
89
|
+
// recommended-type-checked base. configured via the extends above.
|
|
90
|
+
'@typescript-eslint/consistent-type-imports': [
|
|
91
|
+
'warn',
|
|
92
|
+
{ prefer: 'type-imports' }
|
|
93
|
+
],
|
|
94
|
+
'@typescript-eslint/no-non-null-assertion': 'warn',
|
|
95
|
+
'@typescript-eslint/no-implicit-any': 'warn',
|
|
96
|
+
// G-lint-1 §二 enum → as const: warn-only (escape hatch preserved).
|
|
97
|
+
'@typescript-eslint/no-restricted-syntax': [
|
|
98
|
+
'warn',
|
|
99
|
+
{
|
|
100
|
+
selector: 'TSEnumDeclaration',
|
|
101
|
+
message: 'Use "as const" union instead of TS enum.'
|
|
102
|
+
}
|
|
103
|
+
],
|
|
104
|
+
|
|
105
|
+
// L2 (eslint-plugin-import) — boundary hygiene.
|
|
106
|
+
'import/no-duplicates': 'warn',
|
|
107
|
+
'import/no-unresolved': 'off',
|
|
108
|
+
'import/named': 'off',
|
|
109
|
+
'import/default': 'off',
|
|
110
|
+
'import/namespace': 'off'
|
|
111
|
+
},
|
|
112
|
+
overrides: [
|
|
113
|
+
{
|
|
114
|
+
files: ['*.test.ts', '*.test.tsx', 'tests/**/*.ts', 'tests/**/*.tsx'],
|
|
115
|
+
rules: {
|
|
116
|
+
'no-magic-numbers': 'off',
|
|
117
|
+
complexity: 'off',
|
|
118
|
+
'max-lines-per-function': 'off',
|
|
119
|
+
'@typescript-eslint/no-explicit-any': 'off'
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
]
|
|
123
|
+
};
|
|
@@ -26,6 +26,7 @@ import { registerCodeRunCommand } from './code-run-command.js';
|
|
|
26
26
|
import { registerCodeModeGateCommands } from './code-mode-gate-commands.js';
|
|
27
27
|
import { registerCodeJobShapeCommands } from './code-job-shape-commands.js';
|
|
28
28
|
import { registerCodeRuntimeCommands } from './code-runtime-commands.js';
|
|
29
|
+
import { registerCodeOrchestratorCanDoCommand } from './code-orchestrator-can-do.js';
|
|
29
30
|
const STEP_ORDER = [
|
|
30
31
|
'load-memory',
|
|
31
32
|
'standards-preflight',
|
|
@@ -114,5 +115,6 @@ export function registerCodeCommands(program, io) {
|
|
|
114
115
|
registerCodeModeGateCommands(code, io);
|
|
115
116
|
registerCodeJobShapeCommands(code, io);
|
|
116
117
|
registerCodeRuntimeCommands(code, io);
|
|
118
|
+
registerCodeOrchestratorCanDoCommand(code, io);
|
|
117
119
|
registerCodeRunCommand(code, io);
|
|
118
120
|
}
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Slice 2026-08-05-orchestrator-can-do-probe — CLI shim for
|
|
3
|
+
* `peaks code orchestrator-can-do`.
|
|
4
|
+
*
|
|
5
|
+
* Thin entry point that wires the `orchestrator-can-do` subcommand
|
|
6
|
+
* to the parent `code` command. The actual probe logic lives in
|
|
7
|
+
* `../../services/code/orchestrator-can-do.ts`.
|
|
8
|
+
*
|
|
9
|
+
* Encodes the 2026-08-05 lesson (`.peaks/memory/2026-08-05-peaks-code-
|
|
10
|
+
* orchestrator-capability-misjudgment.md`) — the peaks-code orchestrator
|
|
11
|
+
* MUST delegate source-code changes via sub-agent dispatch, not via
|
|
12
|
+
* direct Edit/Write. The probe returns a structured verdict so the LLM
|
|
13
|
+
* does not need to vibes-call "can this slice run in the current session".
|
|
14
|
+
*/
|
|
15
|
+
import type { Command } from 'commander';
|
|
16
|
+
import { type ProgramIO } from '../cli-helpers.js';
|
|
17
|
+
export declare function registerCodeOrchestratorCanDoCommand(code: Command, io: ProgramIO): void;
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Slice 2026-08-05-orchestrator-can-do-probe — CLI shim for
|
|
3
|
+
* `peaks code orchestrator-can-do`.
|
|
4
|
+
*
|
|
5
|
+
* Thin entry point that wires the `orchestrator-can-do` subcommand
|
|
6
|
+
* to the parent `code` command. The actual probe logic lives in
|
|
7
|
+
* `../../services/code/orchestrator-can-do.ts`.
|
|
8
|
+
*
|
|
9
|
+
* Encodes the 2026-08-05 lesson (`.peaks/memory/2026-08-05-peaks-code-
|
|
10
|
+
* orchestrator-capability-misjudgment.md`) — the peaks-code orchestrator
|
|
11
|
+
* MUST delegate source-code changes via sub-agent dispatch, not via
|
|
12
|
+
* direct Edit/Write. The probe returns a structured verdict so the LLM
|
|
13
|
+
* does not need to vibes-call "can this slice run in the current session".
|
|
14
|
+
*/
|
|
15
|
+
import { addJsonOption, getErrorMessage, printResult } from '../cli-helpers.js';
|
|
16
|
+
import { fail, ok } from 'peaks-loop-shared/result';
|
|
17
|
+
import { evaluateOrchestratorCanDo, OrchestratorCanDoError, ORCHESTRATOR_PRECOMPACT_RATIO, ORCHESTRATOR_REDLINE_RATIO, } from '../../services/code/orchestrator-can-do.js';
|
|
18
|
+
import { findProjectRoot } from '../../services/config/config-safety.js';
|
|
19
|
+
export function registerCodeOrchestratorCanDoCommand(code, io) {
|
|
20
|
+
addJsonOption(code
|
|
21
|
+
.command('orchestrator-can-do')
|
|
22
|
+
.description('2026-08-05 lesson: probe whether the LLM orchestrator can execute a slice in the ' +
|
|
23
|
+
'current session. Evaluates 4 boundary questions (source-code touched? sub-agent ' +
|
|
24
|
+
'available? requires user decision? context sustainable?) and returns a structured ' +
|
|
25
|
+
'verdict with canDoInSession, blockers, warnings, and concrete next-action ' +
|
|
26
|
+
'suggestions. Default to canDoInSession=true unless hard blockers are present; ' +
|
|
27
|
+
'sub-agent dispatch (`peaks sub-agent dispatch rd`) is the canonical delegation path ' +
|
|
28
|
+
'for source-code changes.')
|
|
29
|
+
.requiredOption('--slice-spec <text>', 'short description of the slice (e.g. "modify src/services/foo.ts")')
|
|
30
|
+
.option('--project <path>', 'target project root (default: findProjectRoot(cwd))')
|
|
31
|
+
.option('--peaks-bin <path>', 'peaks binary path (test seam; default: peaks on PATH)')).action(async (opts) => {
|
|
32
|
+
try {
|
|
33
|
+
const projectRoot = opts.project ?? findProjectRoot(process.cwd()) ?? process.cwd();
|
|
34
|
+
const peaksBin = opts.peaksBin ?? 'peaks';
|
|
35
|
+
const result = await evaluateOrchestratorCanDo({
|
|
36
|
+
sliceSpec: opts.sliceSpec,
|
|
37
|
+
projectRoot,
|
|
38
|
+
probeSubAgentAvailable: () => probeSubAgentAvailableWithBin(projectRoot, peaksBin),
|
|
39
|
+
probeContextRatio: () => probeContextRatioWithBin(projectRoot, peaksBin),
|
|
40
|
+
});
|
|
41
|
+
printResult(io, ok('code.orchestrator-can-do', result, [...result.warnings], [...result.suggestions, ...summaryLines(result)]), opts.json);
|
|
42
|
+
if (!result.canDoInSession)
|
|
43
|
+
process.exitCode = 1;
|
|
44
|
+
}
|
|
45
|
+
catch (err) {
|
|
46
|
+
if (err instanceof OrchestratorCanDoError) {
|
|
47
|
+
printResult(io, fail('code.orchestrator-can-do', err.code, err.message, null, [
|
|
48
|
+
'Pass --slice-spec <text> describing what the slice should change',
|
|
49
|
+
]), opts.json);
|
|
50
|
+
process.exitCode = 1;
|
|
51
|
+
return;
|
|
52
|
+
}
|
|
53
|
+
printResult(io, fail('code.orchestrator-can-do', 'PROBE_FAILED', getErrorMessage(err), null, [
|
|
54
|
+
'Verify --slice-spec is non-empty and --project is a valid path',
|
|
55
|
+
]), opts.json);
|
|
56
|
+
process.exitCode = 1;
|
|
57
|
+
}
|
|
58
|
+
});
|
|
59
|
+
}
|
|
60
|
+
function summaryLines(result) {
|
|
61
|
+
const lines = [];
|
|
62
|
+
lines.push(`verdict: ${result.canDoInSession ? 'canDoInSession=true' : 'canDoInSession=false (blockers present)'}; ` +
|
|
63
|
+
`q1=src[${result.q1SourceCodeTouched ? 'Y' : 'N'}] q2=subagent[${result.q2SubAgentAvailable ? 'Y' : 'N'}] ` +
|
|
64
|
+
`q3=user-decision[${result.q3RequiresUserDecision ? 'Y' : 'N'}] q4=ratio=${result.contextRatio.toFixed(2)} ` +
|
|
65
|
+
`(red-line ${ORCHESTRATOR_REDLINE_RATIO} / pre-compact ${ORCHESTRATOR_PRECOMPACT_RATIO})`);
|
|
66
|
+
return lines;
|
|
67
|
+
}
|
|
68
|
+
async function probeSubAgentAvailableWithBin(projectRoot, peaksBin) {
|
|
69
|
+
const { probeSubAgentAvailable } = await import('../../services/code/orchestrator-can-do.js');
|
|
70
|
+
return probeSubAgentAvailable(projectRoot, peaksBin);
|
|
71
|
+
}
|
|
72
|
+
async function probeContextRatioWithBin(projectRoot, peaksBin) {
|
|
73
|
+
const { probeContextRatio } = await import('../../services/code/orchestrator-can-do.js');
|
|
74
|
+
return probeContextRatio(projectRoot, peaksBin);
|
|
75
|
+
}
|
|
@@ -26,9 +26,12 @@ export function registerWorkspaceReconcileCommand(workspace, io) {
|
|
|
26
26
|
'By default (no --apply) the command performs four actions:\n' +
|
|
27
27
|
' 1. Migrates legacy runtime files into .peaks/_runtime/: ' +
|
|
28
28
|
'.peaks/.session.json -> .peaks/_runtime/session.json, ' +
|
|
29
|
-
'.peaks/.active-skill.json -> .peaks/_runtime/active-skill.json, ' +
|
|
30
29
|
'.peaks/sop-state/ -> .peaks/_runtime/sop-state/ ' +
|
|
31
|
-
'(idempotent; no-op if already on the new layout)
|
|
30
|
+
'(idempotent; no-op if already on the new layout). ' +
|
|
31
|
+
'Single-slot presence files (.peaks/.active-skill.json and ' +
|
|
32
|
+
'.peaks/_runtime/active-skill.json) are no longer migrated; ' +
|
|
33
|
+
'they were removed in slice 4.0.11 and should be deleted ' +
|
|
34
|
+
'manually if present.\n' +
|
|
32
35
|
' 2. Re-points .peaks/_runtime/session.json to the canonical session ' +
|
|
33
36
|
'using a 4-tier heuristic: active-skill binding -> latest session.json mtime -> ' +
|
|
34
37
|
'latest any-file mtime -> dir-name sort.\n' +
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
import type { Reporter } from 'vitest/reporters';
|
|
2
|
+
/** Minimal shape of vitest's TestModule tree node. Vitest 4.1.10 does
|
|
3
|
+
* not export these types publicly; this mirrors the runtime shape
|
|
4
|
+
* the reporter hooks actually receive. */
|
|
5
|
+
interface BddTestModuleLike {
|
|
6
|
+
moduleId?: string;
|
|
7
|
+
relativeModuleId?: string;
|
|
8
|
+
children: {
|
|
9
|
+
tests(): Iterable<BddTestCaseLike>;
|
|
10
|
+
suites(): Iterable<BddTestModuleLike>;
|
|
11
|
+
};
|
|
12
|
+
}
|
|
13
|
+
interface BddTestCaseLike {
|
|
14
|
+
name: string;
|
|
15
|
+
fullName?: string;
|
|
16
|
+
state?: 'passed' | 'failed' | 'skipped';
|
|
17
|
+
result?: () => unknown;
|
|
18
|
+
}
|
|
19
|
+
declare class BddReporter implements Reporter {
|
|
20
|
+
/** Key: relative module id; Value: per-feature rendered scenarios. */
|
|
21
|
+
private readonly features;
|
|
22
|
+
/**
|
|
23
|
+
* Vitest calls `onTestModuleEnd` after a module finishes. We use it
|
|
24
|
+
* to drain the per-module scenarios into the document map and
|
|
25
|
+
* mark the file's pass/fail status.
|
|
26
|
+
*/
|
|
27
|
+
onTestModuleEnd(testModule: BddTestModuleLike): void;
|
|
28
|
+
/**
|
|
29
|
+
* Final emit. We deliberately print to stdout with `console.log`
|
|
30
|
+
* (vitest captures stdout when needed) and never call `process.exit`
|
|
31
|
+
* — that is the orchestrator's job. Failure reasons surface as plain
|
|
32
|
+
* text so a downstream LLM prompt can grep for `FAILED:`.
|
|
33
|
+
*/
|
|
34
|
+
onTestRunEnd(): void;
|
|
35
|
+
}
|
|
36
|
+
export default BddReporter;
|
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
// src/reporters/bdd-reporter.ts
|
|
2
|
+
//
|
|
3
|
+
// rid-2026-08-05-bdd-test-style Slice C — vitest custom reporter that
|
|
4
|
+
// emits a pure BDD document view of the run. Designed for business
|
|
5
|
+
// reviewers and downstream LLM prompts; it is NOT a replacement for
|
|
6
|
+
// the default reporter.
|
|
7
|
+
//
|
|
8
|
+
// Why a custom reporter:
|
|
9
|
+
// The default reporter focuses on pass/fail and timing. The BDD
|
|
10
|
+
// reporter transcribes `Feature: <file>` / `Scenario: <describe> ->
|
|
11
|
+
// it` into a single human-readable document so a non-engineer can
|
|
12
|
+
// scan what the suite actually exercises.
|
|
13
|
+
//
|
|
14
|
+
// Why no new dep:
|
|
15
|
+
// vitest 4.1.10 (frozen 2026-07-25) ships the `Reporter` interface
|
|
16
|
+
// in `vitest/reporters`. The custom reporter must have a `default`
|
|
17
|
+
// export — the CLI loads it via `runner.import(path)` and validates
|
|
18
|
+
// `customReporterModule.default` is defined (see vitest cli-api chunks
|
|
19
|
+
// line 11371). Importing the `Reporter` type from vitest does not add
|
|
20
|
+
// a runtime dep; tsc resolves it through vitest's dts shim.
|
|
21
|
+
//
|
|
22
|
+
// Why a flag-only reporter:
|
|
23
|
+
// Per rid design section 4 Slice C, the default vitest run is
|
|
24
|
+
// unchanged. This file is opt-in via:
|
|
25
|
+
//
|
|
26
|
+
// pnpm vitest run --reporter ./src/reporters/bdd-reporter.ts <file>
|
|
27
|
+
//
|
|
28
|
+
// Anti-fake-green rule (CLI silent-catch):
|
|
29
|
+
// The reporter does not swallow vitest result shapes. Every state
|
|
30
|
+
// branch (`passed` / `failed` / `skipped`) is rendered explicitly so
|
|
31
|
+
// downstream reviewers cannot misread a hidden failure.
|
|
32
|
+
//
|
|
33
|
+
// Karpathy note:
|
|
34
|
+
// The reporter deliberately emits ONE document per file with the
|
|
35
|
+
// 4-line Feature/Scenario/Given/When/Then shape — no extra layout
|
|
36
|
+
// metadata, no JSON sidecar. Anything beyond what the spec asked
|
|
37
|
+
// for is excluded by Simplicity First.
|
|
38
|
+
class BddReporter {
|
|
39
|
+
/** Key: relative module id; Value: per-feature rendered scenarios. */
|
|
40
|
+
features = new Map();
|
|
41
|
+
/**
|
|
42
|
+
* Vitest calls `onTestModuleEnd` after a module finishes. We use it
|
|
43
|
+
* to drain the per-module scenarios into the document map and
|
|
44
|
+
* mark the file's pass/fail status.
|
|
45
|
+
*/
|
|
46
|
+
onTestModuleEnd(testModule) {
|
|
47
|
+
const moduleId = testModule.relativeModuleId ?? testModule.moduleId ?? '';
|
|
48
|
+
const file = basename(moduleId);
|
|
49
|
+
const scenarios = [];
|
|
50
|
+
collectScenarios(testModule, file, scenarios);
|
|
51
|
+
this.features.set(file, scenarios);
|
|
52
|
+
}
|
|
53
|
+
/**
|
|
54
|
+
* Final emit. We deliberately print to stdout with `console.log`
|
|
55
|
+
* (vitest captures stdout when needed) and never call `process.exit`
|
|
56
|
+
* — that is the orchestrator's job. Failure reasons surface as plain
|
|
57
|
+
* text so a downstream LLM prompt can grep for `FAILED:`.
|
|
58
|
+
*/
|
|
59
|
+
onTestRunEnd() {
|
|
60
|
+
const lines = [];
|
|
61
|
+
const features = [];
|
|
62
|
+
for (const [feature, scenarios] of this.features) {
|
|
63
|
+
const ok = scenarios.every((s) => s.state === 'passed' || s.state === 'skipped');
|
|
64
|
+
features.push({ feature, scenarios, ok });
|
|
65
|
+
}
|
|
66
|
+
// Deterministic order: alphabetical by file basename so two runs on
|
|
67
|
+
// the same diff produce byte-identical docs (avoids noisy diffs).
|
|
68
|
+
features.sort((a, b) => a.feature.localeCompare(b.feature));
|
|
69
|
+
for (const f of features) {
|
|
70
|
+
lines.push(`Feature: ${f.feature}`);
|
|
71
|
+
if (f.scenarios.length === 0) {
|
|
72
|
+
// Empty file still surfaces the Feature line so the document
|
|
73
|
+
// is a faithful list of files the runner touched.
|
|
74
|
+
lines.push('');
|
|
75
|
+
continue;
|
|
76
|
+
}
|
|
77
|
+
for (const s of f.scenarios) {
|
|
78
|
+
lines.push(` Scenario: ${s.scenario || '<root>'}`);
|
|
79
|
+
lines.push(` Given ${s.title}`);
|
|
80
|
+
lines.push(` When vitest runs this test`);
|
|
81
|
+
if (s.state === 'passed') {
|
|
82
|
+
lines.push(` Then should pass`);
|
|
83
|
+
}
|
|
84
|
+
else if (s.state === 'skipped') {
|
|
85
|
+
lines.push(` Then should skip`);
|
|
86
|
+
}
|
|
87
|
+
else {
|
|
88
|
+
const reason = s.error ? ` (${truncate(s.error, 200)})` : '';
|
|
89
|
+
lines.push(` Then FAILED: ${s.title}${reason}`);
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
lines.push('');
|
|
93
|
+
}
|
|
94
|
+
console.log(lines.join('\n'));
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
function basename(path) {
|
|
98
|
+
// vitest module ids are POSIX-style even on Windows; split on '/'
|
|
99
|
+
// then on '\\' as a defensive fallback.
|
|
100
|
+
const idx = Math.max(path.lastIndexOf('/'), path.lastIndexOf('\\'));
|
|
101
|
+
return idx === -1 ? path : path.slice(idx + 1);
|
|
102
|
+
}
|
|
103
|
+
/**
|
|
104
|
+
* Walk a `TestModule` recursively and collect rendered scenarios.
|
|
105
|
+
* `describe` blocks contribute their name to the Scenario label;
|
|
106
|
+
* tests declared at module root produce a `<root>` Scenario so the
|
|
107
|
+
* structure is uniform.
|
|
108
|
+
*/
|
|
109
|
+
function collectScenarios(entity, file, out) {
|
|
110
|
+
const visited = new WeakSet();
|
|
111
|
+
const walk = (node, scenarioLabel) => {
|
|
112
|
+
if (node === null || typeof node !== 'object')
|
|
113
|
+
return;
|
|
114
|
+
if (visited.has(node))
|
|
115
|
+
return;
|
|
116
|
+
visited.add(node);
|
|
117
|
+
const obj = node;
|
|
118
|
+
if (obj.type === 'test') {
|
|
119
|
+
const tc = node;
|
|
120
|
+
const result = tc.result ? tc.result() : undefined;
|
|
121
|
+
const resultObj = (result ?? {});
|
|
122
|
+
const state = (resultObj.state === 'passed' || resultObj.state === 'failed' || resultObj.state === 'skipped')
|
|
123
|
+
? resultObj.state
|
|
124
|
+
: 'skipped';
|
|
125
|
+
const err = resultObj.state === 'failed' && resultObj.errors && resultObj.errors[0]
|
|
126
|
+
? (resultObj.errors[0].message ?? 'unknown failure')
|
|
127
|
+
: undefined;
|
|
128
|
+
out.push({
|
|
129
|
+
scenario: scenarioLabel,
|
|
130
|
+
title: tc.name,
|
|
131
|
+
state,
|
|
132
|
+
error: err,
|
|
133
|
+
});
|
|
134
|
+
return;
|
|
135
|
+
}
|
|
136
|
+
// For a suite/module, descend with the suite's name pushed.
|
|
137
|
+
const suiteName = obj.name ?? '';
|
|
138
|
+
const childSuiteLabel = suiteName || scenarioLabel;
|
|
139
|
+
if (obj.children) {
|
|
140
|
+
try {
|
|
141
|
+
for (const t of obj.children.tests()) {
|
|
142
|
+
walk(t, childSuiteLabel);
|
|
143
|
+
}
|
|
144
|
+
for (const s of obj.children.suites()) {
|
|
145
|
+
walk(s, childSuiteLabel);
|
|
146
|
+
}
|
|
147
|
+
}
|
|
148
|
+
catch {
|
|
149
|
+
// Defensive: vitest internals may throw on teardown. We do not
|
|
150
|
+
// mask the document — we just stop collecting from this node.
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
};
|
|
154
|
+
walk(entity, '');
|
|
155
|
+
}
|
|
156
|
+
function truncate(s, n) {
|
|
157
|
+
return s.length <= n ? s : `${s.slice(0, n - 3)}...`;
|
|
158
|
+
}
|
|
159
|
+
export default BddReporter;
|
|
@@ -127,7 +127,10 @@ export function resolveActiveSkillForCaller(projectRoot, opts) {
|
|
|
127
127
|
const raw = readFileSync(filePath, 'utf8');
|
|
128
128
|
const parsed = JSON.parse(raw);
|
|
129
129
|
if (typeof parsed.skill === 'string' && parsed.skill.length > 0) {
|
|
130
|
-
|
|
130
|
+
const legacyMode = typeof parsed.mode === 'string' && parsed.mode.length > 0
|
|
131
|
+
? parsed.mode
|
|
132
|
+
: null;
|
|
133
|
+
return { skill: parsed.skill, callerId, sessionId, mode: legacyMode, source: 'file' };
|
|
131
134
|
}
|
|
132
135
|
}
|
|
133
136
|
catch { // TODO(g2): legacy silent catch — grace: 1 minor release (v2.14.0)
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Slice 2026-08-05-orchestrator-can-do-probe — service layer for
|
|
3
|
+
* `peaks code orchestrator-can-do`.
|
|
4
|
+
*
|
|
5
|
+
* Encodes the 2026-08-05 lesson (`.peaks/memory/2026-08-05-peaks-code-
|
|
6
|
+
* orchestrator-capability-misjudgment.md`): the peaks-code orchestrator
|
|
7
|
+
* MUST NOT Edit/Write `src/` files directly, but MUST delegate via
|
|
8
|
+
* `peaks sub-agent dispatch`. The decision "can this slice run in the
|
|
9
|
+
* current session" must be a structured probe — not a vibes call.
|
|
10
|
+
*
|
|
11
|
+
* The probe answers 4 boundary questions:
|
|
12
|
+
* Q1 — Is the change to source code? (keywords: src/, *.ts, *.tsx,
|
|
13
|
+
* *.js, package.json, tsconfig, workflows/). If yes → NOT a
|
|
14
|
+
* blocker; orchestrator delegates to sub-agent.
|
|
15
|
+
* Q2 — Can a sub-agent be dispatched? (probe `peaks sub-agent
|
|
16
|
+
* dispatch --role rd --help`). If no → blocker.
|
|
17
|
+
* Q3 — Does the slice require user decisions? (keywords: design,
|
|
18
|
+
* decide, ?, 选择, 决定). If yes → soft warning, NOT a blocker
|
|
19
|
+
* (the LLM should AskUserQuestion, which is cheap).
|
|
20
|
+
* Q4 — Is context usage sustainable? (probe `peaks code context-now
|
|
21
|
+
* --json`). ratio ≥ 0.95 → blocker (red-line); ≥ 0.85 →
|
|
22
|
+
* blocker (auto-compact-now).
|
|
23
|
+
*
|
|
24
|
+
* Decision rule:
|
|
25
|
+
* canDoInSession === (blockers.length === 0)
|
|
26
|
+
*
|
|
27
|
+
* Concrete suggestion when canDoInSession=true and slice touches
|
|
28
|
+
* source code:
|
|
29
|
+
* `peaks sub-agent dispatch rd --prompt "<slice-spec>" --request-id
|
|
30
|
+
* <rid> --project . --batch-id <uuid>`
|
|
31
|
+
*
|
|
32
|
+
* Pure-function module. The CLI shim (code-orchestrator-can-do.ts)
|
|
33
|
+
* adapts the envelope into the program's `ResultEnvelope<T>` shape.
|
|
34
|
+
*/
|
|
35
|
+
/** Slice 2026-08-05-orchestrator-can-do-probe: red-line threshold. */
|
|
36
|
+
export declare const ORCHESTRATOR_REDLINE_RATIO = 0.95;
|
|
37
|
+
/** Slice 2026-08-05-orchestrator-can-do-probe: pre-compact threshold. */
|
|
38
|
+
export declare const ORCHESTRATOR_PRECOMPACT_RATIO = 0.85;
|
|
39
|
+
/** Source-code keywords that signal "do NOT Edit/Write directly". */
|
|
40
|
+
export declare const SOURCE_CODE_KEYWORDS: readonly string[];
|
|
41
|
+
/** Decision-marker keywords that signal "needs user AskUserQuestion". */
|
|
42
|
+
export declare const DECISION_KEYWORDS: readonly string[];
|
|
43
|
+
export interface ContextProbe {
|
|
44
|
+
/** 0.0–1.0; ≥0.85 = pre-compact; ≥0.95 = red-line. */
|
|
45
|
+
readonly ratio: number;
|
|
46
|
+
/** Source tag from `peaks code context-now`. */
|
|
47
|
+
readonly source: string;
|
|
48
|
+
}
|
|
49
|
+
export interface OrchestratorCanDoInput {
|
|
50
|
+
readonly sliceSpec: string;
|
|
51
|
+
readonly projectRoot: string;
|
|
52
|
+
/**
|
|
53
|
+
* Test seam — caller injects probe results. Production CLI builds
|
|
54
|
+
* these via `probeSubAgentAvailable` + `probeContextRatio`. When the
|
|
55
|
+
* test seam is set, the CLI's actual probes are skipped.
|
|
56
|
+
*/
|
|
57
|
+
readonly probeSubAgentAvailable?: () => Promise<boolean>;
|
|
58
|
+
readonly probeContextRatio?: () => Promise<ContextProbe>;
|
|
59
|
+
}
|
|
60
|
+
export interface OrchestratorCanDoResult {
|
|
61
|
+
/** canDoInSession === (blockers.length === 0). */
|
|
62
|
+
readonly canDoInSession: boolean;
|
|
63
|
+
readonly blockers: readonly string[];
|
|
64
|
+
readonly warnings: readonly string[];
|
|
65
|
+
readonly suggestions: readonly string[];
|
|
66
|
+
readonly contextRatio: number;
|
|
67
|
+
readonly subAgentAvailable: boolean;
|
|
68
|
+
/** Diagnostic — which of the 4 boundary questions fired. */
|
|
69
|
+
readonly q1SourceCodeTouched: boolean;
|
|
70
|
+
readonly q2SubAgentAvailable: boolean;
|
|
71
|
+
readonly q3RequiresUserDecision: boolean;
|
|
72
|
+
readonly q4ContextRatio: number;
|
|
73
|
+
}
|
|
74
|
+
export declare class OrchestratorCanDoError extends Error {
|
|
75
|
+
readonly code: 'MISSING_SLICE_SPEC' | 'PROBE_FAILED';
|
|
76
|
+
constructor(message: string, code: 'MISSING_SLICE_SPEC' | 'PROBE_FAILED');
|
|
77
|
+
}
|
|
78
|
+
/**
|
|
79
|
+
* Q1: does the slice touch source code? Pure keyword scan over
|
|
80
|
+
* the slice-spec string. Case-insensitive substring match.
|
|
81
|
+
*/
|
|
82
|
+
export declare function detectSourceCodeTouched(sliceSpec: string): boolean;
|
|
83
|
+
/**
|
|
84
|
+
* Q3: does the slice require user decisions? Pure keyword scan
|
|
85
|
+
* over the slice-spec string.
|
|
86
|
+
*/
|
|
87
|
+
export declare function detectRequiresUserDecision(sliceSpec: string): boolean;
|
|
88
|
+
/**
|
|
89
|
+
* Q2: probe `peaks sub-agent dispatch --role rd --help`. Returns
|
|
90
|
+
* true when the subprocess exits 0. Resolves to false on spawn
|
|
91
|
+
* failure or non-zero exit.
|
|
92
|
+
*/
|
|
93
|
+
export declare function probeSubAgentAvailable(projectRoot: string, peaksBin?: string): Promise<boolean>;
|
|
94
|
+
/**
|
|
95
|
+
* Q4: probe `peaks code context-now --json`. Parses the data.ratio
|
|
96
|
+
* field. Falls back to {ratio: 0, source: 'unavailable'} when the
|
|
97
|
+
* subprocess fails or returns malformed JSON.
|
|
98
|
+
*/
|
|
99
|
+
export declare function probeContextRatio(projectRoot: string, peaksBin?: string): Promise<ContextProbe>;
|
|
100
|
+
/**
|
|
101
|
+
* Build the structured OrchestratorCanDoResult. Pure over the 4 Q
|
|
102
|
+
* signals + sliceSpec. Decision rule is: canDoInSession === !blockers.
|
|
103
|
+
*/
|
|
104
|
+
export declare function buildOrchestratorCanDoResult(input: OrchestratorCanDoInput, signals: {
|
|
105
|
+
q1SourceCodeTouched: boolean;
|
|
106
|
+
q2SubAgentAvailable: boolean;
|
|
107
|
+
q3RequiresUserDecision: boolean;
|
|
108
|
+
q4ContextRatio: number;
|
|
109
|
+
}): OrchestratorCanDoResult;
|
|
110
|
+
/**
|
|
111
|
+
* Evaluate a slice-spec end-to-end. Probes Q2/Q4 via subprocess
|
|
112
|
+
* (overridable via test seams in `input`). Q1/Q3 are pure.
|
|
113
|
+
*/
|
|
114
|
+
export declare function evaluateOrchestratorCanDo(input: OrchestratorCanDoInput): Promise<OrchestratorCanDoResult>;
|