mandrel 2.64.0 → 2.65.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/agents/acceptance-critic.md +3 -2
- package/.agents/agents/auditor.md +3 -2
- package/.agents/agents/plan-critic.md +3 -2
- package/.agents/agents/story-worker.md +2 -2
- package/.agents/audit-checklists/quality.md +3 -0
- package/.agents/docs/agentrc-reference.json +1 -9
- package/.agents/docs/configuration.md +8 -7
- package/.agents/schemas/agentrc.schema.json +6 -13
- package/.agents/schemas/audit-rules.schema.json +1 -1
- package/.agents/schemas/story-deliver-terminal.schema.json +5 -0
- package/.agents/scripts/bootstrap.js +8 -2
- package/.agents/scripts/check-context-budget.js +1 -1
- package/.agents/scripts/lib/ITicketingProvider.js +1 -3
- package/.agents/scripts/lib/audit-suite/findings.js +1 -17
- package/.agents/scripts/lib/audit-suite/frontmatter.js +0 -28
- package/.agents/scripts/lib/audit-suite/index.js +0 -6
- package/.agents/scripts/lib/audit-suite/selector.js +0 -31
- package/.agents/scripts/lib/bootstrap/agents-md-fold.js +156 -0
- package/.agents/scripts/lib/bootstrap/commit-push.js +1 -1
- package/.agents/scripts/lib/bootstrap/manifest.js +2 -2
- package/.agents/scripts/lib/bootstrap/project-bootstrap.js +13 -29
- package/.agents/scripts/lib/config/review-chain-default.js +13 -0
- package/.agents/scripts/lib/config-settings-schema-delivery.js +2 -2
- package/.agents/scripts/lib/config-settings-schema-quality.js +11 -13
- package/.agents/scripts/lib/doc-tiers.js +25 -6
- package/.agents/scripts/lib/generated/agentrc-validator.js +1 -1
- package/.agents/scripts/lib/observability/metrics-ledger.js +0 -72
- package/.agents/scripts/lib/orchestration/code-review.js +11 -6
- package/.agents/scripts/lib/orchestration/epic-rollup.js +29 -12
- package/.agents/scripts/lib/orchestration/merge-block-class.js +20 -4
- package/.agents/scripts/lib/orchestration/merge-poll.js +41 -22
- package/.agents/scripts/lib/orchestration/required-checks.js +147 -0
- package/.agents/scripts/lib/orchestration/review-providers/code-review.js +203 -0
- package/.agents/scripts/lib/orchestration/review-providers/review-provider-factory.js +6 -4
- package/.agents/scripts/lib/orchestration/review-providers/security-review.js +3 -2
- package/.agents/scripts/lib/orchestration/single-story-close/failed-terminal.js +1 -0
- package/.agents/scripts/lib/orchestration/single-story-close/phases/code-review.js +0 -12
- package/.agents/scripts/lib/orchestration/single-story-close/phases/confirm-merge.js +27 -11
- package/.agents/scripts/lib/orchestration/single-story-close/phases/post-land.js +112 -82
- package/.agents/scripts/lib/orchestration/single-story-close/runner.js +72 -5
- package/.agents/scripts/lib/orchestration/story-close/phases/review-core.js +12 -87
- package/.agents/scripts/lib/orchestration/story-deliver-terminal.js +3 -0
- package/.agents/scripts/lib/templates/decomposer-prompts.js +5 -24
- package/.agents/scripts/providers/github/issues.js +14 -23
- package/.agents/scripts/sync-claude-agents.js +1 -1
- package/.agents/workflows/audit-quality.md +42 -7
- package/.agents/workflows/helpers/acceptance-self-eval.md +1 -1
- package/.agents/workflows/helpers/code-review.md +15 -38
- package/.agents/workflows/helpers/deliver-reference.md +4 -2
- package/.agents/workflows/helpers/deliver-story.md +3 -0
- package/.agents/workflows/helpers/plan-reference.md +9 -8
- package/.agents/workflows/mandrel-deliver.md +2 -1
- package/.agents/workflows/mandrel-plan.md +10 -7
- package/.agents/workflows/mandrel-update.md +5 -3
- package/docs/CHANGELOG.md +31 -0
- package/lib/cli/claude-code-version.js +73 -0
- package/lib/cli/doctor.js +2 -2
- package/lib/cli/registry.js +9 -0
- package/lib/cli/uninstall.js +37 -9
- package/lib/migrations/index.js +2 -0
- package/lib/migrations/steps/2.65.0-fold-claude-md-into-agents-md.js +38 -0
- package/package.json +2 -1
- package/.agents/scripts/lib/audit-suite/lens-diff-floor.js +0 -99
- package/.agents/scripts/lib/audit-suite/runner.js +0 -205
- package/.agents/scripts/lib/audit-suite/substitutions.js +0 -96
- package/.agents/scripts/lib/audit-suite/workflow-loader.js +0 -37
- package/.agents/scripts/lib/orchestration/story-close/phases/local-lens-review.js +0 -234
|
@@ -33,6 +33,7 @@ import { runConfirmMergePhase } from './phases/confirm-merge.js';
|
|
|
33
33
|
import { runGraphqlPreflight } from './phases/graphql-preflight.js';
|
|
34
34
|
import { lockWaitPending } from './phases/lock-wait-pending.js';
|
|
35
35
|
import { parseCloseOptions, resolveWaitForMerge } from './phases/options.js';
|
|
36
|
+
import { runPostLandTail } from './phases/post-land.js';
|
|
36
37
|
import { ensurePullRequestWith } from './phases/pull-request.js';
|
|
37
38
|
import { pushStoryBranch } from './phases/push.js';
|
|
38
39
|
import { handleCriticalReviewBlock } from './phases/review-block.js';
|
|
@@ -42,12 +43,54 @@ import { runWrongTreeGuardPhase } from './phases/wrong-tree-guard.js';
|
|
|
42
43
|
|
|
43
44
|
const progress = Logger.createProgress('single-story-close', { stderr: true });
|
|
44
45
|
|
|
46
|
+
/**
|
|
47
|
+
* Wall-clock seconds per named phase; each transition logs the phase it ends.
|
|
48
|
+
*
|
|
49
|
+
* @param {() => number} [nowMs]
|
|
50
|
+
*/
|
|
51
|
+
function createPhaseTimer(nowMs = Date.now) {
|
|
52
|
+
const durations = {};
|
|
53
|
+
let current = null;
|
|
54
|
+
let since = 0;
|
|
55
|
+
const end = () => {
|
|
56
|
+
if (current === null) return;
|
|
57
|
+
const seconds = Math.round((nowMs() - since) / 100) / 10;
|
|
58
|
+
durations[current] = (durations[current] ?? 0) + seconds;
|
|
59
|
+
progress('TIMING', `⏱ ${current}: ${seconds}s`);
|
|
60
|
+
current = null;
|
|
61
|
+
};
|
|
62
|
+
return {
|
|
63
|
+
enter(phase) {
|
|
64
|
+
end();
|
|
65
|
+
if (phase === 'init') return;
|
|
66
|
+
current = phase;
|
|
67
|
+
since = nowMs();
|
|
68
|
+
},
|
|
69
|
+
finish() {
|
|
70
|
+
end();
|
|
71
|
+
return Object.keys(durations).length > 0 ? { ...durations } : null;
|
|
72
|
+
},
|
|
73
|
+
stamp(terminal) {
|
|
74
|
+
const phaseDurations = this.finish();
|
|
75
|
+
if (phaseDurations) terminal.phaseDurations = phaseDurations;
|
|
76
|
+
},
|
|
77
|
+
};
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
const UNTIMED = Object.freeze({ stamp() {} });
|
|
81
|
+
|
|
45
82
|
/**
|
|
46
83
|
* The single terminal writer: the result summary, the envelope callers parse,
|
|
47
84
|
* and terminal friction — so no ending can forget one. Must be awaited: the
|
|
48
85
|
* CLI `process.exit`s as soon as `main` resolves.
|
|
49
86
|
*/
|
|
50
|
-
async function emitTerminal({
|
|
87
|
+
async function emitTerminal({
|
|
88
|
+
terminal,
|
|
89
|
+
result,
|
|
90
|
+
config,
|
|
91
|
+
phaseTimer = UNTIMED,
|
|
92
|
+
}) {
|
|
93
|
+
phaseTimer.stamp(terminal);
|
|
51
94
|
if (result) {
|
|
52
95
|
emitTerseResult({
|
|
53
96
|
label: 'STORY CLOSE RESULT',
|
|
@@ -516,8 +559,10 @@ export async function runSingleStoryClose({
|
|
|
516
559
|
// builds the `failed` envelope from those tags.
|
|
517
560
|
let phase = 'init';
|
|
518
561
|
let observedGates = null;
|
|
562
|
+
const phaseTimer = createPhaseTimer();
|
|
519
563
|
const setPhase = (next) => {
|
|
520
564
|
phase = next;
|
|
565
|
+
phaseTimer.enter(next);
|
|
521
566
|
};
|
|
522
567
|
const setObservedGates = (gates) => {
|
|
523
568
|
observedGates = gates;
|
|
@@ -527,6 +572,7 @@ export async function runSingleStoryClose({
|
|
|
527
572
|
options,
|
|
528
573
|
setPhase,
|
|
529
574
|
setObservedGates,
|
|
575
|
+
phaseTimer,
|
|
530
576
|
injectedProvider,
|
|
531
577
|
injectedConfig,
|
|
532
578
|
injectedNotify,
|
|
@@ -539,6 +585,7 @@ export async function runSingleStoryClose({
|
|
|
539
585
|
});
|
|
540
586
|
} catch (err) {
|
|
541
587
|
if (err && typeof err === 'object') {
|
|
588
|
+
err.closePhaseDurations = phaseTimer.finish();
|
|
542
589
|
if (!err.closePhase) err.closePhase = phase;
|
|
543
590
|
if (!err.closeGates && observedGates) err.closeGates = observedGates;
|
|
544
591
|
}
|
|
@@ -617,6 +664,10 @@ async function finishWithMergeWait(prCtx, deps) {
|
|
|
617
664
|
progress,
|
|
618
665
|
injectedGh: deps.injectedGh,
|
|
619
666
|
injectedNotify: deps.injectedNotify,
|
|
667
|
+
runPostLandTailFn: (args) => {
|
|
668
|
+
deps.setPhase('post-land');
|
|
669
|
+
return runPostLandTail(args);
|
|
670
|
+
},
|
|
620
671
|
});
|
|
621
672
|
const terminal = terminalFromWaitOutcome({
|
|
622
673
|
waitOutcome,
|
|
@@ -650,7 +701,12 @@ async function finishWithMergeWait(prCtx, deps) {
|
|
|
650
701
|
}),
|
|
651
702
|
landCompleted: waitOutcome.confirmed === true,
|
|
652
703
|
});
|
|
653
|
-
await emitTerminal({
|
|
704
|
+
await emitTerminal({
|
|
705
|
+
terminal,
|
|
706
|
+
result,
|
|
707
|
+
config: prCtx.config,
|
|
708
|
+
phaseTimer: prCtx.phaseTimer,
|
|
709
|
+
});
|
|
654
710
|
reportWaitTerminal(terminal, { storyId: prCtx.storyId, prUrl: prCtx.prUrl });
|
|
655
711
|
return { success: terminal.status === 'landed', result, terminal };
|
|
656
712
|
}
|
|
@@ -698,7 +754,12 @@ async function finishWithoutMergeWait(prCtx, waitForMergeReason) {
|
|
|
698
754
|
nextCommand: NEXT_COMMANDS.confirmMerge(prCtx.storyId),
|
|
699
755
|
elapsedSeconds: elapsedSecondsSince(prCtx.startedAtMs),
|
|
700
756
|
});
|
|
701
|
-
await emitTerminal({
|
|
757
|
+
await emitTerminal({
|
|
758
|
+
terminal,
|
|
759
|
+
result,
|
|
760
|
+
config: prCtx.config,
|
|
761
|
+
phaseTimer: prCtx.phaseTimer,
|
|
762
|
+
});
|
|
702
763
|
progress(
|
|
703
764
|
'DONE',
|
|
704
765
|
`✅ Story #${prCtx.storyId}: PR ready → ${prCtx.prUrl} (${waitForMergeReason})`,
|
|
@@ -714,13 +775,16 @@ async function finishWithoutMergeWait(prCtx, waitForMergeReason) {
|
|
|
714
775
|
* config: object, startedAtMs: number }} ctx
|
|
715
776
|
* @returns {Promise<{ success: false, result: object, terminal: object }>}
|
|
716
777
|
*/
|
|
717
|
-
async function finishDeferred(
|
|
778
|
+
async function finishDeferred(
|
|
779
|
+
lockWait,
|
|
780
|
+
{ config, startedAtMs, phaseTimer, ...ids },
|
|
781
|
+
) {
|
|
718
782
|
const { result, terminal, note } = lockWaitPending({
|
|
719
783
|
...ids,
|
|
720
784
|
lockWait,
|
|
721
785
|
elapsedSeconds: elapsedSecondsSince(startedAtMs),
|
|
722
786
|
});
|
|
723
|
-
await emitTerminal({ terminal, result, config });
|
|
787
|
+
await emitTerminal({ terminal, result, config, phaseTimer });
|
|
724
788
|
progress('PENDING', note);
|
|
725
789
|
return { success: false, result, terminal };
|
|
726
790
|
}
|
|
@@ -759,6 +823,7 @@ async function runClosePipeline({
|
|
|
759
823
|
options,
|
|
760
824
|
setPhase,
|
|
761
825
|
setObservedGates,
|
|
826
|
+
phaseTimer,
|
|
762
827
|
injectedProvider,
|
|
763
828
|
injectedConfig,
|
|
764
829
|
injectedNotify,
|
|
@@ -848,6 +913,7 @@ async function runClosePipeline({
|
|
|
848
913
|
baseBranch,
|
|
849
914
|
config,
|
|
850
915
|
startedAtMs,
|
|
916
|
+
phaseTimer,
|
|
851
917
|
});
|
|
852
918
|
}
|
|
853
919
|
|
|
@@ -943,6 +1009,7 @@ async function runClosePipeline({
|
|
|
943
1009
|
directMerged,
|
|
944
1010
|
config,
|
|
945
1011
|
startedAtMs,
|
|
1012
|
+
phaseTimer,
|
|
946
1013
|
lockWait: prePush.lockWait,
|
|
947
1014
|
gates: closeEnvelopeGates(options, prePush.validationGates, reviewOverride),
|
|
948
1015
|
};
|
|
@@ -3,23 +3,20 @@
|
|
|
3
3
|
* maker-blind `runCodeReview` invocation out of any phase file.
|
|
4
4
|
*/
|
|
5
5
|
|
|
6
|
-
import { countChangedLines } from '../../../audit-suite/index.js';
|
|
7
6
|
import { gitSpawn } from '../../../git-utils.js';
|
|
8
|
-
import { appendFindingsYield } from '../../../observability/metrics-ledger.js';
|
|
9
7
|
import { computeChangeSet } from '../../change-set.js';
|
|
10
8
|
import { runCodeReview } from '../../code-review.js';
|
|
11
|
-
import { runLocalLensReview } from './local-lens-review.js';
|
|
12
9
|
|
|
13
10
|
/**
|
|
14
|
-
* Run
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
*
|
|
11
|
+
* Run `runCodeReview` over one change set and return the review result.
|
|
12
|
+
* Throws propagate; the caller picks the advisory posture. Review depth is
|
|
13
|
+
* derived by `runCodeReview` from the changed files and is input-only — it
|
|
14
|
+
* never alters the output envelope.
|
|
18
15
|
*
|
|
19
|
-
* The diff is enumerated exactly once here and injected into
|
|
20
|
-
*
|
|
21
|
-
* lands in between. An unenumerable diff injects `null` ("already
|
|
22
|
-
* which
|
|
16
|
+
* The diff is enumerated exactly once here and injected into the review, so
|
|
17
|
+
* review depth scores the same file set the change set names even if a
|
|
18
|
+
* commit lands in between. An unenumerable diff injects `null` ("already
|
|
19
|
+
* tried"), which the review honours without re-spawning git.
|
|
23
20
|
*
|
|
24
21
|
* @param {{
|
|
25
22
|
* storyId: number|string,
|
|
@@ -32,12 +29,9 @@ import { runLocalLensReview } from './local-lens-review.js';
|
|
|
32
29
|
* gitSpawnFn?: import('../../change-set.js').GitSpawnFn,
|
|
33
30
|
* computeChangeSetFn?: typeof computeChangeSet,
|
|
34
31
|
* runCodeReviewFn?: typeof runCodeReview,
|
|
35
|
-
* runLocalLensReviewFn?: typeof runLocalLensReview,
|
|
36
|
-
* countChangedLinesFn?: typeof countChangedLines,
|
|
37
|
-
* appendFindingsYieldFn?: typeof appendFindingsYield,
|
|
38
32
|
* }} args
|
|
39
|
-
* @returns {Promise<object>} The `runCodeReview` result plus
|
|
40
|
-
* `
|
|
33
|
+
* @returns {Promise<object>} The `runCodeReview` result plus the computed
|
|
34
|
+
* `changeSet`.
|
|
41
35
|
*/
|
|
42
36
|
export async function runStoryReviewCore({
|
|
43
37
|
storyId,
|
|
@@ -50,24 +44,12 @@ export async function runStoryReviewCore({
|
|
|
50
44
|
gitSpawnFn = gitSpawn,
|
|
51
45
|
computeChangeSetFn = computeChangeSet,
|
|
52
46
|
runCodeReviewFn = runCodeReview,
|
|
53
|
-
runLocalLensReviewFn = runLocalLensReview,
|
|
54
|
-
countChangedLinesFn = countChangedLines,
|
|
55
|
-
appendFindingsYieldFn = appendFindingsYield,
|
|
56
47
|
}) {
|
|
57
|
-
const storyIdNum = Number(storyId);
|
|
58
|
-
|
|
59
48
|
const changeSet = computeChangeSetFn({ baseRef, headRef, gitSpawnFn });
|
|
60
49
|
|
|
61
|
-
// Line count for the lens diff-floor, probed only for a non-empty file set;
|
|
62
|
-
// `null` = unknown, and the floor fails open.
|
|
63
|
-
const changedLineCount =
|
|
64
|
-
Array.isArray(changeSet.files) && changeSet.files.length > 0
|
|
65
|
-
? countChangedLinesFn({ baseRef, headRef, gitSpawnFn })
|
|
66
|
-
: null;
|
|
67
|
-
|
|
68
50
|
const opts = {
|
|
69
51
|
scope: 'story',
|
|
70
|
-
ticketId:
|
|
52
|
+
ticketId: Number(storyId),
|
|
71
53
|
baseRef,
|
|
72
54
|
headRef,
|
|
73
55
|
provider,
|
|
@@ -82,63 +64,6 @@ export async function runStoryReviewCore({
|
|
|
82
64
|
opts.commentTargetId = commentTargetId;
|
|
83
65
|
}
|
|
84
66
|
|
|
85
|
-
const localLensReview = await runLocalLensReviewFn({
|
|
86
|
-
baseRef,
|
|
87
|
-
headRef,
|
|
88
|
-
changedFiles: changeSet.files,
|
|
89
|
-
changedLineCount,
|
|
90
|
-
storyId: storyIdNum,
|
|
91
|
-
progress,
|
|
92
|
-
progressTag,
|
|
93
|
-
gitSpawnFn,
|
|
94
|
-
});
|
|
95
|
-
|
|
96
67
|
const result = await runCodeReviewFn(opts);
|
|
97
|
-
|
|
98
|
-
// Best-effort findings-yield ledger, for tuning the roster on measurement.
|
|
99
|
-
try {
|
|
100
|
-
const yieldEntries = buildLensYieldEntries(localLensReview);
|
|
101
|
-
if (yieldEntries !== null) {
|
|
102
|
-
await appendFindingsYieldFn({
|
|
103
|
-
storyId: storyIdNum,
|
|
104
|
-
cli: 'story-close-review',
|
|
105
|
-
lenses: yieldEntries,
|
|
106
|
-
diffFloor: localLensReview?.floorSkip ?? null,
|
|
107
|
-
});
|
|
108
|
-
}
|
|
109
|
-
} catch (err) {
|
|
110
|
-
progress(
|
|
111
|
-
progressTag,
|
|
112
|
-
`⚠️ findings-yield ledger append failed (continuing): ${err?.message ?? err}`,
|
|
113
|
-
);
|
|
114
|
-
}
|
|
115
|
-
|
|
116
|
-
return { ...result, localLensReview, changeSet };
|
|
117
|
-
}
|
|
118
|
-
|
|
119
|
-
/**
|
|
120
|
-
* One findings-yield entry per matched lens; `null` for an empty roster.
|
|
121
|
-
*
|
|
122
|
-
* @param {object|null|undefined} localLensReview
|
|
123
|
-
* @returns {Array<{ lens: string, findings: number, skippedByFloor: boolean }>|null}
|
|
124
|
-
*/
|
|
125
|
-
function buildLensYieldEntries(localLensReview) {
|
|
126
|
-
const lenses = Array.isArray(localLensReview?.lenses)
|
|
127
|
-
? localLensReview.lenses.filter((l) => typeof l === 'string' && l.length)
|
|
128
|
-
: [];
|
|
129
|
-
if (lenses.length === 0) return null;
|
|
130
|
-
const skippedByFloor = localLensReview?.floorSkip?.skip === true;
|
|
131
|
-
const findingsByLens = new Map();
|
|
132
|
-
for (const finding of localLensReview?.materialized?.findings ?? []) {
|
|
133
|
-
if (typeof finding?.audit !== 'string') continue;
|
|
134
|
-
findingsByLens.set(
|
|
135
|
-
finding.audit,
|
|
136
|
-
(findingsByLens.get(finding.audit) ?? 0) + 1,
|
|
137
|
-
);
|
|
138
|
-
}
|
|
139
|
-
return lenses.map((lens) => ({
|
|
140
|
-
lens,
|
|
141
|
-
findings: skippedByFloor ? 0 : (findingsByLens.get(lens) ?? 0),
|
|
142
|
-
skippedByFloor,
|
|
143
|
-
}));
|
|
68
|
+
return { ...result, changeSet };
|
|
144
69
|
}
|
|
@@ -116,6 +116,7 @@ function compact(obj) {
|
|
|
116
116
|
* @param {object|null} [args.waitBudget]
|
|
117
117
|
* @param {{ waitedSeconds: number, expired: boolean }|null} [args.lockWait]
|
|
118
118
|
* Full-suite lock wait; `waitBudget` is merge-wait only.
|
|
119
|
+
* @param {Record<string, number>|null} [args.phaseDurations] Seconds per phase.
|
|
119
120
|
* @param {string} [args.timestamp]
|
|
120
121
|
* @param {{ schema: object|null, error: string|null }} [args.schemaSource]
|
|
121
122
|
* Test seam.
|
|
@@ -137,6 +138,7 @@ export function buildTerminalEnvelope({
|
|
|
137
138
|
elapsedSeconds = 0,
|
|
138
139
|
waitBudget,
|
|
139
140
|
lockWait,
|
|
141
|
+
phaseDurations,
|
|
140
142
|
timestamp = new Date().toISOString(),
|
|
141
143
|
schemaSource,
|
|
142
144
|
}) {
|
|
@@ -158,6 +160,7 @@ export function buildTerminalEnvelope({
|
|
|
158
160
|
elapsedSeconds: Math.max(0, Number(elapsedSeconds) || 0),
|
|
159
161
|
waitBudget: waitBudget ?? null,
|
|
160
162
|
lockWait: lockWait ?? null,
|
|
163
|
+
phaseDurations,
|
|
161
164
|
timestamp,
|
|
162
165
|
});
|
|
163
166
|
|
|
@@ -31,13 +31,11 @@ export function renderStoryAuthorCore() {
|
|
|
31
31
|
(lint) =>
|
|
32
32
|
`- **${lint.id}** — ${lint.summary} Example: \`${lint.goodExample}\``,
|
|
33
33
|
).join('\n');
|
|
34
|
-
return `
|
|
35
|
-
Your job is to turn a plan seed / Tech Spec into a Story ticket array for an AI Agent to execute.
|
|
34
|
+
return `Turn a plan seed / Tech Spec into Story tickets for an AI agent to execute. The emitted stories template (see STORY BODY SCHEMA) is the ticket shape.
|
|
36
35
|
|
|
37
36
|
### HIERARCHY RULES (v2 default-single):
|
|
38
37
|
1. **Emit exactly one Story by default.** Split into N>1 only when pieces have near-zero overlap or sit across an architectural seam. Coupled work stays one Story — put intra-session checkpoints in \`## Slicing\` and fold the Tech Spec into \`## Spec\`.
|
|
39
38
|
2. **Stories**: Specific user-facing or architectural capabilities (e.g., "Implement JWT Token Exchange").
|
|
40
|
-
- There is NO Epic parent ticket, NO Feature tier, and NO Task layer.
|
|
41
39
|
- **Story-Level Execution**: Each Story is executed end-to-end on a single branch by a single agent. Acceptance criteria and verification commands live as top-level \`acceptance[]\` / \`verify[]\` arrays on the Story ticket (see STORY BODY SCHEMA below).
|
|
42
40
|
- Thematic grouping is prose in the Story's folded \`## Spec\` / \`## Slicing\`, never sibling tickets for coupled work.
|
|
43
41
|
|
|
@@ -46,27 +44,10 @@ Your job is to turn a plan seed / Tech Spec into a Story ticket array for an AI
|
|
|
46
44
|
- \`labels[]\` is **optional**. Emit it only to request an *additional* label; persist sanitizes the list before applying it.
|
|
47
45
|
- Do **not** emit \`agent::*\` labels — lifecycle state is runtime-owned, and persist applies \`agent::ready\` itself once every checkpoint is on the ticket.
|
|
48
46
|
|
|
49
|
-
### OUTPUT FORMAT:
|
|
50
|
-
You MUST respond ONLY with a valid JSON array of objects. No prose, no markdown blocks.
|
|
51
|
-
|
|
52
|
-
### JSON SCHEMA:
|
|
53
|
-
[
|
|
54
|
-
{
|
|
55
|
-
"slug": "hyphen-case-id",
|
|
56
|
-
"type": "story",
|
|
57
|
-
"title": "Short descriptive title",
|
|
58
|
-
"body": <string — see STORY BODY SCHEMA below>,
|
|
59
|
-
"acceptance": ["<outcome a PR reviewer can confirm>", ...],
|
|
60
|
-
"verify": ["<exact command or test path>", ...],
|
|
61
|
-
"labels": ["<extra-label>"] (optional — type::story is applied automatically; omit this field unless you need an additional label),
|
|
62
|
-
"depends_on": ["slug-of-blocking-dependency"] (optional array of Story slugs that block execution)
|
|
63
|
-
}
|
|
64
|
-
]
|
|
65
|
-
|
|
66
47
|
**Slug format**: \`^[a-z0-9][a-z0-9-]*$\` — hyphen-case only. Underscores are rejected by the validator.
|
|
67
48
|
|
|
68
49
|
### STORY BODY SCHEMA (REQUIRED FOR EVERY STORY):
|
|
69
|
-
\`body\` is either the serialized markdown **string** (the section format below) or a **structured object** carrying the same fields (\`goal\`, optional \`slicing\` / \`spec\`, \`changes\`, optional \`non_goals\`) — persist parses either shape and serializes the canonical markdown itself, so you never need to read \`story-body.js\` or hand-assemble the markdown (the \`stories.template.json\` file emitted next to the plan-context envelope is a ready-to-fill structured-object skeleton).
|
|
50
|
+
\`body\` is either the serialized markdown **string** (the section format below) or a **structured object** carrying the same fields (\`goal\`, optional \`slicing\` / \`spec\`, \`changes\`, optional \`non_goals\`) — persist parses either shape and serializes the canonical markdown itself, so you never need to read \`story-body.js\` or hand-assemble the markdown (the \`stories.template.json\` file emitted next to the plan-context envelope is a ready-to-fill structured-object skeleton). The executing sub-agent is non-interactive and self-verifies from the ticket alone, so the ticket carries everything it needs.
|
|
70
51
|
|
|
71
52
|
The \`acceptance[]\` and \`verify[]\` arrays live at the **top level** of the Story ticket object — that is the machine contract the validator reads. Author each list **once, at top level**, and **omit** the \`## Acceptance\` / \`## Verify\` sections from the authored \`body\` string: persist syncs the top-level arrays into those sections so the GitHub issue stays a complete executable document. The validator resolves both fields from the top level, so an omitted section is the expected shape, not a violation.
|
|
72
53
|
|
|
@@ -153,9 +134,9 @@ ${envelopeFloor}
|
|
|
153
134
|
- A Story touching UI (\`*.tsx\`, \`*.astro\`, \`*.svelte\`, \`*.vue\`, a components folder) states the \`data-testid\` contract in \`acceptance[]\` per the testid contract in \`.agents/skills/stack/qa/playwright/SKILL.md\`.
|
|
154
135
|
- A Story touching user-visible copy, brand assets or visual style cites the relevant section of \`docs/style-guide.md\` in \`acceptance[]\` when that file exists.
|
|
155
136
|
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
137
|
+
#### ORDERING:
|
|
138
|
+
|
|
139
|
+
Express execution ordering between Stories with \`depends_on\` — the slugs of the Stories that must land first. Never emit a parent field.`;
|
|
159
140
|
}
|
|
160
141
|
|
|
161
142
|
/**
|
|
@@ -138,38 +138,29 @@ export class IssuesGateway {
|
|
|
138
138
|
|
|
139
139
|
/**
|
|
140
140
|
* Resolve an issue's container parent in one request via `Issue.parent`.
|
|
141
|
-
*
|
|
142
|
-
* return `null`, leaving the caller's checklist fallback to run.
|
|
141
|
+
* `null` means "no parent"; a degraded lookup throws after retries.
|
|
143
142
|
*
|
|
144
143
|
* @param {number} number Issue number whose parent to resolve.
|
|
145
144
|
* @returns {Promise<object|null>} Mapped parent ticket, or null.
|
|
145
|
+
* @throws {Error} When the lookup degrades.
|
|
146
146
|
* @field-manifest GraphQL Issue.parent: number, id, title, body, state,
|
|
147
147
|
* labels.nodes.name, assignees.nodes.login
|
|
148
148
|
*/
|
|
149
149
|
async getParentIssue(number) {
|
|
150
150
|
const issueNumber = Number(number);
|
|
151
151
|
if (!Number.isInteger(issueNumber) || issueNumber <= 0) return null;
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
this.
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
},
|
|
165
|
-
);
|
|
166
|
-
} catch (err) {
|
|
167
|
-
Logger.warn(
|
|
168
|
-
`[GitHubProvider] parent lookup for #${issueNumber} degraded to none ` +
|
|
169
|
-
`(${err?.message ?? err}).`,
|
|
170
|
-
);
|
|
171
|
-
return null;
|
|
172
|
-
}
|
|
152
|
+
const data = await withTransientRetry(
|
|
153
|
+
() =>
|
|
154
|
+
this.ghGraphql(
|
|
155
|
+
PARENT_ISSUE_QUERY,
|
|
156
|
+
{ owner: this.owner, repo: this.repo, number: issueNumber },
|
|
157
|
+
{ headers: { 'GraphQL-Features': 'sub_issues' } },
|
|
158
|
+
),
|
|
159
|
+
{
|
|
160
|
+
label: `getParentIssue #${issueNumber}`,
|
|
161
|
+
onRetry: defaultRetryWarn,
|
|
162
|
+
},
|
|
163
|
+
);
|
|
173
164
|
return subIssueNodeToTicket(data?.repository?.issue?.parent ?? null);
|
|
174
165
|
}
|
|
175
166
|
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
* Project `.agents/agents/` and `.agents/local/agents/` into a flat
|
|
5
5
|
* `.claude/agents/` tree — the sibling of `sync-claude-commands.js`, with the
|
|
6
6
|
* same payload-wins shadowing and orphan-reap. A role agent runs on its own
|
|
7
|
-
* system prompt (no
|
|
7
|
+
* system prompt (no entry-doc `@`-import closure), which is the point of routing to it.
|
|
8
8
|
*/
|
|
9
9
|
|
|
10
10
|
// cli-opt-out: top-level-await script with no main() function — runAsCli wraps an async main, which doesn't apply here.
|
|
@@ -12,7 +12,8 @@ the Story under audit. The shared lens machinery lives in
|
|
|
12
12
|
`{{auditOutputDir}}/audit-quality-results.md`. Each finding carries a
|
|
13
13
|
**Category:** (`Flakiness | Coverage | Performance | Mocking | Test Plans`); the
|
|
14
14
|
report adds a **Test Strategy Assessment** table (Unit / Integration / E2E /
|
|
15
|
-
Test Plans: Healthy / Needs Work / Missing
|
|
15
|
+
Test Plans / Property-Based Testing: Healthy / Needs Work / Missing, or `N/A`
|
|
16
|
+
for Property-Based Testing when no module is a candidate).
|
|
16
17
|
|
|
17
18
|
## Scope
|
|
18
19
|
|
|
@@ -130,6 +131,36 @@ Evaluate the gathered context against the following test quality dimensions:
|
|
|
130
131
|
finding here. Route the *architectural* framing of the same defect to
|
|
131
132
|
[`audit-architecture`](audit-architecture.md)'s Shipped-But-Never-Wired
|
|
132
133
|
dimension; this lens owns the **missing-test** framing.
|
|
134
|
+
8. **Property-Based Coverage — Invariants Tested Only by Examples.** Flag a
|
|
135
|
+
module whose correctness rests on an invariant but whose tests are all
|
|
136
|
+
hand-picked examples, which structurally cannot reach the inputs nobody
|
|
137
|
+
thought to pick. A module is a candidate **only** on code evidence: a
|
|
138
|
+
documented invariant or "never"/"always" claim in a header comment; an
|
|
139
|
+
explicit state machine or status/label transition table; an
|
|
140
|
+
encode/decode, parse/serialize or normalise pair (round-trip); a
|
|
141
|
+
merge/dedup/sort/scheduler function; bounded-concurrency or retry
|
|
142
|
+
coordination over async I/O; or an idempotency claim. A module with no
|
|
143
|
+
stated or implied invariant is **never** a finding. Rank candidates by
|
|
144
|
+
Step 0's churn × coverage gap and cite their `baselines/` coverage/CRAP row
|
|
145
|
+
where one exists.
|
|
146
|
+
|
|
147
|
+
- **Toolchain by ecosystem, never one library.** Detect an existing
|
|
148
|
+
property-testing library from the consumer's manifests (e.g.
|
|
149
|
+
`fast-check` for JS/TS, `hypothesis` for Python, `proptest`/`quickcheck`
|
|
150
|
+
for Rust, `jqwik` for the JVM, `rapid`/`gopter` for Go); recommend the
|
|
151
|
+
ecosystem-idiomatic one only when none is present.
|
|
152
|
+
- Severity: an async/concurrency coordinator, or a guard whose
|
|
153
|
+
invariant protects an irreversible write, tested only by examples →
|
|
154
|
+
**High**; any other invariant-bearing module with example-only tests →
|
|
155
|
+
**Medium**; toolchain absent with no High/Medium candidate → **one Low**
|
|
156
|
+
roll-up finding, not one per module.
|
|
157
|
+
- Category: file under `Coverage`. A property test whose seed is
|
|
158
|
+
neither pinned nor printed on failure goes under `Flakiness`: a red that
|
|
159
|
+
cannot be reproduced breaks the reproducibility the rubric demands.
|
|
160
|
+
- **Name the property.** Each finding states the invariant as a testable
|
|
161
|
+
property (e.g. `decode(encode(x)) === x`; "no transition leaves a
|
|
162
|
+
terminal state") plus its generator shape (the input domain to draw
|
|
163
|
+
from) — never a bare "add property tests".
|
|
133
164
|
|
|
134
165
|
## Constraint (lens-specific carve-out)
|
|
135
166
|
|
|
@@ -150,10 +181,14 @@ table:
|
|
|
150
181
|
|
|
151
182
|
## Test Strategy Assessment
|
|
152
183
|
|
|
153
|
-
| Layer
|
|
154
|
-
|
|
|
155
|
-
| Unit Testing
|
|
156
|
-
| Integration Testing
|
|
157
|
-
| E2E Testing
|
|
158
|
-
| Test Plans
|
|
184
|
+
| Layer | Status | Notes |
|
|
185
|
+
| ---------------------- | -------------------------------------- | -------------- |
|
|
186
|
+
| Unit Testing | [Healthy / Needs Work / Missing] | [Brief reason] |
|
|
187
|
+
| Integration Testing | [Healthy / Needs Work / Missing] | [Brief reason] |
|
|
188
|
+
| E2E Testing | [Healthy / Needs Work / Missing] | [Brief reason] |
|
|
189
|
+
| Test Plans | [Healthy / Needs Work / Missing] | [Brief reason] |
|
|
190
|
+
| Property-Based Testing | [Healthy / Needs Work / Missing / N/A] | [Brief reason] |
|
|
159
191
|
```
|
|
192
|
+
|
|
193
|
+
`Property-Based Testing` reads `N/A` when the repo has no candidate module
|
|
194
|
+
(dimension 8), so a repo with no invariant-bearing code is not nagged.
|
|
@@ -69,7 +69,7 @@ per-criterion, mid-delivery, and evaluates the actual work product.
|
|
|
69
69
|
> `delivery.routing.roleScopedAgents` is enabled (the **default**), use
|
|
70
70
|
> `subagent_type: acceptance-critic`: it boots on the role-scoped
|
|
71
71
|
> [`acceptance-critic`](../../agents/acceptance-critic.md) context (its own
|
|
72
|
-
> system prompt, no
|
|
72
|
+
> system prompt, no entry-doc @-closure) carrying the maker-blind
|
|
73
73
|
> invariant and the verdict schema standalone. With the kill-switch off
|
|
74
74
|
> (`roleScopedAgents: false`), fall back to
|
|
75
75
|
> `subagent_type: general-purpose`. This loop already runs inside a Story
|
|
@@ -74,8 +74,8 @@ How each tier changes the review protocol:
|
|
|
74
74
|
adversarial pass over the diff hunting for integration regressions and
|
|
75
75
|
security-relevant edges before findings are finalized.
|
|
76
76
|
|
|
77
|
-
The LLM-backed review providers (codex, security-review,
|
|
78
|
-
the resolved `depth` into the prompt/instructions they emit so the underlying
|
|
77
|
+
The LLM-backed review providers (code-review, codex, security-review,
|
|
78
|
+
ultrareview) render the resolved `depth` into the prompt/instructions they emit so the underlying
|
|
79
79
|
model actually changes thoroughness. The native provider deliberately ignores
|
|
80
80
|
`depth` — its mechanical lint + maintainability sweep already scales with diff
|
|
81
81
|
size, and there is no "review harder" knob a deterministic scorer can turn (its
|
|
@@ -110,44 +110,21 @@ The pipeline will:
|
|
|
110
110
|
- Run a focused lint check on the change set.
|
|
111
111
|
- Post a structured summary report to the `[TICKET_ID]` issue.
|
|
112
112
|
|
|
113
|
-
###
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
`git diff --name-only`).
|
|
122
|
-
2. Selects the **local-tier** lenses that own a concern decidable from a single
|
|
123
|
-
Story's diff — `resolveLensTier(lens) === 'local'` **plus** the pure
|
|
124
|
-
`matchesAnyFilePattern` matcher against the diff (the audit-suite SDK's
|
|
125
|
-
[`selectLocalLenses`](../../scripts/lib/audit-suite/selector.js)). This is
|
|
126
|
-
deliberately **not** `selectAudits`: `selectAudits` unions in keyword and
|
|
127
|
-
gate matches and has no per-tier gate, so it would widen the roster past the
|
|
128
|
-
footprint-matched local set this tier owns.
|
|
129
|
-
3. Materializes the matched roster at **`light`** depth
|
|
130
|
-
(`STORY_SCOPE_LENS_DEPTH`) via `runAuditSuite`, surfacing the outcome on the
|
|
131
|
-
review envelope's `localLensReview` field.
|
|
132
|
-
|
|
133
|
-
A diff that matches no local lens adds **no** lens work (the roster is empty and
|
|
134
|
-
`runAuditSuite` is never invoked). The pass is advisory and best-effort: a git
|
|
135
|
-
or materialization failure degrades to a skipped envelope and never blocks the
|
|
136
|
-
close.
|
|
137
|
-
|
|
138
|
-
The live close entry point —
|
|
139
|
-
[`runStoryScopeReview`](../../scripts/lib/orchestration/single-story-close/phases/code-review.js)
|
|
140
|
-
— reaches this pass through the shared `runStoryReviewCore` spine. Because
|
|
141
|
-
the pass lives inside the close subprocess (invoked after the delivering
|
|
142
|
-
child exits), it honors the maker-blind invariant above: a maker never runs
|
|
143
|
-
its own local-lens review.
|
|
113
|
+
### Story scope runs no lens pass
|
|
114
|
+
|
|
115
|
+
Close runs **no** audit-lens pass of its own (Story #5416 retired it: it
|
|
116
|
+
materialized prompt files no workflow read, then armed auto-merge without
|
|
117
|
+
waiting). The Story-scope review is this pipeline plus CI. Local-tier lens
|
|
118
|
+
concerns are covered shift-left by the write-time authoring checklists
|
|
119
|
+
threaded into the Story prompt; the on-demand `/audit-*` workflows remain the
|
|
120
|
+
way to run a full lens over a change.
|
|
144
121
|
|
|
145
122
|
## Step 2 — Review Pillars
|
|
146
123
|
|
|
147
124
|
For each changed file, execute a strict review against four pillars. The
|
|
148
125
|
second pillar (**Integration Review**) deliberately defers the security /
|
|
149
126
|
performance / quality / coverage sweeps to the change-set-scoped lenses —
|
|
150
|
-
those
|
|
127
|
+
those are covered shift-left by the write-time lens checklists.
|
|
151
128
|
Re-walking those sweeps a second time in this pillar is duplication, not
|
|
152
129
|
defense-in-depth.
|
|
153
130
|
|
|
@@ -173,9 +150,9 @@ Does the implementation match the Story's acceptance criteria and folded Spec?
|
|
|
173
150
|
|
|
174
151
|
The diff under review is `baseRef..headRef`
|
|
175
152
|
(`main..story-<storyId>`, or the configured base branch to the Story
|
|
176
|
-
branch). The
|
|
177
|
-
|
|
178
|
-
|
|
153
|
+
branch). The write-time lens checklists have already covered the local-tier
|
|
154
|
+
concerns. Pillar findings land in the single `verification-results` comment
|
|
155
|
+
this pass posts. The
|
|
179
156
|
integration view here focuses on cross-cutting ripple within the Story and
|
|
180
157
|
contract drift against the base branch. Look for:
|
|
181
158
|
|
|
@@ -260,7 +237,7 @@ prior baseline before merging.
|
|
|
260
237
|
|
|
261
238
|
Findings are **persisted as a `verification-results` structured comment on
|
|
262
239
|
the `[TICKET_ID]` issue** by `runCodeReview` (the unified findings contract —
|
|
263
|
-
this single comment carries the Story-scope
|
|
240
|
+
this single comment carries the Story-scope review findings). The target
|
|
264
241
|
ticket is the Story. The comment
|
|
265
242
|
is idempotent — re-runs replace the prior one — and its body includes
|
|
266
243
|
severity-tier counts plus the full findings list so downstream workflows
|
|
@@ -175,7 +175,7 @@ into batches of `cap` and dispatch each batch in its own turn.
|
|
|
175
175
|
exposes agent dispatch, spawn each ready Story as its own
|
|
176
176
|
`subagent_type: story-worker` sub-agent — it boots on the role-scoped
|
|
177
177
|
[`story-worker`](../../agents/story-worker.md) context (its own system prompt, no
|
|
178
|
-
|
|
178
|
+
entry-doc @-closure) carrying the load-bearing delivery MUSTs standalone. The
|
|
179
179
|
sub-agent executes [`deliver-story.md`](deliver-story.md) Steps 0–2.5
|
|
180
180
|
(init → implement → acceptance self-eval → **push**) and stops there; **you**
|
|
181
181
|
own Step 3, serialized — see `/mandrel-deliver` § Closing what the workers hand back.
|
|
@@ -291,7 +291,9 @@ confirm instead (`captureStoryFollowUps`).
|
|
|
291
291
|
reopen the issue.
|
|
292
292
|
- The parent lookup resolves the native parent edge in **one** call
|
|
293
293
|
(`getParentIssue`), falling back to a `type::epic` scan for a child linked by
|
|
294
|
-
checklist alone.
|
|
294
|
+
checklist alone. An authoritative "no parent" narrows that scan to Epics
|
|
295
|
+
whose body checklist names the Story (no per-Epic native read); only a
|
|
296
|
+
degraded lookup reads every scanned Epic's native children. Children are the body checklist **union** the native
|
|
295
297
|
sub-issue edges — the same reader `/mandrel-deliver`'s expansion uses.
|
|
296
298
|
- A checklist row citing an id that resolves to nothing is **dropped with a
|
|
297
299
|
warning** when the native read succeeded; an unresolvable *native* edge
|
|
@@ -72,6 +72,9 @@ One branch, one PR to `main`, commits against the inline `acceptance[]` /
|
|
|
72
72
|
1. Read the Story body; its acceptance criteria are the contract. Docs are
|
|
73
73
|
digest-first; read a caller-provided `checklistPath` first, and walk any
|
|
74
74
|
`## Slicing` rows as **intra-session checkpoints** (reference § Step 1).
|
|
75
|
+
After a context summary, re-derive progress from `git log` on
|
|
76
|
+
`story-<id>` against the `## Slicing` rows (each checkpoint is a commit
|
|
77
|
+
boundary) before continuing.
|
|
75
78
|
2. Implement and commit on the Story branch, iterating with quick advisory
|
|
76
79
|
gates (`typecheck`, `lint`, scoped tests) — the full chain runs in Step 3,
|
|
77
80
|
and the **one** full-suite run at Step 2.5.
|