mandrel 2.55.0 → 2.57.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/agents/plan-critic.md +13 -18
- package/.agents/agents/story-worker.md +25 -34
- package/.agents/docs/agentrc-reference.json +4 -30
- package/.agents/docs/configuration.md +11 -28
- package/.agents/docs/execution-reference.md +5 -5
- package/.agents/docs/quality-gates.md +8 -7
- package/.agents/instructions.md +9 -10
- package/.agents/rules/ci-remediation.md +39 -21
- package/.agents/schemas/agentrc.schema.json +28 -185
- package/.agents/schemas/story-deliver-terminal.schema.json +1 -1
- package/.agents/scripts/acceptance-eval.js +107 -17
- package/.agents/scripts/audit-to-stories.js +222 -75
- package/.agents/scripts/ceremony-derive.js +191 -0
- package/.agents/scripts/check-context-budget.js +28 -33
- package/.agents/scripts/check-cyclomatic.js +4 -3
- package/.agents/scripts/deliver-light.js +31 -94
- package/.agents/scripts/file-ci-gap.js +306 -0
- package/.agents/scripts/lib/audit-suite/checklist-threading.js +15 -2
- package/.agents/scripts/lib/audit-to-stories/audit-label-taxonomy.js +25 -1
- package/.agents/scripts/lib/audit-to-stories/dedupe-against-github.js +40 -52
- package/.agents/scripts/lib/audit-to-stories/finding-adapter.js +5 -1
- package/.agents/scripts/lib/audit-to-stories/issue-corpus.js +162 -0
- package/.agents/scripts/lib/audit-to-stories/issues-file.js +121 -0
- package/.agents/scripts/lib/audit-to-stories/ledger-commit.js +1 -1
- package/.agents/scripts/lib/audit-to-stories/ledger-record.js +126 -0
- package/.agents/scripts/lib/audit-to-stories/seed-from-findings.js +11 -0
- package/.agents/scripts/lib/baselines/coverage-updater-cli.js +110 -0
- package/.agents/scripts/lib/baselines/crap-preview-scan.js +25 -0
- package/.agents/scripts/lib/baselines/crap-updater-cli.js +223 -0
- package/.agents/scripts/lib/bdd-scenario-budget.js +21 -3
- package/.agents/scripts/lib/bootstrap/quality-bootstrap.js +0 -1
- package/.agents/scripts/lib/close-validation/gates.js +52 -1
- package/.agents/scripts/lib/config/acceptance-eval.js +25 -57
- package/.agents/scripts/lib/config/delivery-routing.js +7 -33
- package/.agents/scripts/lib/config/explain.js +0 -19
- package/.agents/scripts/lib/config/limits.js +18 -78
- package/.agents/scripts/lib/config/quality.js +6 -3
- package/.agents/scripts/lib/config/runners.js +3 -2
- package/.agents/scripts/lib/config-settings-schema-delivery.js +15 -68
- package/.agents/scripts/lib/config-settings-schema-quality.js +0 -14
- package/.agents/scripts/lib/config-settings-schema.js +49 -143
- package/.agents/scripts/lib/crap-engine.js +35 -4
- package/.agents/scripts/lib/crap-utils.js +17 -1
- package/.agents/scripts/lib/cyclomatic-ceiling.js +19 -7
- package/.agents/scripts/lib/feedback-loop/graduator-core.js +53 -13
- package/.agents/scripts/lib/feedback-loop/prior-feedback-fetcher.js +71 -25
- package/.agents/scripts/lib/feedback-loop/retro-proposals-graduator.js +18 -25
- package/.agents/scripts/lib/{audit-to-stories/ledger.js → findings/audit-ledger.js} +131 -24
- package/.agents/scripts/lib/findings/route-finding.js +38 -0
- package/.agents/scripts/lib/generated/agentrc-validator.js +1 -1
- package/.agents/scripts/lib/github/framework-repo.js +148 -2
- package/.agents/scripts/lib/label-constants.js +6 -1
- package/.agents/scripts/lib/observability/runtime-friction.js +1 -1
- package/.agents/scripts/lib/observability/source-classifier.js +2 -0
- package/.agents/scripts/lib/orchestration/acceptance-eval-decision.js +5 -4
- package/.agents/scripts/lib/orchestration/ceremony-routing.js +19 -73
- package/.agents/scripts/lib/orchestration/ci-gap-intake.js +605 -0
- package/.agents/scripts/lib/orchestration/ci-rerun-guard.js +13 -8
- package/.agents/scripts/lib/orchestration/complexity-gate.js +46 -212
- package/.agents/scripts/lib/orchestration/file-assumptions.js +32 -17
- package/.agents/scripts/lib/orchestration/light-escalation.js +3 -3
- package/.agents/scripts/lib/orchestration/light-suitability.js +66 -233
- package/.agents/scripts/lib/orchestration/plan-context.js +181 -387
- package/.agents/scripts/lib/orchestration/plan-critic-conditions.js +42 -153
- package/.agents/scripts/lib/orchestration/plan-critics-evaluate.js +14 -70
- package/.agents/scripts/lib/orchestration/plan-persist/audit-provenance.js +197 -0
- package/.agents/scripts/lib/orchestration/plan-persist/changes-repair.js +300 -0
- package/.agents/scripts/lib/orchestration/plan-persist/persist-helpers.js +131 -168
- package/.agents/scripts/lib/orchestration/plan-persist/run-plan-persist.js +133 -299
- package/.agents/scripts/lib/orchestration/plan-persist/soft-findings.js +55 -0
- package/.agents/scripts/lib/orchestration/plan-persist/story-ops.js +16 -65
- package/.agents/scripts/lib/orchestration/plan-persist/wave-serialisation.js +22 -35
- package/.agents/scripts/lib/orchestration/plan-text-hygiene.js +30 -139
- package/.agents/scripts/lib/orchestration/planning/memory-pool-advisory.js +61 -223
- package/.agents/scripts/lib/orchestration/run-epilogue.js +4 -4
- package/.agents/scripts/lib/orchestration/single-story-close/phases/close-validation.js +5 -0
- package/.agents/scripts/lib/orchestration/single-story-close/phases/pre-gate-steps.js +46 -16
- package/.agents/scripts/lib/orchestration/story-close/context-budget-writeback.js +213 -0
- package/.agents/scripts/lib/orchestration/story-follow-ups.js +32 -20
- package/.agents/scripts/lib/orchestration/task-body-validator.js +10 -63
- package/.agents/scripts/lib/orchestration/ticket-validator-conflicts.js +33 -539
- package/.agents/scripts/lib/orchestration/ticket-validator-sizing.js +21 -414
- package/.agents/scripts/lib/orchestration/ticket-validator.js +54 -118
- package/.agents/scripts/lib/orchestration/verify-credit.js +69 -24
- package/.agents/scripts/lib/story-body/body-format-lints.js +15 -85
- package/.agents/scripts/lib/story-body/story-body.js +17 -237
- package/.agents/scripts/lib/templates/decomposer-prompts.js +84 -121
- package/.agents/scripts/lib/test-isolate/cli-options.js +93 -0
- package/.agents/scripts/lib/test-isolate/progress-log.js +45 -0
- package/.agents/scripts/lib/test-isolate/render-report.js +97 -0
- package/.agents/scripts/lib/test-isolate/run-isolate.js +87 -0
- package/.agents/scripts/lib/test-run-credit.js +266 -0
- package/.agents/scripts/lib/wave-runner/footprint.js +48 -358
- package/.agents/scripts/lib/wave-runner/ready-set.js +6 -5
- package/.agents/scripts/lib/workers/crap-worker.js +32 -41
- package/.agents/scripts/plan-context.js +7 -9
- package/.agents/scripts/plan-critics.js +28 -54
- package/.agents/scripts/plan-persist.js +25 -68
- package/.agents/scripts/pr-watch-with-update.js +3 -2
- package/.agents/scripts/quality-preview.js +51 -0
- package/.agents/scripts/run-tests.js +12 -0
- package/.agents/scripts/stories-wave-tick.js +23 -45
- package/.agents/scripts/test-isolate.js +13 -180
- package/.agents/scripts/update-coverage-baseline.js +25 -70
- package/.agents/scripts/update-crap-baseline.js +19 -123
- package/.agents/skills/core/scope-triage/SKILL.md +3 -3
- package/.agents/workflows/audit-clean-code.md +4 -3
- package/.agents/workflows/audit-to-stories.md +63 -27
- package/.agents/workflows/helpers/acceptance-self-eval.md +41 -41
- package/.agents/workflows/helpers/code-quality-guardrails.md +4 -4
- package/.agents/workflows/helpers/code-review.md +2 -3
- package/.agents/workflows/helpers/deliver-digest.md +41 -57
- package/.agents/workflows/helpers/deliver-light.md +40 -105
- package/.agents/workflows/helpers/deliver-reference.md +1 -1
- package/.agents/workflows/helpers/deliver-story-reference.md +56 -62
- package/.agents/workflows/helpers/deliver-story.md +9 -13
- package/.agents/workflows/helpers/plan-reference.md +132 -196
- package/.agents/workflows/mandrel-plan.md +28 -41
- package/.agents/workflows/memory-consolidate.md +9 -13
- package/docs/CHANGELOG.md +33 -0
- package/lib/migrations/index.js +4 -0
- package/lib/migrations/steps/2.57.0-retire-delivery-limit-knobs.js +45 -0
- package/lib/migrations/steps/2.57.0-retire-planning-limit-knobs.js +59 -0
- package/package.json +1 -1
- package/.agents/scripts/lib/framework-version.js +0 -39
- package/.agents/scripts/lib/orchestration/consolidation-precondition.js +0 -223
- package/.agents/scripts/lib/orchestration/plan-persist/fan-out-gate.js +0 -97
- package/.agents/scripts/lib/orchestration/planning/decomposer-context.js +0 -26
- package/.agents/scripts/lib/orchestration/spec-budget.js +0 -89
- package/.agents/scripts/lib/orchestration/spec-spill.js +0 -74
- package/.agents/scripts/lib/orchestration/verify-tier-repair.js +0 -107
|
@@ -169,6 +169,25 @@
|
|
|
169
169
|
"description": "Default `timeoutMs` applied to every `gh` subprocess the provider facade spawns, so a stalled socket or long-poll cannot hang an orchestration indefinitely. A `GhExecTimeoutError` from a hit ceiling is classified `transient` and retried by `withTransientRetry`. Story #2860.",
|
|
170
170
|
"default": 60000
|
|
171
171
|
},
|
|
172
|
+
"followUpRepos": {
|
|
173
|
+
"type": "object",
|
|
174
|
+
"description": "Repository slugs for the non-consumer follow-up ownership buckets, used when a CI gap, retro proposal, or audit finding belongs to someone other than the repo that surfaced it.",
|
|
175
|
+
"properties": {
|
|
176
|
+
"framework": {
|
|
177
|
+
"type": "string",
|
|
178
|
+
"pattern": "^[^/\\s]+/[^/\\s]+$",
|
|
179
|
+
"description": "`<owner>/<repo>` that owns framework-level defects. Defaults to the Mandrel mirror — the one bucket with a knowable default.",
|
|
180
|
+
"default": "dsj1984/mandrel"
|
|
181
|
+
},
|
|
182
|
+
"platform": {
|
|
183
|
+
"type": ["string", "null"],
|
|
184
|
+
"pattern": "^[^/\\s]+/[^/\\s]+$",
|
|
185
|
+
"description": "`<owner>/<repo>` for a shared platform or infrastructure tracker (a shared base config, a runner fleet, a cross-repo toolchain). No default — nothing can guess a shared repo. Left unset, platform-owned findings file locally and say so.",
|
|
186
|
+
"default": null
|
|
187
|
+
}
|
|
188
|
+
},
|
|
189
|
+
"additionalProperties": false
|
|
190
|
+
},
|
|
172
191
|
"branchProtection": {
|
|
173
192
|
"type": "object",
|
|
174
193
|
"description": "Branch-protection stance applied to `project.baseBranch` by the GitHub bootstrap, and reproduced locally before every push.",
|
|
@@ -308,145 +327,21 @@
|
|
|
308
327
|
},
|
|
309
328
|
"planning": {
|
|
310
329
|
"type": "object",
|
|
311
|
-
"description": "Inputs to `/mandrel-plan`:
|
|
330
|
+
"description": "Inputs to `/mandrel-plan`: the memory-hygiene advisory ceiling and the opt-in navigability reachability gate.",
|
|
312
331
|
"properties": {
|
|
313
|
-
"riskHeuristics": {
|
|
314
|
-
"oneOf": [
|
|
315
|
-
{
|
|
316
|
-
"type": "array",
|
|
317
|
-
"items": {
|
|
318
|
-
"type": "string"
|
|
319
|
-
}
|
|
320
|
-
},
|
|
321
|
-
{
|
|
322
|
-
"type": "object",
|
|
323
|
-
"properties": {
|
|
324
|
-
"append": {
|
|
325
|
-
"type": "array",
|
|
326
|
-
"items": {
|
|
327
|
-
"type": "string"
|
|
328
|
-
}
|
|
329
|
-
},
|
|
330
|
-
"prepend": {
|
|
331
|
-
"type": "array",
|
|
332
|
-
"items": {
|
|
333
|
-
"type": "string"
|
|
334
|
-
}
|
|
335
|
-
}
|
|
336
|
-
},
|
|
337
|
-
"additionalProperties": false
|
|
338
|
-
}
|
|
339
|
-
],
|
|
340
|
-
"description": "Prose heuristics the planner escalates a Story against. A plain array replaces the framework list; the `{ append, prepend }` extender form deep-merges with it.",
|
|
341
|
-
"default": [
|
|
342
|
-
"Destructive or irreversible data mutations (dropping tables, deleting rows without soft-delete or backup, truncating production state).",
|
|
343
|
-
"Modifications to shared security or auth infrastructure (IAM policies, auth middleware, session or token handling, secret rotation).",
|
|
344
|
-
"Changes to CI/CD, deployment pipelines, or release gating that could disable safety checks or ship unverified code to production.",
|
|
345
|
-
"Monorepo-wide AST or text replacements touching overlapping files in parallel (catastrophic merge-conflict risk across concurrent agents).",
|
|
346
|
-
"Schema migrations that rewrite existing rows or drop columns without a backfill or rollback plan."
|
|
347
|
-
]
|
|
348
|
-
},
|
|
349
|
-
"complexityGate": {
|
|
350
|
-
"type": "object",
|
|
351
|
-
"description": "Shape-derived ceremony-lite complexity routing. A lite claim is validated against the authored Story shape at persist and re-derived from the Story body at dispatch; conservative (full on any doubt). Never relaxes the Story-ticket / PR-to-main / repo-gates / security-baseline non-negotiables.",
|
|
352
|
-
"properties": {
|
|
353
|
-
"enabled": {
|
|
354
|
-
"type": "boolean",
|
|
355
|
-
"description": "Master switch. When false, lite routing is disabled everywhere: persist refuses lite claims and dispatch always takes the sub-agent path. Default true."
|
|
356
|
-
},
|
|
357
|
-
"maxArtifacts": {
|
|
358
|
-
"type": "integer",
|
|
359
|
-
"minimum": 0,
|
|
360
|
-
"description": "Enumerated-artifact threshold reported by the plan-context complexity signals. An input signal for the planner verdict — carries no routing authority. Default 1."
|
|
361
|
-
}
|
|
362
|
-
},
|
|
363
|
-
"additionalProperties": false
|
|
364
|
-
},
|
|
365
332
|
"memoryPool": {
|
|
366
333
|
"type": "object",
|
|
367
|
-
"description": "
|
|
334
|
+
"description": "Threshold for the memory-hygiene advisory `/mandrel-plan` surfaces at Gate #1. Advisory only: it recommends `/memory-consolidate` and never gates, reroutes, or mutates the memory pool.",
|
|
368
335
|
"properties": {
|
|
369
|
-
"staleAfterDays": {
|
|
370
|
-
"type": "integer",
|
|
371
|
-
"minimum": 1,
|
|
372
|
-
"description": "Recommend a consolidation pass once the pool's stamp is older than this many days. Default 30.",
|
|
373
|
-
"default": 30
|
|
374
|
-
},
|
|
375
|
-
"growthDelta": {
|
|
376
|
-
"type": "integer",
|
|
377
|
-
"minimum": 1,
|
|
378
|
-
"description": "Recommend a consolidation pass once this many entries have been written since the last one. Measured against the entry count the last pass stamped, so a stamp predating that field leaves growth unmeasured and only the age threshold applies. Default 25.",
|
|
379
|
-
"default": 25
|
|
380
|
-
},
|
|
381
336
|
"indexByteCeiling": {
|
|
382
337
|
"type": "integer",
|
|
383
338
|
"minimum": 1,
|
|
384
|
-
"description": "Recommend a consolidation pass once the pool's `MEMORY.md` index exceeds this many bytes.
|
|
339
|
+
"description": "Recommend a consolidation pass once the pool's `MEMORY.md` index exceeds this many bytes. The harness truncates the index it loads into each session at its own byte cap, so an oversized index is a loss already happening — every entry listed after the cut is invisible — rather than a hygiene forecast. Default 24576, the harness cap itself.",
|
|
385
340
|
"default": 24576
|
|
386
341
|
}
|
|
387
342
|
},
|
|
388
343
|
"additionalProperties": false
|
|
389
344
|
},
|
|
390
|
-
"failOnSharedEditors": {
|
|
391
|
-
"type": "boolean",
|
|
392
|
-
"description": "When true, upgrade shared-editor conflict findings to hard errors (default false — advisory soft findings only).",
|
|
393
|
-
"default": false
|
|
394
|
-
},
|
|
395
|
-
"requireExplicitCrossStoryDeps": {
|
|
396
|
-
"type": "boolean",
|
|
397
|
-
"description": "When true, upgrade implicit cross-Story dependency findings to hard errors (default false — advisory soft findings only).",
|
|
398
|
-
"default": false
|
|
399
|
-
},
|
|
400
|
-
"crossCuttingRegistries": {
|
|
401
|
-
"oneOf": [
|
|
402
|
-
{
|
|
403
|
-
"type": "array",
|
|
404
|
-
"items": {
|
|
405
|
-
"type": "string"
|
|
406
|
-
}
|
|
407
|
-
},
|
|
408
|
-
{
|
|
409
|
-
"type": "object",
|
|
410
|
-
"properties": {
|
|
411
|
-
"append": {
|
|
412
|
-
"type": "array",
|
|
413
|
-
"items": {
|
|
414
|
-
"type": "string"
|
|
415
|
-
}
|
|
416
|
-
},
|
|
417
|
-
"prepend": {
|
|
418
|
-
"type": "array",
|
|
419
|
-
"items": {
|
|
420
|
-
"type": "string"
|
|
421
|
-
}
|
|
422
|
-
}
|
|
423
|
-
},
|
|
424
|
-
"additionalProperties": false
|
|
425
|
-
}
|
|
426
|
-
],
|
|
427
|
-
"description": "Registry path patterns whose concurrent edits across Stories are flagged as conflicts. Defaults to the framework listener/handler index patterns when omitted.",
|
|
428
|
-
"default": [
|
|
429
|
-
"lib/orchestration/lifecycle/listeners/index.js",
|
|
430
|
-
"**/listeners/index.js",
|
|
431
|
-
"**/handlers/index.js"
|
|
432
|
-
]
|
|
433
|
-
},
|
|
434
|
-
"failOnRegistryConflicts": {
|
|
435
|
-
"type": "boolean",
|
|
436
|
-
"description": "When true, upgrade cross-cutting registry conflict findings to hard errors (default false).",
|
|
437
|
-
"default": false
|
|
438
|
-
},
|
|
439
|
-
"failOnLargeFanOut": {
|
|
440
|
-
"type": "boolean",
|
|
441
|
-
"description": "When true, upgrade fan-out-warning findings (delete blast radius) to hard errors (default false — soft advisory).",
|
|
442
|
-
"default": false
|
|
443
|
-
},
|
|
444
|
-
"largeFanOutThreshold": {
|
|
445
|
-
"type": "integer",
|
|
446
|
-
"minimum": 0,
|
|
447
|
-
"description": "Call-site count above which a Story that deletes a module emits a fan-out-warning. Counts base-branch references to the deleted path basename. Soft by default; does not size or reject Stories. Default 10.",
|
|
448
|
-
"default": 10
|
|
449
|
-
},
|
|
450
345
|
"navigation": {
|
|
451
346
|
"type": "object",
|
|
452
347
|
"description": "Opt-in navigability reachability gate. Absent or empty routeGlobs is a silent no-op.",
|
|
@@ -475,7 +370,7 @@
|
|
|
475
370
|
},
|
|
476
371
|
"delivery": {
|
|
477
372
|
"type": "object",
|
|
478
|
-
"description": "Everything `/mandrel-deliver` and `single-story-close` consume: execution timeouts, worktree isolation, runner concurrency, docs freshness,
|
|
373
|
+
"description": "Everything `/mandrel-deliver` and `single-story-close` consume: execution timeouts, worktree isolation, runner concurrency, docs freshness, quality gates, merge/CI watch, review ceremony, and the feedback loop.",
|
|
479
374
|
"properties": {
|
|
480
375
|
"execution": {
|
|
481
376
|
"type": "object",
|
|
@@ -652,39 +547,6 @@
|
|
|
652
547
|
}
|
|
653
548
|
]
|
|
654
549
|
},
|
|
655
|
-
"signals": {
|
|
656
|
-
"type": "object",
|
|
657
|
-
"description": "Detector thresholds for the surviving performance-signal categories. Each block is shallow-merged by the resolver.",
|
|
658
|
-
"properties": {
|
|
659
|
-
"rework": {
|
|
660
|
-
"type": "object",
|
|
661
|
-
"description": "Rework detector — repeated edits to one file in a run.",
|
|
662
|
-
"properties": {
|
|
663
|
-
"editsPerFile": {
|
|
664
|
-
"type": "integer",
|
|
665
|
-
"minimum": 1,
|
|
666
|
-
"description": "Edits to a single file within one run that trip the rework signal.",
|
|
667
|
-
"default": 5
|
|
668
|
-
}
|
|
669
|
-
},
|
|
670
|
-
"additionalProperties": false
|
|
671
|
-
},
|
|
672
|
-
"retry": {
|
|
673
|
-
"type": "object",
|
|
674
|
-
"description": "Retry detector — the same command failing repeatedly.",
|
|
675
|
-
"properties": {
|
|
676
|
-
"repeatCount": {
|
|
677
|
-
"type": "integer",
|
|
678
|
-
"minimum": 1,
|
|
679
|
-
"description": "Repeats of an identical failing command that trip the retry signal.",
|
|
680
|
-
"default": 3
|
|
681
|
-
}
|
|
682
|
-
},
|
|
683
|
-
"additionalProperties": false
|
|
684
|
-
}
|
|
685
|
-
},
|
|
686
|
-
"additionalProperties": false
|
|
687
|
-
},
|
|
688
550
|
"quality": {
|
|
689
551
|
"type": "object",
|
|
690
552
|
"description": "Quality-gate configuration. Every gate lives under `gates.<tier>` and shares the same `{ enabled, baselinePath, tolerance, floors, components }` base; shared scoping lives at this block root.",
|
|
@@ -1591,12 +1453,6 @@
|
|
|
1591
1453
|
"description": "Cyclomatic complexity at which a new or changed method is flagged for a refactor look.",
|
|
1592
1454
|
"default": 8
|
|
1593
1455
|
},
|
|
1594
|
-
"cyclomaticMustFix": {
|
|
1595
|
-
"type": "integer",
|
|
1596
|
-
"minimum": 1,
|
|
1597
|
-
"description": "Cyclomatic complexity at which a new or changed method must be decomposed before the diff closes.",
|
|
1598
|
-
"default": 12
|
|
1599
|
-
},
|
|
1600
1456
|
"requireSiblingTest": {
|
|
1601
1457
|
"type": "boolean",
|
|
1602
1458
|
"description": "When true, a new source file with no colocated sibling test is reported by the guardrails pass.",
|
|
@@ -1838,12 +1694,6 @@
|
|
|
1838
1694
|
"description": "Maximum auto-fix retry attempts per finding in /mandrel-deliver Phase 5 (code-review). 0 disables auto-fix. Default 3.",
|
|
1839
1695
|
"default": 3
|
|
1840
1696
|
},
|
|
1841
|
-
"maxFixScopeFiles": {
|
|
1842
|
-
"type": "integer",
|
|
1843
|
-
"minimum": 1,
|
|
1844
|
-
"description": "Maximum file count a single auto-fix may modify before escalating to agent::blocked. Default 5.",
|
|
1845
|
-
"default": 5
|
|
1846
|
-
},
|
|
1847
1697
|
"autoFixSeverity": {
|
|
1848
1698
|
"type": "string",
|
|
1849
1699
|
"enum": ["high", "medium"],
|
|
@@ -1879,12 +1729,12 @@
|
|
|
1879
1729
|
},
|
|
1880
1730
|
"acceptanceEval": {
|
|
1881
1731
|
"type": "object",
|
|
1882
|
-
"description": "Story #3819. Bounded per-Story acceptance self-eval loop. After the implementation commits land and before the Story-implementation phase flips to `closing`, an independent (fresh-context) critic pass scores the caller-injected change set against each inline `acceptance[]` item, redrafts the unmet items, and re-evaluates — capped at `maxRounds` redraft rounds, then escalates to `agent::blocked` when criteria remain unmet. There is no `enabled` flag: the
|
|
1732
|
+
"description": "Story #3819. Bounded per-Story acceptance self-eval loop. After the implementation commits land and before the Story-implementation phase flips to `closing`, an independent (fresh-context) critic pass scores the caller-injected change set against each inline `acceptance[]` item, redrafts the unmet items, and re-evaluates — capped at `maxRounds` redraft rounds (0 = scored once, no redraft), then escalates to `agent::blocked` when criteria remain unmet. There is no `enabled` flag: the scoring pass is a hard cutover (always on).",
|
|
1883
1733
|
"properties": {
|
|
1884
1734
|
"maxRounds": {
|
|
1885
1735
|
"type": "integer",
|
|
1886
|
-
"minimum":
|
|
1887
|
-
"description": "Maximum number of redraft rounds before escalation. Default 2;
|
|
1736
|
+
"minimum": 0,
|
|
1737
|
+
"description": "Maximum number of redraft rounds before escalation. Default 2; 0 means the verdict is scored once with no redraft round (Story #5313 dropped the hard ceiling and the floor-of-one clamp).",
|
|
1888
1738
|
"default": 2
|
|
1889
1739
|
}
|
|
1890
1740
|
},
|
|
@@ -1993,24 +1843,17 @@
|
|
|
1993
1843
|
},
|
|
1994
1844
|
"routing": {
|
|
1995
1845
|
"type": "object",
|
|
1996
|
-
"description": "v2 delivery-spawn routing: role-scoped boot contexts and
|
|
1846
|
+
"description": "v2 delivery-spawn routing: role-scoped boot contexts and the ceremony profile. The v1 singleDelivery epic-route kill-switch was removed in Stage 6; the freshCriticSampleRate sampling floor was retired in Story #5313.",
|
|
1997
1847
|
"properties": {
|
|
1998
1848
|
"roleScopedAgents": {
|
|
1999
1849
|
"type": "boolean",
|
|
2000
1850
|
"description": "Epic #4478 (M7-B). Kill-switch for the role-scoped boot contexts. When true (default), a converted delivery spawn (`story-worker`, `acceptance-critic`) boots on its own `.claude/agents/<role>.md` system prompt instead of re-paying the full CLAUDE.md @-import closure. When false, every converted spawn falls back to `subagent_type: general-purpose` — the instant, code-rollback-free per-consumer revert, and the universal escape for hosts that ignore `.claude/agents/`. The fallback is the full-closure agent that ran before M7-B, so flipping it off never drops a gate.",
|
|
2001
1851
|
"default": true
|
|
2002
1852
|
},
|
|
2003
|
-
"freshCriticSampleRate": {
|
|
2004
|
-
"type": "number",
|
|
2005
|
-
"minimum": 0,
|
|
2006
|
-
"maximum": 1,
|
|
2007
|
-
"description": "Epic #4478 (M7-B, Part 2). Maker-checker sampling floor. Under the standard profile, a change set touching no sensitive path routes its acceptance clusters down the contract-identical inline critic path, but this fraction of them is still forced through a fresh-context critic so a low derived level never means zero independent checking. Clamped to [0, 1]; 0 disables the floor, 1 forces every cluster fresh. Consumed by resolveCeremonyForRisk (lib/orchestration/ceremony-routing.js).",
|
|
2008
|
-
"default": 0.2
|
|
2009
|
-
},
|
|
2010
1853
|
"ceremonyProfile": {
|
|
2011
1854
|
"type": "string",
|
|
2012
1855
|
"enum": ["minimal", "standard", "strict"],
|
|
2013
|
-
"description": "Acceptance-ceremony depth. minimal = always inline critic; strict = always fresh-context critic; standard (default) = routed off the change level derived from the Story diff
|
|
1856
|
+
"description": "Acceptance-ceremony depth. minimal = always inline critic; strict = always fresh-context critic; standard (default) = routed off the change level derived from the Story diff: high or underivable → fresh, low → inline.",
|
|
2014
1857
|
"default": "standard"
|
|
2015
1858
|
},
|
|
2016
1859
|
"closeAndLand": {
|
|
@@ -161,7 +161,7 @@
|
|
|
161
161
|
"type": "array",
|
|
162
162
|
"minItems": 1,
|
|
163
163
|
"items": { "type": "string", "minLength": 1 },
|
|
164
|
-
"description": "The gate's reasons verbatim — the same strings the
|
|
164
|
+
"description": "The gate's reasons verbatim — the same strings the gate logs, so an escalation explains itself identically on every surface."
|
|
165
165
|
},
|
|
166
166
|
"created": {
|
|
167
167
|
"type": "object",
|
|
@@ -17,8 +17,8 @@
|
|
|
17
17
|
* 1. Validate the verdict file against the verdict JSON Schema (a
|
|
18
18
|
* malformed verdict is a hard error — the loop refuses to guess).
|
|
19
19
|
* 2. Decide `proceed | redraft | block` from the per-criterion verdicts
|
|
20
|
-
* and the resolved
|
|
21
|
-
*
|
|
20
|
+
* and the resolved redraft budget (`delivery.acceptanceEval.maxRounds`;
|
|
21
|
+
* `0` means the verdict is scored once with no redraft round).
|
|
22
22
|
* 3. Emit one per-criterion `acceptance-eval` signal into the retro /
|
|
23
23
|
* feedback substrate so the retro and `/mandrel-plan` Phase 0 feedback
|
|
24
24
|
* fetch can see which acceptance items needed rework and the round
|
|
@@ -46,14 +46,19 @@
|
|
|
46
46
|
* the caller into ONE verdict — `criteria[]` in acceptance-array order — and
|
|
47
47
|
* scored here exactly once. Invoking the gate per cluster instead would burn
|
|
48
48
|
* one Story-level round per cluster (distinct fingerprints defeat the replay
|
|
49
|
-
* guard) and race the `signals.ndjson` round ledger.
|
|
50
|
-
*
|
|
51
|
-
*
|
|
49
|
+
* guard) and race the `signals.ndjson` round ledger. The gate reads the
|
|
50
|
+
* Story's own `acceptance[]` count off its body (Story #5313) and rejects a
|
|
51
|
+
* verdict whose `criteria[]` length differs **before** scoring, so the
|
|
52
|
+
* mistake costs no round. `--expected-criteria` is still accepted but is
|
|
53
|
+
* redundant with the derived count: when both are known they must agree.
|
|
54
|
+
* An inline-owned verdict is one file scored in one call — the cluster
|
|
55
|
+
* merge applies only to fresh critics.
|
|
52
56
|
*
|
|
53
57
|
* CLI:
|
|
54
58
|
* --story <id> Story ID (required).
|
|
55
59
|
* --verdict <path> Path to the round's verdict JSON (required).
|
|
56
|
-
* --expected-criteria <n>
|
|
60
|
+
* --expected-criteria <n> Optional; must equal the Story's acceptance[]
|
|
61
|
+
* count when the body is readable.
|
|
57
62
|
* --no-signal Suppress the signal emit (tests).
|
|
58
63
|
*
|
|
59
64
|
* Stdout: a single JSON envelope
|
|
@@ -93,6 +98,8 @@ import {
|
|
|
93
98
|
FULL_SUITE_SHAPE_WARNING,
|
|
94
99
|
isFullSuiteCommand,
|
|
95
100
|
} from './lib/orchestration/verify-credit.js';
|
|
101
|
+
import { createProvider } from './lib/provider-factory.js';
|
|
102
|
+
import { parse as parseStoryBody } from './lib/story-body/story-body.js';
|
|
96
103
|
|
|
97
104
|
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
|
98
105
|
|
|
@@ -181,10 +188,84 @@ const MERGE_CONTRACT =
|
|
|
181
188
|
"merge every cluster's records into a single criteria[] in acceptance[] order, " +
|
|
182
189
|
'one per acceptance item, before scoring.';
|
|
183
190
|
|
|
191
|
+
/**
|
|
192
|
+
* Read the Story's `acceptance[]` count off its body (Story #5313), so the
|
|
193
|
+
* coverage assertion no longer depends on the caller restating a number it
|
|
194
|
+
* already read. Total: any failure — no provider, an unreadable ticket, an
|
|
195
|
+
* unparseable body — yields `null`, which routes the caller to the flag (when
|
|
196
|
+
* passed) and otherwise to a stated, logged skip. It never manufactures a
|
|
197
|
+
* count.
|
|
198
|
+
*
|
|
199
|
+
* @param {{ storyId: number, config: object }} args
|
|
200
|
+
* @param {{ createProviderFn?: typeof createProvider, parseBodyFn?: typeof parseStoryBody }} [deps]
|
|
201
|
+
* @returns {Promise<number|null>}
|
|
202
|
+
*/
|
|
203
|
+
export async function readStoryAcceptanceCount(
|
|
204
|
+
{ storyId, config },
|
|
205
|
+
{ createProviderFn = createProvider, parseBodyFn = parseStoryBody } = {},
|
|
206
|
+
) {
|
|
207
|
+
try {
|
|
208
|
+
const ticket = await createProviderFn(config).getTicket(storyId);
|
|
209
|
+
const body = typeof ticket?.body === 'string' ? ticket.body : null;
|
|
210
|
+
if (body === null) return null;
|
|
211
|
+
const acceptance = parseBodyFn(body)?.body?.acceptance;
|
|
212
|
+
return Array.isArray(acceptance) && acceptance.length > 0
|
|
213
|
+
? acceptance.length
|
|
214
|
+
: null;
|
|
215
|
+
} catch {
|
|
216
|
+
return null;
|
|
217
|
+
}
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
/**
|
|
221
|
+
* Reconcile the body-derived count with the optional flag (Story #5313): the
|
|
222
|
+
* derived count is authoritative; a flag that disagrees with it is a wiring
|
|
223
|
+
* error worth failing on rather than a value to prefer silently.
|
|
224
|
+
*
|
|
225
|
+
* @param {{ derived: number|null, flagged: number|null }} args
|
|
226
|
+
* @returns {number|null}
|
|
227
|
+
*/
|
|
228
|
+
export function reconcileExpectedCriteria({ derived, flagged }) {
|
|
229
|
+
if (derived === null) return flagged;
|
|
230
|
+
if (flagged !== null && flagged !== derived) {
|
|
231
|
+
throw new Error(
|
|
232
|
+
`acceptance-eval: --expected-criteria ${flagged} disagrees with the Story's acceptance[] count (${derived}); drop the flag — the gate reads the count itself.`,
|
|
233
|
+
);
|
|
234
|
+
}
|
|
235
|
+
return derived;
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
/**
|
|
239
|
+
* The count the coverage assertion runs against: the body-derived count,
|
|
240
|
+
* reconciled with the optional flag, with a stated skip when neither is
|
|
241
|
+
* readable (Story #5313). Split out of `runAcceptanceEvalCli` so the CLI
|
|
242
|
+
* body's own branching stays inside its committed cyclomatic budget.
|
|
243
|
+
*
|
|
244
|
+
* @param {{ storyId: number, config: object, flagged: number|null, readAcceptanceCountImpl: typeof readStoryAcceptanceCount, logger: { warn?: Function } }} args
|
|
245
|
+
* @returns {Promise<number|null>}
|
|
246
|
+
*/
|
|
247
|
+
async function resolveExpectedCriteriaCount({
|
|
248
|
+
storyId,
|
|
249
|
+
config,
|
|
250
|
+
flagged,
|
|
251
|
+
readAcceptanceCountImpl,
|
|
252
|
+
logger,
|
|
253
|
+
}) {
|
|
254
|
+
const derived = await readAcceptanceCountImpl({ storyId, config });
|
|
255
|
+
const expected = reconcileExpectedCriteria({ derived, flagged });
|
|
256
|
+
if (expected === null) {
|
|
257
|
+
logger.warn?.(
|
|
258
|
+
'[acceptance-eval] ⚠ the Story acceptance[] count could not be read and no --expected-criteria was passed — the coverage assertion is skipped for this round.',
|
|
259
|
+
);
|
|
260
|
+
}
|
|
261
|
+
return expected;
|
|
262
|
+
}
|
|
263
|
+
|
|
184
264
|
/**
|
|
185
265
|
* Resolve the optional `--expected-criteria` flag to a positive integer, or
|
|
186
|
-
* `null` when the flag is absent
|
|
187
|
-
*
|
|
266
|
+
* `null` when the flag is absent. Since Story #5313 the gate derives the
|
|
267
|
+
* count from the Story body; the flag remains accepted for callers that
|
|
268
|
+
* still pass it and is checked against the derived value.
|
|
188
269
|
*
|
|
189
270
|
* Exported for tests.
|
|
190
271
|
*
|
|
@@ -225,7 +306,7 @@ export function assertCriteriaCoverage(verdict, expectedCriteria) {
|
|
|
225
306
|
const actual = Array.isArray(verdict?.criteria) ? verdict.criteria.length : 0;
|
|
226
307
|
if (actual === expectedCriteria) return;
|
|
227
308
|
throw new Error(
|
|
228
|
-
`acceptance-eval: verdict covers ${actual} criteria but
|
|
309
|
+
`acceptance-eval: verdict covers ${actual} criteria but the Story's acceptance[] count is ${expectedCriteria}. ` +
|
|
229
310
|
`${MERGE_CONTRACT} No round was consumed.`,
|
|
230
311
|
);
|
|
231
312
|
}
|
|
@@ -409,6 +490,7 @@ export async function runAcceptanceEval(
|
|
|
409
490
|
* resolveConfigImpl?: typeof resolveConfig,
|
|
410
491
|
* validateVerdictImpl?: typeof validateVerdict,
|
|
411
492
|
* runAcceptanceEvalImpl?: typeof runAcceptanceEval,
|
|
493
|
+
* readAcceptanceCountImpl?: typeof readStoryAcceptanceCount,
|
|
412
494
|
* logger?: { info: Function, warn?: Function },
|
|
413
495
|
* }} [deps]
|
|
414
496
|
* @returns {Promise<object>} the emitted envelope.
|
|
@@ -422,11 +504,12 @@ export async function runAcceptanceEvalCli(
|
|
|
422
504
|
resolveConfigImpl = resolveConfig,
|
|
423
505
|
validateVerdictImpl = validateVerdict,
|
|
424
506
|
runAcceptanceEvalImpl = runAcceptanceEval,
|
|
507
|
+
readAcceptanceCountImpl = readStoryAcceptanceCount,
|
|
425
508
|
logger = Logger,
|
|
426
509
|
} = deps;
|
|
427
510
|
const { storyId, verdictPath, expectedCriteria, emitSignal } =
|
|
428
511
|
parseCliArgs(argv);
|
|
429
|
-
const
|
|
512
|
+
const flagged = resolveExpectedCriteria(expectedCriteria);
|
|
430
513
|
|
|
431
514
|
if (!storyId) {
|
|
432
515
|
throw new Error(
|
|
@@ -461,9 +544,17 @@ export async function runAcceptanceEvalCli(
|
|
|
461
544
|
|
|
462
545
|
const verdict = validateVerdictImpl(parsed);
|
|
463
546
|
|
|
464
|
-
// Story #4951: a merged verdict must cover every acceptance[] item
|
|
465
|
-
//
|
|
466
|
-
// a free mistake.
|
|
547
|
+
// Story #4951 / #5313: a merged verdict must cover every acceptance[] item,
|
|
548
|
+
// and the count comes from the Story body itself. This runs before the
|
|
549
|
+
// round ledger is touched, so a partial cluster verdict is a free mistake.
|
|
550
|
+
const config = resolveConfigImpl();
|
|
551
|
+
const expected = await resolveExpectedCriteriaCount({
|
|
552
|
+
storyId,
|
|
553
|
+
config,
|
|
554
|
+
flagged,
|
|
555
|
+
readAcceptanceCountImpl,
|
|
556
|
+
logger,
|
|
557
|
+
});
|
|
467
558
|
assertCriteriaCoverage(verdict, expected);
|
|
468
559
|
|
|
469
560
|
// A verdict whose embedded storyId disagrees with the CLI flag is a
|
|
@@ -479,7 +570,6 @@ export async function runAcceptanceEvalCli(
|
|
|
479
570
|
// caller that injects its own scorer.
|
|
480
571
|
warnOnFullSuiteVerify(verdict, logger);
|
|
481
572
|
|
|
482
|
-
const config = resolveConfigImpl();
|
|
483
573
|
const { envelope, exitCode } = await runAcceptanceEvalImpl({
|
|
484
574
|
storyId,
|
|
485
575
|
verdict,
|
|
@@ -522,9 +612,9 @@ runAsCli(import.meta.url, main, {
|
|
|
522
612
|
['--verdict <path>', 'Path to the authored verdict JSON (required).'],
|
|
523
613
|
[
|
|
524
614
|
'--expected-criteria <n>',
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
'
|
|
615
|
+
"Optional and redundant since Story #5313: the gate reads the Story's " +
|
|
616
|
+
'acceptance[] count itself and rejects a shorter verdict before ' +
|
|
617
|
+
'scoring. When passed it must agree with that count.',
|
|
528
618
|
],
|
|
529
619
|
[
|
|
530
620
|
'--no-signal',
|