@opengsd/gsd-core 1.13.0 → 1.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -1
- package/.claude-plugin/plugin.json +1 -1
- package/README.ja-JP.md +3 -3
- package/README.ko-KR.md +3 -3
- package/README.pt-BR.md +3 -3
- package/README.zh-CN.md +3 -3
- package/agents/gsd-advisor-researcher.compact.md +85 -0
- package/agents/gsd-ai-researcher.compact.md +96 -0
- package/agents/gsd-assumptions-analyzer.compact.md +81 -0
- package/agents/gsd-code-fixer.compact.md +459 -0
- package/agents/gsd-code-fixer.md +9 -8
- package/agents/gsd-code-reviewer.compact.md +269 -0
- package/agents/gsd-code-reviewer.md +15 -3
- package/agents/gsd-codebase-mapper.compact.md +760 -0
- package/agents/gsd-debug-session-manager.compact.md +360 -0
- package/agents/gsd-debug-session-manager.md +17 -2
- package/agents/gsd-debugger.md +2 -2
- package/agents/gsd-doc-classifier.compact.md +192 -0
- package/agents/gsd-doc-synthesizer.compact.md +200 -0
- package/agents/gsd-doc-verifier.compact.md +143 -0
- package/agents/gsd-doc-writer.compact.md +440 -0
- package/agents/gsd-dom-verifier.compact.md +138 -0
- package/agents/gsd-domain-researcher.compact.md +141 -0
- package/agents/gsd-eval-auditor.compact.md +160 -0
- package/agents/gsd-eval-auditor.md +1 -1
- package/agents/gsd-eval-planner.compact.md +137 -0
- package/agents/gsd-executor.md +13 -8
- package/agents/gsd-framework-selector.compact.md +82 -0
- package/agents/gsd-integration-checker.compact.md +245 -0
- package/agents/gsd-intel-updater.compact.md +226 -0
- package/agents/gsd-intel-updater.md +1 -1
- package/agents/gsd-mempalace-curator.compact.md +45 -0
- package/agents/gsd-nyquist-auditor.compact.md +179 -0
- package/agents/gsd-pattern-mapper.compact.md +275 -0
- package/agents/gsd-phase-researcher.md +19 -11
- package/agents/gsd-plan-checker.md +8 -7
- package/agents/gsd-planner.md +12 -8
- package/agents/gsd-project-researcher.compact.md +587 -0
- package/agents/gsd-project-researcher.md +1 -1
- package/agents/gsd-research-synthesizer.compact.md +212 -0
- package/agents/gsd-research-synthesizer.md +1 -1
- package/agents/gsd-roadmapper.compact.md +454 -0
- package/agents/gsd-roadmapper.md +13 -0
- package/agents/gsd-security-auditor.compact.md +162 -0
- package/agents/gsd-ui-auditor.compact.md +404 -0
- package/agents/gsd-ui-auditor.md +155 -17
- package/agents/gsd-ui-checker.compact.md +277 -0
- package/agents/gsd-ui-researcher.compact.md +282 -0
- package/agents/gsd-ui-researcher.md +1 -1
- package/agents/gsd-user-profiler.compact.md +108 -0
- package/agents/gsd-verifier.md +10 -9
- package/bin/install.js +848 -163
- package/commands/gsd/autonomous.md +2 -2
- package/commands/gsd/capture.md +1 -1
- package/commands/gsd/cleanup.md +1 -0
- package/commands/gsd/code-review.md +2 -1
- package/commands/gsd/complete-milestone.md +1 -0
- package/commands/gsd/config.md +1 -0
- package/commands/gsd/debug.md +1 -0
- package/commands/gsd/graphify.md +1 -0
- package/commands/gsd/health.md +1 -0
- package/commands/gsd/mempalace-capture.md +8 -3
- package/commands/gsd/mempalace-recall.md +1 -0
- package/commands/gsd/new-milestone.md +1 -0
- package/commands/gsd/new-project.md +1 -0
- package/commands/gsd/next.md +1 -0
- package/commands/gsd/pause-work.md +1 -0
- package/commands/gsd/phase.md +1 -0
- package/commands/gsd/plan-review-convergence.md +6 -6
- package/commands/gsd/pr-branch.md +1 -0
- package/commands/gsd/progress.md +1 -1
- package/commands/gsd/quick-batch.md +1 -1
- package/commands/gsd/resume-work.md +1 -0
- package/commands/gsd/review-backlog.md +1 -0
- package/commands/gsd/review.md +2 -3
- package/commands/gsd/settings.md +2 -1
- package/commands/gsd/stats.md +1 -0
- package/commands/gsd/thread.md +1 -0
- package/commands/gsd/workspace.md +1 -0
- package/commands/gsd/workstreams.md +1 -0
- package/gsd-core/bin/check-latest-version.cjs +8 -3
- package/gsd-core/bin/gsd-tools.cjs +672 -146
- package/gsd-core/bin/lib/adr-parser.cjs +4 -2
- package/gsd-core/bin/lib/artifacts.cjs +2 -1
- package/gsd-core/bin/lib/audit.cjs +119 -34
- package/gsd-core/bin/lib/broken-windows.cjs +168 -49
- package/gsd-core/bin/lib/capability-lifecycle.cjs +10 -6
- package/gsd-core/bin/lib/capability-loader.cjs +135 -1
- package/gsd-core/bin/lib/capability-registry.cjs +96 -189
- package/gsd-core/bin/lib/capability-source.cjs +19 -2
- package/gsd-core/bin/lib/capability-validator.cjs +14 -2
- package/gsd-core/bin/lib/check-command-router.cjs +213 -49
- package/gsd-core/bin/lib/code-review-depth.cjs +2 -2
- package/gsd-core/bin/lib/codex-agent-toml.cjs +21 -25
- package/gsd-core/bin/lib/commands.cjs +823 -112
- package/gsd-core/bin/lib/config-loader.cjs +66 -4
- package/gsd-core/bin/lib/config.cjs +186 -45
- package/gsd-core/bin/lib/coverage.cjs +1 -1
- package/gsd-core/bin/lib/decisions.cjs +164 -45
- package/gsd-core/bin/lib/external-descriptor-trust.cjs +29 -14
- package/gsd-core/bin/lib/frontmatter.cjs +13 -0
- package/gsd-core/bin/lib/graphify.cjs +10 -2
- package/gsd-core/bin/lib/gsd2-import.cjs +1 -2
- package/gsd-core/bin/lib/health-diagnostic-rules/state-consistency.cjs +12 -1
- package/gsd-core/bin/lib/health-diagnostic-rules/worktree-health.cjs +1 -1
- package/gsd-core/bin/lib/host-runtime-detection.cjs +9 -0
- package/gsd-core/bin/lib/init.cjs +614 -86
- package/gsd-core/bin/lib/install-engine.cjs +29 -3
- package/gsd-core/bin/lib/install-profiles.cjs +14 -0
- package/gsd-core/bin/lib/installer-migrations.cjs +41 -5
- package/gsd-core/bin/lib/loop-resolver.cjs +50 -31
- package/gsd-core/bin/lib/mcp-catalog.cjs +2 -2
- package/gsd-core/bin/lib/milestone.cjs +37 -13
- package/gsd-core/bin/lib/model-resolver.cjs +253 -53
- package/gsd-core/bin/lib/phase-command-router.cjs +16 -2
- package/gsd-core/bin/lib/phase-id-card.cjs +32 -0
- package/gsd-core/bin/lib/phase-id-display.cjs +78 -0
- package/gsd-core/bin/lib/phase-id.cjs +268 -27
- package/gsd-core/bin/lib/phase-lifecycle.cjs +61 -0
- package/gsd-core/bin/lib/phase-locator.cjs +29 -10
- package/gsd-core/bin/lib/phase.cjs +393 -88
- package/gsd-core/bin/lib/plan-document.cjs +49 -1
- package/gsd-core/bin/lib/planning-document.cjs +459 -0
- package/gsd-core/bin/lib/planning-inspect.cjs +52 -19
- package/gsd-core/bin/lib/planning-snapshot.cjs +61 -12
- package/gsd-core/bin/lib/planning-workspace.cjs +57 -3
- package/gsd-core/bin/lib/pr-branch-patterns.cjs +57 -0
- package/gsd-core/bin/lib/pristine-baseline.cjs +182 -0
- package/gsd-core/bin/lib/probe-core.cjs +7 -1
- package/gsd-core/bin/lib/prohibition-enforcement.cjs +91 -4
- package/gsd-core/bin/lib/project-root.cjs +41 -2
- package/gsd-core/bin/lib/quick-batch.cjs +1 -1
- package/gsd-core/bin/lib/refactor-trigger-command-router.cjs +61 -2
- package/gsd-core/bin/lib/research-store.cjs +11 -12
- package/gsd-core/bin/lib/review-lane-descriptor.cjs +10 -30
- package/gsd-core/bin/lib/review-lane-invocation.cjs +23 -0
- package/gsd-core/bin/lib/review-reviewer-selection.cjs +2 -2
- package/gsd-core/bin/lib/reviewer-step-dispatch.cjs +337 -0
- package/gsd-core/bin/lib/roadmap-command-router.cjs +12 -4
- package/gsd-core/bin/lib/roadmap-parser.cjs +219 -18
- package/gsd-core/bin/lib/roadmap-upgrade.cjs +1539 -13
- package/gsd-core/bin/lib/roadmap.cjs +356 -42
- package/gsd-core/bin/lib/runtime-artifact-conversion.cjs +310 -41
- package/gsd-core/bin/lib/runtime-artifact-install-plan.cjs +15 -4
- package/gsd-core/bin/lib/runtime-artifact-layout.cjs +13 -5
- package/gsd-core/bin/lib/runtime-homes.cjs +4 -0
- package/gsd-core/bin/lib/runtime-hooks-surface.cjs +408 -37
- package/gsd-core/bin/lib/runtime-name-policy.cjs +111 -1
- package/gsd-core/bin/lib/security.cjs +126 -7
- package/gsd-core/bin/lib/shell-command-projection.cjs +10 -6
- package/gsd-core/bin/lib/state-document.cjs +130 -28
- package/gsd-core/bin/lib/state-md-schema.cjs +21 -14
- package/gsd-core/bin/lib/state-transition.cjs +181 -30
- package/gsd-core/bin/lib/state.cjs +265 -27
- package/gsd-core/bin/lib/surface.cjs +77 -3
- package/gsd-core/bin/lib/task-command-router.cjs +12 -6
- package/gsd-core/bin/lib/tdd-red-evidence.cjs +78 -5
- package/gsd-core/bin/lib/uat-predicate.cjs +47 -4
- package/gsd-core/bin/lib/uat.cjs +9 -1
- package/gsd-core/bin/lib/ui-consideration-probe.cjs +15 -2
- package/gsd-core/bin/lib/ui-frontend-evidence.cjs +100 -9
- package/gsd-core/bin/lib/undo-commit-selection.cjs +131 -0
- package/gsd-core/bin/lib/update-context.cjs +30 -24
- package/gsd-core/bin/lib/vendor/js-yaml.cjs +11 -3
- package/gsd-core/bin/lib/verification.cjs +315 -30
- package/gsd-core/bin/lib/verify-command-grounding.cjs +47 -3
- package/gsd-core/bin/lib/verify.cjs +320 -48
- package/gsd-core/bin/lib/workstream-inventory.cjs +1 -0
- package/gsd-core/bin/lib/worktree-base-ref.cjs +482 -73
- package/gsd-core/bin/lib/worktree-safety.cjs +797 -58
- package/gsd-core/bin/shared/config-defaults.manifest.json +4 -0
- package/gsd-core/bin/shared/config-schema.manifest.json +6 -0
- package/gsd-core/bin/verify-reapply-patches.cjs +439 -80
- package/gsd-core/references/checkpoints.md +5 -3
- package/gsd-core/references/compact-content-gate.md +66 -0
- package/gsd-core/references/edge-probe-fixtures/01-round-half-even/expected-coverage.json +28 -3
- package/gsd-core/references/edge-probe-fixtures/02-merge-intervals/expected-coverage.json +37 -4
- package/gsd-core/references/edge-probe-fixtures/03-truncate-graphemes/expected-coverage.json +28 -3
- package/gsd-core/references/edge-probe-fixtures/04-money-rounding/expected-coverage.json +28 -3
- package/gsd-core/references/edge-probe-fixtures/05-list-dedupe/expected-coverage.json +37 -4
- package/gsd-core/references/edge-probe-fixtures/06-resolved-mixed/expected-coverage.json +37 -4
- package/gsd-core/references/edge-probe.md +195 -21
- package/gsd-core/references/execute-phase-between-wave-reset.md +7 -6
- package/gsd-core/references/execute-phase-wave-guard.md +22 -11
- package/gsd-core/references/gsd-run-resolver.md +1 -1
- package/gsd-core/references/loop-hook-dispatch.md +18 -0
- package/gsd-core/references/model-profiles.md +13 -4
- package/gsd-core/references/phase-argument-parsing.md +9 -7
- package/gsd-core/references/phase-id-convention.md +28 -0
- package/gsd-core/references/planner-gap-closure.md +2 -0
- package/gsd-core/references/planner-load-graph-context.md +24 -13
- package/gsd-core/references/planner-verify-command-grounding.md +14 -0
- package/gsd-core/references/planning-config.md +14 -2
- package/gsd-core/references/tdd.md +30 -4
- package/gsd-core/references/thinking-models-planning.md +18 -2
- package/gsd-core/references/ui-consideration-probe.md +10 -5
- package/gsd-core/references/verification-patterns.md +17 -4
- package/gsd-core/references/verify-command-path-resolvability.md +10 -2
- package/gsd-core/references/worktree-path-safety.md +433 -2
- package/gsd-core/templates/README.md +7 -1
- package/gsd-core/templates/state.md +6 -3
- package/gsd-core/templates/summary.compact.md +212 -0
- package/gsd-core/templates/user-setup.compact.md +199 -0
- package/gsd-core/templates/user-setup.md +0 -9
- package/gsd-core/templates/verification-report.md +1 -1
- package/gsd-core/workflows/_runtime-launcher.snippet.sh +1 -1
- package/gsd-core/workflows/add-backlog.md +1 -1
- package/gsd-core/workflows/add-phase.md +1 -1
- package/gsd-core/workflows/add-tests.md +2 -2
- package/gsd-core/workflows/add-todo.md +6 -5
- package/gsd-core/workflows/ai-integration-phase.md +11 -3
- package/gsd-core/workflows/audit-fix.md +1 -1
- package/gsd-core/workflows/audit-milestone.md +1 -1
- package/gsd-core/workflows/audit-uat.md +1 -1
- package/gsd-core/workflows/autonomous/steps/converge-fail-fast.md +9 -18
- package/gsd-core/workflows/autonomous.md +29 -16
- package/gsd-core/workflows/check-todos.md +6 -4
- package/gsd-core/workflows/cleanup.md +5 -3
- package/gsd-core/workflows/code-review/steps/dispatch-fix.md +4 -3
- package/gsd-core/workflows/code-review/steps/structural-pre-pass.md +8 -1
- package/gsd-core/workflows/code-review-fix.md +108 -22
- package/gsd-core/workflows/code-review.md +216 -73
- package/gsd-core/workflows/complete-milestone/detail/elaboration.md +274 -0
- package/gsd-core/workflows/complete-milestone.md +41 -264
- package/gsd-core/workflows/debug.md +3 -3
- package/gsd-core/workflows/diagnose-issues.md +1 -1
- package/gsd-core/workflows/discuss-phase/modes/advisor.md +1 -1
- package/gsd-core/workflows/discuss-phase/modes/chain.md +1 -1
- package/gsd-core/workflows/discuss-phase-assumptions.md +1 -1
- package/gsd-core/workflows/discuss-phase.md +1 -1
- package/gsd-core/workflows/do.md +2 -2
- package/gsd-core/workflows/docs-update/detail/elaboration.md +179 -0
- package/gsd-core/workflows/docs-update.md +17 -158
- package/gsd-core/workflows/edit-phase.md +1 -1
- package/gsd-core/workflows/eval-review.md +10 -3
- package/gsd-core/workflows/execute-phase/detail/elaboration.md +124 -0
- package/gsd-core/workflows/execute-phase/steps/code-review-disposition.md +1017 -0
- package/gsd-core/workflows/execute-phase/steps/codebase-drift-gate.md +19 -4
- package/gsd-core/workflows/execute-phase/steps/completion-reconciliation.md +56 -0
- package/gsd-core/workflows/execute-phase/steps/executor-isolation-dispatch.md +43 -4
- package/gsd-core/workflows/execute-phase/steps/executor-progress-policy.md +43 -0
- package/gsd-core/workflows/execute-phase/steps/gap-closure-artifacts.md +1 -1
- package/gsd-core/workflows/execute-phase/steps/partial-wave.md +1 -1
- package/gsd-core/workflows/execute-phase/steps/per-plan-executor-routing.md +1 -1
- package/gsd-core/workflows/execute-phase/steps/per-plan-worktree-gate.md +1 -1
- package/gsd-core/workflows/execute-phase/steps/post-merge-gate.md +46 -9
- package/gsd-core/workflows/execute-phase/steps/protected-branch.md +1 -1
- package/gsd-core/workflows/execute-phase/steps/ready-wave-gate.md +37 -0
- package/gsd-core/workflows/execute-phase/steps/regression-gate-run.md +1 -1
- package/gsd-core/workflows/execute-phase/steps/sequential-root-pin.md +35 -0
- package/gsd-core/workflows/execute-phase/steps/stale-reverification.md +24 -0
- package/gsd-core/workflows/execute-phase/steps/tdd-applicability-resolution.md +1 -1
- package/gsd-core/workflows/execute-phase/steps/threat-id-gate.md +28 -0
- package/gsd-core/workflows/execute-phase/steps/wave-post-gate-hooks.md +1 -1
- package/gsd-core/workflows/execute-phase/steps/worktree-base-check.md +25 -0
- package/gsd-core/workflows/execute-phase.md +83 -172
- package/gsd-core/workflows/execute-plan.md +24 -10
- package/gsd-core/workflows/explore.md +3 -3
- package/gsd-core/workflows/extract-learnings.md +2 -1
- package/gsd-core/workflows/fast.md +1 -1
- package/gsd-core/workflows/forensics.md +1 -1
- package/gsd-core/workflows/graduation.md +1 -1
- package/gsd-core/workflows/health.md +2 -2
- package/gsd-core/workflows/help/modes/full.compact.md +398 -0
- package/gsd-core/workflows/help/modes/full.md +5 -5
- package/gsd-core/workflows/help/modes/topic.md +15 -5
- package/gsd-core/workflows/help.md +1 -1
- package/gsd-core/workflows/import.md +2 -2
- package/gsd-core/workflows/inbox.md +2 -2
- package/gsd-core/workflows/ingest-docs.md +3 -3
- package/gsd-core/workflows/insert-phase.md +1 -1
- package/gsd-core/workflows/list-seeds.md +1 -1
- package/gsd-core/workflows/list-workspaces.md +1 -1
- package/gsd-core/workflows/manager.md +2 -2
- package/gsd-core/workflows/map-codebase.md +52 -5
- package/gsd-core/workflows/milestone-summary.md +1 -1
- package/gsd-core/workflows/mvp-phase.md +1 -1
- package/gsd-core/workflows/new-milestone.md +56 -14
- package/gsd-core/workflows/new-project/detail/elaboration.md +216 -0
- package/gsd-core/workflows/new-project/steps/auto-mode-config.md +3 -3
- package/gsd-core/workflows/new-project/steps/codebase-map-offer.md +1 -1
- package/gsd-core/workflows/new-project.md +39 -209
- package/gsd-core/workflows/new-workspace.md +2 -2
- package/gsd-core/workflows/next.md +1 -1
- package/gsd-core/workflows/note.md +1 -1
- package/gsd-core/workflows/onboard.md +1 -1
- package/gsd-core/workflows/pause-work.md +1 -1
- package/gsd-core/workflows/plan-phase/detail/elaboration.md +209 -0
- package/gsd-core/workflows/plan-phase/steps/chunked-planning-mode.md +17 -5
- package/gsd-core/workflows/plan-phase/steps/prd-express-path.md +1 -1
- package/gsd-core/workflows/plan-phase/steps/stall-detection-helpers.md +23 -5
- package/gsd-core/workflows/plan-phase.md +45 -187
- package/gsd-core/workflows/plan-review-convergence.md +21 -5
- package/gsd-core/workflows/plant-seed.md +62 -20
- package/gsd-core/workflows/pr-branch.md +132 -20
- package/gsd-core/workflows/profile-user.md +2 -2
- package/gsd-core/workflows/progress.md +1 -1
- package/gsd-core/workflows/quick/steps/plan-checker-loop.md +25 -0
- package/gsd-core/workflows/quick/steps/quick-verification.md +1 -1
- package/gsd-core/workflows/quick/steps/worktree-pre-dispatch-commit.md +1 -1
- package/gsd-core/workflows/quick-batch/steps/batch-init.md +1 -1
- package/gsd-core/workflows/quick-batch/steps/completion.md +1 -1
- package/gsd-core/workflows/quick-batch/steps/merge-wave.md +1 -1
- package/gsd-core/workflows/quick-batch/steps/planner-wave.md +1 -1
- package/gsd-core/workflows/quick-batch/steps/research-phase.md +1 -1
- package/gsd-core/workflows/quick-batch/steps/resume-mode.md +1 -1
- package/gsd-core/workflows/quick-batch/steps/verification-wave.md +1 -1
- package/gsd-core/workflows/quick-batch/steps/worktree-dispatch.md +1 -1
- package/gsd-core/workflows/quick-batch.md +1 -1
- package/gsd-core/workflows/quick.md +29 -10
- package/gsd-core/workflows/reapply-patches.md +86 -6
- package/gsd-core/workflows/remove-phase.md +1 -1
- package/gsd-core/workflows/remove-workspace.md +2 -2
- package/gsd-core/workflows/resume-project.md +1 -1
- package/gsd-core/workflows/review.md +31 -16
- package/gsd-core/workflows/scan.md +1 -1
- package/gsd-core/workflows/secure-phase.md +3 -2
- package/gsd-core/workflows/settings-advanced.md +30 -10
- package/gsd-core/workflows/settings-integrations.md +2 -3
- package/gsd-core/workflows/settings.md +22 -9
- package/gsd-core/workflows/ship.md +3 -2
- package/gsd-core/workflows/sketch-wrap-up.md +1 -1
- package/gsd-core/workflows/sketch.md +1 -1
- package/gsd-core/workflows/smart-entry.md +2 -2
- package/gsd-core/workflows/spec-phase.md +15 -5
- package/gsd-core/workflows/spike-wrap-up.md +1 -1
- package/gsd-core/workflows/spike.md +1 -1
- package/gsd-core/workflows/stats.md +1 -1
- package/gsd-core/workflows/sync-skills.md +5 -5
- package/gsd-core/workflows/thread.md +1 -1
- package/gsd-core/workflows/transition.md +1 -1
- package/gsd-core/workflows/ui-phase.md +44 -8
- package/gsd-core/workflows/ui-review.md +18 -4
- package/gsd-core/workflows/ultraplan-phase.md +1 -1
- package/gsd-core/workflows/undo.md +339 -20
- package/gsd-core/workflows/update.md +14 -12
- package/gsd-core/workflows/validate-phase.md +3 -2
- package/gsd-core/workflows/verify-work/detail/elaboration.md +230 -0
- package/gsd-core/workflows/verify-work/steps/automated-ui-verification.md +1 -1
- package/gsd-core/workflows/verify-work/steps/mvp-uat-framing.md +1 -1
- package/gsd-core/workflows/verify-work.md +101 -196
- package/hooks/dist/gsd-agent-isolation-guard.js +66 -16
- package/hooks/dist/gsd-context-monitor.js +88 -15
- package/hooks/dist/gsd-cursor-subagent-start.js +34 -14
- package/hooks/dist/gsd-secret-read-guard.js +71 -19
- package/hooks/dist/gsd-statusline.js +81 -20
- package/hooks/dist/gsd-validate-commit.sh +97 -8
- package/hooks/dist/gsd-worktree-path-guard.js +25 -14
- package/hooks/dist/gsd-write-guard.js +46 -1
- package/hooks/dist/lib/dispatch-identity.js +187 -0
- package/hooks/dist/lib/filename-classification.js +64 -0
- package/hooks/dist/lib/isolation-deny-reason.js +53 -1
- package/hooks/dist/lib/isolation-sentinel.js +58 -19
- package/hooks/gsd-agent-isolation-guard.js +66 -16
- package/hooks/gsd-context-monitor.js +88 -15
- package/hooks/gsd-cursor-subagent-start.js +34 -14
- package/hooks/gsd-secret-read-guard.js +71 -19
- package/hooks/gsd-statusline.js +81 -20
- package/hooks/gsd-validate-commit.sh +97 -8
- package/hooks/gsd-worktree-path-guard.js +25 -14
- package/hooks/gsd-write-guard.js +46 -1
- package/hooks/lib/dispatch-identity.js +187 -0
- package/hooks/lib/filename-classification.js +64 -0
- package/hooks/lib/isolation-deny-reason.js +53 -1
- package/hooks/lib/isolation-sentinel.js +58 -19
- package/package.json +11 -6
- package/scripts/benchmark-compact-content-variants.cjs +298 -0
- package/scripts/benchmark-compact-content.cjs +368 -0
- package/scripts/build-hooks.js +15 -6
- package/scripts/check-contract-drift.cjs +131 -12
- package/scripts/check-env.cjs +36 -8
- package/scripts/check-glossary-refs.cjs +25 -21
- package/scripts/ci-next-health.cjs +271 -0
- package/scripts/ci-prepare-test-scope.cjs +7 -7
- package/scripts/ci-test-scope.cjs +126 -20
- package/scripts/ci-timeout-report.cjs +1 -1
- package/scripts/command-contract-helpers.cjs +3 -0
- package/scripts/diff-touches-shipped-paths.cjs +1 -1
- package/scripts/docs-guard-registry.cjs +35 -2
- package/scripts/gen-adr-index.cjs +8 -2
- package/scripts/gen-inventory-manifest.cjs +12 -0
- package/scripts/gen-loop-host-contract.cjs +69 -0
- package/scripts/gen-platform-conformance-tier.cjs +557 -0
- package/scripts/lib/drift-scan.cjs +1 -1
- package/scripts/lib/macos-conformance-tier.generated.cjs +224 -0
- package/scripts/lib/ndjson-reporter.cjs +3 -2
- package/scripts/lib/npm-version-check-diagnosis.cjs +59 -0
- package/scripts/lib/platform-conformance-tier.generated.cjs +287 -0
- package/scripts/lib/suite-detection.cjs +32 -0
- package/scripts/lint-allowed-tools-parity.cjs +221 -0
- package/scripts/lint-docs-guard-registration.exempt-baseline.cjs +47 -3
- package/scripts/lint-phase-arg-assignment.cjs +257 -0
- package/scripts/lint-phase-id-drift.cjs +623 -13
- package/scripts/lint-pr-branch-pattern-drift.cjs +148 -0
- package/scripts/lint-response-language-coverage.cjs +9 -3
- package/scripts/lint-retired-runtime-name.cjs +619 -0
- package/scripts/lint-source-test-name-collision.cjs +1 -1
- package/scripts/lint-state-write-path-drift.cjs +93 -0
- package/scripts/lint-test-file-count.allowlist.json +29 -9
- package/scripts/lint-vendored-deps.cjs +128 -17
- package/scripts/lint-workflow-shellcheck-baseline.json +100 -0
- package/scripts/prompt-injection-scan.sh +18 -0
- package/scripts/release-tarball-smoke.cjs +194 -1
- package/scripts/workflow-size.cjs +139 -0
- package/skills/gsd-autonomous/SKILL.md +2 -2
- package/skills/gsd-capture/SKILL.md +1 -1
- package/skills/gsd-cleanup/SKILL.md +1 -0
- package/skills/gsd-code-review/SKILL.md +2 -1
- package/skills/gsd-complete-milestone/SKILL.md +1 -0
- package/skills/gsd-config/SKILL.md +1 -0
- package/skills/gsd-debug/SKILL.md +1 -0
- package/skills/gsd-graphify/SKILL.md +1 -0
- package/skills/gsd-health/SKILL.md +1 -0
- package/skills/gsd-mempalace-capture/SKILL.md +8 -3
- package/skills/gsd-mempalace-recall/SKILL.md +1 -0
- package/skills/gsd-new-milestone/SKILL.md +1 -0
- package/skills/gsd-new-project/SKILL.md +1 -0
- package/skills/gsd-next/SKILL.md +1 -0
- package/skills/gsd-pause-work/SKILL.md +1 -0
- package/skills/gsd-phase/SKILL.md +1 -0
- package/skills/gsd-plan-review-convergence/SKILL.md +5 -5
- package/skills/gsd-pr-branch/SKILL.md +1 -0
- package/skills/gsd-progress/SKILL.md +1 -1
- package/skills/gsd-quick-batch/SKILL.md +1 -1
- package/skills/gsd-resume-work/SKILL.md +1 -0
- package/skills/gsd-review/SKILL.md +2 -3
- package/skills/gsd-review-backlog/SKILL.md +1 -0
- package/skills/gsd-settings/SKILL.md +2 -1
- package/skills/gsd-stats/SKILL.md +1 -0
- package/skills/gsd-thread/SKILL.md +1 -0
- package/skills/gsd-workspace/SKILL.md +1 -0
- package/skills/gsd-workstreams/SKILL.md +1 -0
- package/vscode/package.json +1 -1
- package/gsd-core/templates/claude-md.md +0 -145
- package/gsd-core/templates/codebase/concerns.md +0 -310
- package/gsd-core/templates/codebase/conventions.md +0 -307
- package/gsd-core/templates/codebase/integrations.md +0 -280
- package/gsd-core/templates/codebase/structure.md +0 -285
- package/gsd-core/templates/codebase/testing.md +0 -480
- package/gsd-core/templates/debug-subagent-prompt.md +0 -91
- package/gsd-core/templates/discovery.md +0 -146
|
@@ -0,0 +1,298 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Benchmarks the token-count effect of every registered variant-swap pair
|
|
6
|
+
* (ADR-4139, Phase 6 #4406 — `gsd-core/workflows/<name>/{modes,steps,templates}/*.compact.md`
|
|
7
|
+
* and `gsd-core/templates/**\/*.compact.md`). Sibling to
|
|
8
|
+
* `scripts/benchmark-compact-content.cjs` (Phase 3's spine/detail benchmark) rather than an
|
|
9
|
+
* extension of it — the two are different data shapes (a variant pair is two independent,
|
|
10
|
+
* deliberately-overlapping complete files; a spine/detail split is one document partitioned in
|
|
11
|
+
* two disjoint halves), and mixing them into one report would conflate an "off" total that means
|
|
12
|
+
* something different in each case.
|
|
13
|
+
*
|
|
14
|
+
* For every registered pair it reports the "off" token count (the canonical file — what
|
|
15
|
+
* `workflow.compact_content=false`, the default, pays at that call site) against the "on" token
|
|
16
|
+
* count (the `.compact.md` sibling — what `workflow.compact_content=true` pays once the gate
|
|
17
|
+
* resolves to it). See `gsd-core/references/compact-content-gate.md` §"Streams 1b and 4" for the
|
|
18
|
+
* resolution rule this measures.
|
|
19
|
+
*
|
|
20
|
+
* PROXY-TOKENIZER CAVEAT: same as the sibling script — `gpt-tokenizer` is a stand-in tokenizer;
|
|
21
|
+
* Anthropic publishes none for Claude 3+. The on/off comparison is exact under this one pinned
|
|
22
|
+
* tokenizer applied identically to both sides; the absolute counts are not Claude's real counts.
|
|
23
|
+
*
|
|
24
|
+
* Discovery is REIMPLEMENTED here rather than imported from
|
|
25
|
+
* `tests/helpers/compact-content-variant.cjs`, for the same reason
|
|
26
|
+
* `benchmark-compact-content.cjs` reimplements spine/detail discovery instead of importing it: a
|
|
27
|
+
* `scripts/` reporting tool depending on a test-only helper module inverts this repo's normal
|
|
28
|
+
* layering, and a test-only module changing shape should never be able to break a benchmark.
|
|
29
|
+
*
|
|
30
|
+
* Usage:
|
|
31
|
+
* node scripts/benchmark-compact-content-variants.cjs # print JSON to stdout
|
|
32
|
+
* node scripts/benchmark-compact-content-variants.cjs --write # write the committed baseline
|
|
33
|
+
* node scripts/benchmark-compact-content-variants.cjs --check # recompute, diff vs committed baseline
|
|
34
|
+
* node scripts/benchmark-compact-content-variants.cjs --check --baseline-path=<path>
|
|
35
|
+
*
|
|
36
|
+
* Same CRITICAL contract as the sibling script: this — and `--check` especially — MUST NEVER
|
|
37
|
+
* exit non-zero because a baseline is drifted, stale, or missing. Only a genuine I/O error
|
|
38
|
+
* reading a SOURCE file the benchmark measures may throw. This is a reporting instrument, never
|
|
39
|
+
* a gate.
|
|
40
|
+
*/
|
|
41
|
+
|
|
42
|
+
const fs = require('node:fs');
|
|
43
|
+
const path = require('node:path');
|
|
44
|
+
|
|
45
|
+
const { countTokens } = require('gpt-tokenizer');
|
|
46
|
+
const { runMain } = require('./lib/cli-exit.cjs');
|
|
47
|
+
|
|
48
|
+
const ROOT = path.resolve(__dirname, '..');
|
|
49
|
+
// #4407: agents/ joins the scan. Reachability there is a code seam
|
|
50
|
+
// (cmdAgentSkills), not a markdown literal reference, but token accounting
|
|
51
|
+
// doesn't care how a pair is reached — only that it's registered.
|
|
52
|
+
const VARIANT_ROOTS = [
|
|
53
|
+
path.join(ROOT, 'gsd-core', 'workflows'),
|
|
54
|
+
path.join(ROOT, 'gsd-core', 'templates'),
|
|
55
|
+
path.join(ROOT, 'agents'),
|
|
56
|
+
];
|
|
57
|
+
const BASELINE_PATH = path.join(ROOT, 'tests', 'fixtures', 'compact-content-variant-benchmark-baseline.json');
|
|
58
|
+
const COMPACT_SUFFIX = '.compact.md';
|
|
59
|
+
|
|
60
|
+
function getTokenizerVersion() {
|
|
61
|
+
const pkgPath = require.resolve('gpt-tokenizer/package.json');
|
|
62
|
+
const pkg = JSON.parse(fs.readFileSync(pkgPath, 'utf8'));
|
|
63
|
+
return pkg.version;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* Discover every registered variant pair under `roots`: a `.compact.md` file
|
|
68
|
+
* with a same-directory, same-stem canonical `.md` sibling. Mirrors
|
|
69
|
+
* `tests/helpers/compact-content-variant.cjs`'s `discoverRegisteredVariants`
|
|
70
|
+
* in shape but is a from-scratch, self-contained implementation (see module
|
|
71
|
+
* header for why this is not a shared import). Skips an orphaned compact
|
|
72
|
+
* file with no canonical sibling — that is the guard's problem, not this
|
|
73
|
+
* benchmark's; a pair with no canonical baseline has no "off" number to
|
|
74
|
+
* report against.
|
|
75
|
+
*
|
|
76
|
+
* @param {string[]} [roots]
|
|
77
|
+
* @returns {Array<{name: string, canonicalPath: string, compactPath: string}>}
|
|
78
|
+
*/
|
|
79
|
+
function discoverRegisteredVariants(roots = VARIANT_ROOTS) {
|
|
80
|
+
const pairs = [];
|
|
81
|
+
|
|
82
|
+
function walk(dir) {
|
|
83
|
+
let entries;
|
|
84
|
+
try {
|
|
85
|
+
entries = fs.readdirSync(dir, { withFileTypes: true });
|
|
86
|
+
} catch {
|
|
87
|
+
return;
|
|
88
|
+
}
|
|
89
|
+
for (const entry of entries) {
|
|
90
|
+
const full = path.join(dir, entry.name);
|
|
91
|
+
if (entry.isDirectory()) {
|
|
92
|
+
walk(full);
|
|
93
|
+
} else if (entry.isFile() && entry.name.endsWith(COMPACT_SUFFIX)) {
|
|
94
|
+
const stem = entry.name.slice(0, -COMPACT_SUFFIX.length);
|
|
95
|
+
const canonicalPath = path.join(dir, `${stem}.md`);
|
|
96
|
+
if (fs.existsSync(canonicalPath)) {
|
|
97
|
+
const name = path.relative(ROOT, canonicalPath).split(path.sep).join('/');
|
|
98
|
+
pairs.push({ name, canonicalPath, compactPath: full });
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
for (const root of roots) walk(root);
|
|
105
|
+
return pairs.sort((a, b) => a.name.localeCompare(b.name));
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
/**
|
|
109
|
+
* Compute the off/on/reduction numbers for ONE registered variant pair.
|
|
110
|
+
* Throws on a genuine read failure of a source file (the one thing allowed
|
|
111
|
+
* to throw, per the module-header CRITICAL note).
|
|
112
|
+
*
|
|
113
|
+
* @param {{canonicalPath: string, compactPath: string}} pair
|
|
114
|
+
* @returns {{offTokens: number, onTokens: number, reductionPct: number}}
|
|
115
|
+
*/
|
|
116
|
+
function computePairTokens(pair) {
|
|
117
|
+
const offTokens = countTokens(fs.readFileSync(pair.canonicalPath, 'utf8'));
|
|
118
|
+
const onTokens = countTokens(fs.readFileSync(pair.compactPath, 'utf8'));
|
|
119
|
+
const reductionPct = offTokens === 0 ? 0 : round2(((offTokens - onTokens) / offTokens) * 100);
|
|
120
|
+
return { offTokens, onTokens, reductionPct };
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
function round2(n) {
|
|
124
|
+
return Math.round(n * 100) / 100;
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
/**
|
|
128
|
+
* Aggregate per-pair numbers from SUMMED off/on totals, never averaged
|
|
129
|
+
* per-pair percentages. Reports `0` (not `NaN`) for zero registered pairs.
|
|
130
|
+
* @param {Record<string, {offTokens: number, onTokens: number}>} pairResults
|
|
131
|
+
*/
|
|
132
|
+
function computeAggregate(pairResults) {
|
|
133
|
+
let offTokens = 0;
|
|
134
|
+
let onTokens = 0;
|
|
135
|
+
for (const key of Object.keys(pairResults)) {
|
|
136
|
+
offTokens += pairResults[key].offTokens;
|
|
137
|
+
onTokens += pairResults[key].onTokens;
|
|
138
|
+
}
|
|
139
|
+
const reductionPct = offTokens === 0 ? 0 : round2(((offTokens - onTokens) / offTokens) * 100);
|
|
140
|
+
return { offTokens, onTokens, reductionPct };
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
const LABEL =
|
|
144
|
+
"PROXY-TOKENIZER DELTA — gpt-tokenizer is a stand-in; Anthropic publishes no tokenizer for Claude 3+. " +
|
|
145
|
+
"The on/off COMPARISON is exact under this pinned tokenizer; absolute counts are not Claude's real token counts.";
|
|
146
|
+
|
|
147
|
+
/**
|
|
148
|
+
* @param {string[]} [roots]
|
|
149
|
+
* @returns {object}
|
|
150
|
+
*/
|
|
151
|
+
function buildReport(roots = VARIANT_ROOTS) {
|
|
152
|
+
const pairs = discoverRegisteredVariants(roots);
|
|
153
|
+
const pairReports = {};
|
|
154
|
+
for (const pair of pairs) {
|
|
155
|
+
pairReports[pair.name] = computePairTokens(pair);
|
|
156
|
+
}
|
|
157
|
+
return {
|
|
158
|
+
schema_version: 1,
|
|
159
|
+
generated_by: 'scripts/benchmark-compact-content-variants.cjs',
|
|
160
|
+
tokenizer: { name: 'gpt-tokenizer', version: getTokenizerVersion() },
|
|
161
|
+
label: LABEL,
|
|
162
|
+
pairs: pairReports,
|
|
163
|
+
aggregate: computeAggregate(pairReports),
|
|
164
|
+
};
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
/**
|
|
168
|
+
* Format a human-readable drift summary between a (possibly missing/invalid)
|
|
169
|
+
* committed baseline and a freshly-computed live report. Never throws.
|
|
170
|
+
* @param {string} baselinePath
|
|
171
|
+
* @param {object} live
|
|
172
|
+
* @returns {string}
|
|
173
|
+
*/
|
|
174
|
+
function formatDriftReport(baselinePath, live) {
|
|
175
|
+
const lines = [];
|
|
176
|
+
let baseline = null;
|
|
177
|
+
let baselineReadError = null;
|
|
178
|
+
try {
|
|
179
|
+
const raw = fs.readFileSync(baselinePath, 'utf8');
|
|
180
|
+
try {
|
|
181
|
+
baseline = JSON.parse(raw);
|
|
182
|
+
} catch (parseErr) {
|
|
183
|
+
baselineReadError = `baseline at ${baselinePath} could not be parsed as JSON: ${parseErr.message}`;
|
|
184
|
+
}
|
|
185
|
+
} catch {
|
|
186
|
+
baselineReadError = `no baseline found at ${baselinePath}`;
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
if (baselineReadError) {
|
|
190
|
+
lines.push(`DRIFT: ${baselineReadError} — treating as fully drifted (this is reported, not an error).`);
|
|
191
|
+
lines.push('Live pairs:');
|
|
192
|
+
for (const name of Object.keys(live.pairs).sort()) {
|
|
193
|
+
const p = live.pairs[name];
|
|
194
|
+
lines.push(` + ${name}: off=${p.offTokens} on=${p.onTokens} reduction=${p.reductionPct}%`);
|
|
195
|
+
}
|
|
196
|
+
lines.push(
|
|
197
|
+
`Live aggregate: off=${live.aggregate.offTokens} on=${live.aggregate.onTokens} ` +
|
|
198
|
+
`reduction=${live.aggregate.reductionPct}%`,
|
|
199
|
+
);
|
|
200
|
+
return lines.join('\n');
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
const baselinePairs = (baseline && typeof baseline === 'object' && baseline.pairs) || {};
|
|
204
|
+
const livePairs = live.pairs;
|
|
205
|
+
const allNames = new Set([...Object.keys(baselinePairs), ...Object.keys(livePairs)]);
|
|
206
|
+
let anyDrift = false;
|
|
207
|
+
|
|
208
|
+
if (!baseline || typeof baseline.label !== 'string' || !baseline.label.includes('PROXY-TOKENIZER')) {
|
|
209
|
+
anyDrift = true;
|
|
210
|
+
lines.push('DRIFT: committed baseline is missing the required "PROXY-TOKENIZER" label.');
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
for (const name of [...allNames].sort()) {
|
|
214
|
+
const b = baselinePairs[name];
|
|
215
|
+
const l = livePairs[name];
|
|
216
|
+
if (!b) {
|
|
217
|
+
anyDrift = true;
|
|
218
|
+
lines.push(`DRIFT: pair "${name}" is new (not in committed baseline) — live off=${l.offTokens} on=${l.onTokens} reduction=${l.reductionPct}%`);
|
|
219
|
+
} else if (!l) {
|
|
220
|
+
anyDrift = true;
|
|
221
|
+
lines.push(`DRIFT: pair "${name}" was removed (present in committed baseline, not found live) — baseline off=${b.offTokens} on=${b.onTokens} reduction=${b.reductionPct}%`);
|
|
222
|
+
} else if (b.offTokens !== l.offTokens || b.onTokens !== l.onTokens || b.reductionPct !== l.reductionPct) {
|
|
223
|
+
anyDrift = true;
|
|
224
|
+
lines.push(
|
|
225
|
+
`DRIFT: pair "${name}": off ${b.offTokens} -> ${l.offTokens} (${l.offTokens - b.offTokens >= 0 ? '+' : ''}${l.offTokens - b.offTokens}), ` +
|
|
226
|
+
`on ${b.onTokens} -> ${l.onTokens} (${l.onTokens - b.onTokens >= 0 ? '+' : ''}${l.onTokens - b.onTokens}), ` +
|
|
227
|
+
`reduction ${b.reductionPct}% -> ${l.reductionPct}% (${round2(l.reductionPct - b.reductionPct) >= 0 ? '+' : ''}${round2(l.reductionPct - b.reductionPct)}pp)`,
|
|
228
|
+
);
|
|
229
|
+
}
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
const ba = (baseline && baseline.aggregate) || {};
|
|
233
|
+
const la = live.aggregate;
|
|
234
|
+
if (ba.offTokens !== la.offTokens || ba.onTokens !== la.onTokens || ba.reductionPct !== la.reductionPct) {
|
|
235
|
+
anyDrift = true;
|
|
236
|
+
lines.push(
|
|
237
|
+
`DRIFT: aggregate: off ${ba.offTokens} -> ${la.offTokens}, on ${ba.onTokens} -> ${la.onTokens}, ` +
|
|
238
|
+
`reduction ${ba.reductionPct}% -> ${la.reductionPct}%`,
|
|
239
|
+
);
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
if (!anyDrift) {
|
|
243
|
+
lines.push(`Baseline at ${baselinePath} is up to date with the live recompute.`);
|
|
244
|
+
} else {
|
|
245
|
+
lines.push('');
|
|
246
|
+
lines.push('Run `node scripts/benchmark-compact-content-variants.cjs --write` to refresh the committed baseline.');
|
|
247
|
+
lines.push('(This is a REPORT, not a gate — exiting 0 regardless of drift, per this script\'s own contract.)');
|
|
248
|
+
}
|
|
249
|
+
return lines.join('\n');
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
function parseArgs(argv) {
|
|
253
|
+
const opts = { write: false, check: false, baselinePath: BASELINE_PATH };
|
|
254
|
+
for (const arg of argv) {
|
|
255
|
+
if (arg === '--write') opts.write = true;
|
|
256
|
+
else if (arg === '--check') opts.check = true;
|
|
257
|
+
else if (arg.startsWith('--baseline-path=')) opts.baselinePath = arg.slice('--baseline-path='.length);
|
|
258
|
+
}
|
|
259
|
+
return opts;
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
function main() {
|
|
263
|
+
const opts = parseArgs(process.argv.slice(2));
|
|
264
|
+
|
|
265
|
+
if (opts.write) {
|
|
266
|
+
const report = buildReport();
|
|
267
|
+
fs.mkdirSync(path.dirname(BASELINE_PATH), { recursive: true });
|
|
268
|
+
fs.writeFileSync(BASELINE_PATH, JSON.stringify(report, null, 2) + '\n');
|
|
269
|
+
process.stdout.write(`Wrote ${BASELINE_PATH}\n`);
|
|
270
|
+
return;
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
if (opts.check) {
|
|
274
|
+
const live = buildReport();
|
|
275
|
+
process.stdout.write(formatDriftReport(opts.baselinePath, live) + '\n');
|
|
276
|
+
return;
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
process.stdout.write(JSON.stringify(buildReport(), null, 2) + '\n');
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
/* c8 ignore next 3 -- CLI entry guard; this repo measures coverage with c8, which does not honor istanbul pragmas */
|
|
283
|
+
if (require.main === module) {
|
|
284
|
+
runMain(main);
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
module.exports = {
|
|
288
|
+
discoverRegisteredVariants,
|
|
289
|
+
computePairTokens,
|
|
290
|
+
computeAggregate,
|
|
291
|
+
buildReport,
|
|
292
|
+
formatDriftReport,
|
|
293
|
+
getTokenizerVersion,
|
|
294
|
+
parseArgs,
|
|
295
|
+
LABEL,
|
|
296
|
+
BASELINE_PATH,
|
|
297
|
+
VARIANT_ROOTS,
|
|
298
|
+
};
|
|
@@ -0,0 +1,368 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Benchmarks the token-count effect of every registered `workflow.compact_content`
|
|
6
|
+
* spine/detail split (ADR-4139 Decision 3, Phase 3 #4403's `docs/PARTITION-RULES.md`).
|
|
7
|
+
*
|
|
8
|
+
* For every split it reports the "off" token count (spine + all detail parts —
|
|
9
|
+
* what `workflow.compact_content=false` pays today, since an opted-out project's
|
|
10
|
+
* spine reads every detail part back in) against the "on" token count (spine
|
|
11
|
+
* alone — what `workflow.compact_content=true` pays, since the detail `Read`
|
|
12
|
+
* never fires). See `.gsd/phase/enhance-4404-token-benchmark/40-design.md` for
|
|
13
|
+
* the full design rationale.
|
|
14
|
+
*
|
|
15
|
+
* PROXY-TOKENIZER CAVEAT: Anthropic publishes no tokenizer for Claude 3+, so
|
|
16
|
+
* `gpt-tokenizer` (pinned exact-version devDependency, see ADR-4139
|
|
17
|
+
* Consequences) is a stand-in. The on/off COMPARISON is exact under this one
|
|
18
|
+
* pinned tokenizer applied identically to both sides; the absolute counts are
|
|
19
|
+
* NOT Claude's real token counts. Every output surface says so explicitly —
|
|
20
|
+
* see the `label` field below.
|
|
21
|
+
*
|
|
22
|
+
* Discovery is REIMPLEMENTED here rather than imported from
|
|
23
|
+
* `tests/helpers/compact-content-split.cjs`: a `scripts/` reporting tool
|
|
24
|
+
* depending on a test-only helper module inverts this repo's normal layering,
|
|
25
|
+
* and a test-only module changing shape should never be able to break a
|
|
26
|
+
* benchmark. The discovery logic is intentionally small (see
|
|
27
|
+
* `discoverRegisteredSplits` below) and mirrors `docs/PARTITION-RULES.md`'s
|
|
28
|
+
* own description of what a "registered split" is.
|
|
29
|
+
*
|
|
30
|
+
* Usage:
|
|
31
|
+
* node scripts/benchmark-compact-content.cjs # print JSON to stdout
|
|
32
|
+
* node scripts/benchmark-compact-content.cjs --write # write the committed baseline
|
|
33
|
+
* node scripts/benchmark-compact-content.cjs --check # recompute, diff vs committed baseline, print drift
|
|
34
|
+
* node scripts/benchmark-compact-content.cjs --check --baseline-path=<path>
|
|
35
|
+
* # --check against a DIFFERENT baseline
|
|
36
|
+
* # file (testability seam; defaults to the
|
|
37
|
+
* # committed path when omitted)
|
|
38
|
+
*
|
|
39
|
+
* ============================================================================
|
|
40
|
+
* CRITICAL — READ BEFORE "FIXING" ANYTHING BELOW (this is the issue's own
|
|
41
|
+
* Done-when item; getting it wrong defeats the entire point of this tool):
|
|
42
|
+
*
|
|
43
|
+
* This script — and in particular `--check` — MUST NEVER exit non-zero
|
|
44
|
+
* because a baseline is drifted, stale, or missing entirely. A drifted or
|
|
45
|
+
* absent baseline is REPORTED (printed to stdout as a human-readable diff),
|
|
46
|
+
* never treated as failure. The ONLY thing allowed to make this script exit
|
|
47
|
+
* non-zero is a genuine I/O error reading a SOURCE file the benchmark is
|
|
48
|
+
* measuring (a spine or detail `.md` file that `discoverRegisteredSplits`
|
|
49
|
+
* found on disk but that fails to read) — never anything about the baseline
|
|
50
|
+
* file. This is deliberate: the benchmark is a reporting instrument wired
|
|
51
|
+
* into `npm run benchmark:compact-content`, which is never part of a gate,
|
|
52
|
+
* `lint:ci`, or `pretest` — see `tests/benchmark-compact-content.test.cjs`'s
|
|
53
|
+
* "never-fails-CI" tests, which assert this behavior directly against a
|
|
54
|
+
* deliberately-wrong and a wholly-missing baseline. Do not add
|
|
55
|
+
* `process.exit(1)` (or an `ExitError` with a non-zero code) on drift.
|
|
56
|
+
* ============================================================================
|
|
57
|
+
*/
|
|
58
|
+
|
|
59
|
+
const fs = require('node:fs');
|
|
60
|
+
const path = require('node:path');
|
|
61
|
+
|
|
62
|
+
const { countTokens } = require('gpt-tokenizer');
|
|
63
|
+
const { runMain } = require('./lib/cli-exit.cjs');
|
|
64
|
+
|
|
65
|
+
const ROOT = path.resolve(__dirname, '..');
|
|
66
|
+
const WORKFLOWS_DIR = path.join(ROOT, 'gsd-core', 'workflows');
|
|
67
|
+
const BASELINE_PATH = path.join(ROOT, 'tests', 'fixtures', 'compact-content-benchmark-baseline.json');
|
|
68
|
+
|
|
69
|
+
// The tokenizer's own package.json is read live (`require.resolve` + a plain
|
|
70
|
+
// `fs.readFileSync`/JSON.parse — NOT `require('gpt-tokenizer/package.json')`,
|
|
71
|
+
// which would work identically here but would tie this file to Node's CJS
|
|
72
|
+
// JSON-import behavior for no benefit) rather than hardcoding the version as
|
|
73
|
+
// a string literal. This repo's own `package.json` pins `gpt-tokenizer` at an
|
|
74
|
+
// EXACT version (no `^`/`~`), so the two are guaranteed to agree today — but
|
|
75
|
+
// reading the installed package's own version live means a future re-pin of
|
|
76
|
+
// that dependency never requires touching this file too, and the reported
|
|
77
|
+
// version can never silently drift from what actually ran.
|
|
78
|
+
function getTokenizerVersion() {
|
|
79
|
+
const pkgPath = require.resolve('gpt-tokenizer/package.json');
|
|
80
|
+
const pkg = JSON.parse(fs.readFileSync(pkgPath, 'utf8'));
|
|
81
|
+
return pkg.version;
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/**
|
|
85
|
+
* Discover every registered spine/detail split under `workflowsDir`.
|
|
86
|
+
*
|
|
87
|
+
* A split is registered by a `<workflowsDir>/<name>/detail/` directory
|
|
88
|
+
* (containing at least one `*.md` file) where `<name>` is exactly ONE path
|
|
89
|
+
* segment directly under `workflowsDir`. Pairs with the spine at
|
|
90
|
+
* `<workflowsDir>/<name>.md`; skipped entirely if that spine does not exist
|
|
91
|
+
* (this benchmark only measures real, complete splits — an orphaned detail
|
|
92
|
+
* directory with no spine is Phase 3's registration guard's problem, not
|
|
93
|
+
* this benchmark's).
|
|
94
|
+
*
|
|
95
|
+
* Mirrors `tests/helpers/compact-content-split.cjs`'s `discoverRegisteredSplits`
|
|
96
|
+
* in shape (same "one segment below workflowsDir, `detail/` dir, at least one
|
|
97
|
+
* `.md` file" rule) but is a from-scratch, self-contained implementation —
|
|
98
|
+
* see the module header for why this is not a shared import.
|
|
99
|
+
*
|
|
100
|
+
* @param {string} [workflowsDir]
|
|
101
|
+
* @returns {Array<{name: string, spinePath: string, detailPaths: string[]}>}
|
|
102
|
+
*/
|
|
103
|
+
function discoverRegisteredSplits(workflowsDir = WORKFLOWS_DIR) {
|
|
104
|
+
const found = new Map();
|
|
105
|
+
|
|
106
|
+
function walk(dir) {
|
|
107
|
+
let entries;
|
|
108
|
+
try {
|
|
109
|
+
entries = fs.readdirSync(dir, { withFileTypes: true });
|
|
110
|
+
} catch {
|
|
111
|
+
return; // unreadable directory: nothing to discover under it
|
|
112
|
+
}
|
|
113
|
+
for (const entry of entries) {
|
|
114
|
+
if (!entry.isDirectory()) continue;
|
|
115
|
+
const full = path.join(dir, entry.name);
|
|
116
|
+
|
|
117
|
+
if (entry.name === 'detail') {
|
|
118
|
+
let mdFiles = [];
|
|
119
|
+
try {
|
|
120
|
+
mdFiles = fs
|
|
121
|
+
.readdirSync(full, { withFileTypes: true })
|
|
122
|
+
.filter((e) => e.isFile() && e.name.endsWith('.md'))
|
|
123
|
+
.map((e) => e.name);
|
|
124
|
+
} catch {
|
|
125
|
+
mdFiles = [];
|
|
126
|
+
}
|
|
127
|
+
if (mdFiles.length > 0) {
|
|
128
|
+
const segments = path.relative(workflowsDir, dir).split(path.sep).filter(Boolean);
|
|
129
|
+
if (segments.length === 1) {
|
|
130
|
+
const name = segments[0];
|
|
131
|
+
if (!found.has(name)) {
|
|
132
|
+
const spinePath = path.join(workflowsDir, `${name}.md`);
|
|
133
|
+
if (fs.existsSync(spinePath)) {
|
|
134
|
+
const detailPaths = mdFiles.map((f) => path.join(full, f)).sort();
|
|
135
|
+
found.set(name, { name, spinePath, detailPaths });
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
walk(full);
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
walk(workflowsDir);
|
|
147
|
+
return [...found.values()].sort((a, b) => a.name.localeCompare(b.name));
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
/**
|
|
151
|
+
* Compute the off/on/reduction numbers for ONE registered split.
|
|
152
|
+
*
|
|
153
|
+
* Throws on a genuine read failure of a source file (see the module-header
|
|
154
|
+
* CRITICAL note — this is the one thing allowed to throw). `reductionPct` is
|
|
155
|
+
* reported as `0` rather than `NaN`/`Infinity` when `offTokens` is `0` (an
|
|
156
|
+
* empty spine — should never happen for a real split, but must not crash).
|
|
157
|
+
*
|
|
158
|
+
* @param {{name: string, spinePath: string, detailPaths: string[]}} split
|
|
159
|
+
* @returns {{offTokens: number, onTokens: number, reductionPct: number}}
|
|
160
|
+
*/
|
|
161
|
+
function computeSplitTokens(split) {
|
|
162
|
+
const spineContent = fs.readFileSync(split.spinePath, 'utf8');
|
|
163
|
+
const onTokens = countTokens(spineContent);
|
|
164
|
+
let detailTokens = 0;
|
|
165
|
+
for (const detailPath of split.detailPaths) {
|
|
166
|
+
const detailContent = fs.readFileSync(detailPath, 'utf8');
|
|
167
|
+
detailTokens += countTokens(detailContent);
|
|
168
|
+
}
|
|
169
|
+
const offTokens = onTokens + detailTokens;
|
|
170
|
+
const reductionPct = offTokens === 0 ? 0 : round2(((offTokens - onTokens) / offTokens) * 100);
|
|
171
|
+
return { offTokens, onTokens, reductionPct };
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
function round2(n) {
|
|
175
|
+
return Math.round(n * 100) / 100;
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
/**
|
|
179
|
+
* Aggregate per-split numbers. `aggregateReductionPct` is computed from the
|
|
180
|
+
* SUMMED off/on totals, never averaged across per-split percentages — a
|
|
181
|
+
* 2-split fixture with very different sizes is what
|
|
182
|
+
* `tests/benchmark-compact-content.test.cjs` uses to pin this down. Reports
|
|
183
|
+
* `0` (not `NaN`) when there are zero registered splits or `aggregateOff` is
|
|
184
|
+
* `0` — a valid, non-crashing state, not an error.
|
|
185
|
+
*
|
|
186
|
+
* @param {Record<string, {offTokens: number, onTokens: number}>} splitResults
|
|
187
|
+
* @returns {{offTokens: number, onTokens: number, reductionPct: number}}
|
|
188
|
+
*/
|
|
189
|
+
function computeAggregate(splitResults) {
|
|
190
|
+
let offTokens = 0;
|
|
191
|
+
let onTokens = 0;
|
|
192
|
+
for (const key of Object.keys(splitResults)) {
|
|
193
|
+
offTokens += splitResults[key].offTokens;
|
|
194
|
+
onTokens += splitResults[key].onTokens;
|
|
195
|
+
}
|
|
196
|
+
const reductionPct = offTokens === 0 ? 0 : round2(((offTokens - onTokens) / offTokens) * 100);
|
|
197
|
+
return { offTokens, onTokens, reductionPct };
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
const LABEL =
|
|
201
|
+
"PROXY-TOKENIZER DELTA — gpt-tokenizer is a stand-in; Anthropic publishes no tokenizer for Claude 3+. " +
|
|
202
|
+
"The on/off COMPARISON is exact under this pinned tokenizer; absolute counts are not Claude's real token counts.";
|
|
203
|
+
|
|
204
|
+
/**
|
|
205
|
+
* Build the full report object — the "committed baseline" shape. Keys of
|
|
206
|
+
* `splits` are sorted (`discoverRegisteredSplits` already returns
|
|
207
|
+
* name-sorted records; `Object.keys` insertion order on a plain object built
|
|
208
|
+
* in that order preserves it, and `--check`'s comparison re-sorts anyway so
|
|
209
|
+
* this is belt-and-suspenders, not load-bearing).
|
|
210
|
+
*
|
|
211
|
+
* @param {string} [workflowsDir]
|
|
212
|
+
* @returns {object}
|
|
213
|
+
*/
|
|
214
|
+
function buildReport(workflowsDir = WORKFLOWS_DIR) {
|
|
215
|
+
const splits = discoverRegisteredSplits(workflowsDir);
|
|
216
|
+
const splitReports = {};
|
|
217
|
+
for (const split of splits) {
|
|
218
|
+
splitReports[split.name] = computeSplitTokens(split);
|
|
219
|
+
}
|
|
220
|
+
return {
|
|
221
|
+
schema_version: 1,
|
|
222
|
+
generated_by: 'scripts/benchmark-compact-content.cjs',
|
|
223
|
+
tokenizer: { name: 'gpt-tokenizer', version: getTokenizerVersion() },
|
|
224
|
+
label: LABEL,
|
|
225
|
+
splits: splitReports,
|
|
226
|
+
aggregate: computeAggregate(splitReports),
|
|
227
|
+
};
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
/**
|
|
231
|
+
* Format a human-readable drift summary between a (possibly missing/invalid)
|
|
232
|
+
* committed baseline and a freshly-computed live report. Never throws — a
|
|
233
|
+
* missing or unparseable baseline is reported as "no baseline found" /
|
|
234
|
+
* "baseline could not be parsed", not propagated as an error, per the
|
|
235
|
+
* module's CRITICAL contract.
|
|
236
|
+
*
|
|
237
|
+
* @param {string} baselinePath
|
|
238
|
+
* @param {object} live
|
|
239
|
+
* @returns {string}
|
|
240
|
+
*/
|
|
241
|
+
function formatDriftReport(baselinePath, live) {
|
|
242
|
+
const lines = [];
|
|
243
|
+
let baseline = null;
|
|
244
|
+
let baselineReadError = null;
|
|
245
|
+
try {
|
|
246
|
+
const raw = fs.readFileSync(baselinePath, 'utf8');
|
|
247
|
+
try {
|
|
248
|
+
baseline = JSON.parse(raw);
|
|
249
|
+
} catch (parseErr) {
|
|
250
|
+
baselineReadError = `baseline at ${baselinePath} could not be parsed as JSON: ${parseErr.message}`;
|
|
251
|
+
}
|
|
252
|
+
} catch {
|
|
253
|
+
baselineReadError = `no baseline found at ${baselinePath}`;
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
if (baselineReadError) {
|
|
257
|
+
lines.push(`DRIFT: ${baselineReadError} — treating as fully drifted (this is reported, not an error).`);
|
|
258
|
+
lines.push('Live splits:');
|
|
259
|
+
for (const name of Object.keys(live.splits).sort()) {
|
|
260
|
+
const s = live.splits[name];
|
|
261
|
+
lines.push(` + ${name}: off=${s.offTokens} on=${s.onTokens} reduction=${s.reductionPct}%`);
|
|
262
|
+
}
|
|
263
|
+
lines.push(
|
|
264
|
+
`Live aggregate: off=${live.aggregate.offTokens} on=${live.aggregate.onTokens} ` +
|
|
265
|
+
`reduction=${live.aggregate.reductionPct}%`,
|
|
266
|
+
);
|
|
267
|
+
return lines.join('\n');
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
const baselineSplits = (baseline && typeof baseline === 'object' && baseline.splits) || {};
|
|
271
|
+
const liveSplits = live.splits;
|
|
272
|
+
const allNames = new Set([...Object.keys(baselineSplits), ...Object.keys(liveSplits)]);
|
|
273
|
+
let anyDrift = false;
|
|
274
|
+
|
|
275
|
+
if (!baseline || typeof baseline.label !== 'string' || !baseline.label.includes('PROXY-TOKENIZER')) {
|
|
276
|
+
anyDrift = true;
|
|
277
|
+
lines.push('DRIFT: committed baseline is missing the required "PROXY-TOKENIZER" label.');
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
for (const name of [...allNames].sort()) {
|
|
281
|
+
const b = baselineSplits[name];
|
|
282
|
+
const l = liveSplits[name];
|
|
283
|
+
if (!b) {
|
|
284
|
+
anyDrift = true;
|
|
285
|
+
lines.push(`DRIFT: split "${name}" is new (not in committed baseline) — live off=${l.offTokens} on=${l.onTokens} reduction=${l.reductionPct}%`);
|
|
286
|
+
} else if (!l) {
|
|
287
|
+
anyDrift = true;
|
|
288
|
+
lines.push(`DRIFT: split "${name}" was removed (present in committed baseline, not found live) — baseline off=${b.offTokens} on=${b.onTokens} reduction=${b.reductionPct}%`);
|
|
289
|
+
} else if (b.offTokens !== l.offTokens || b.onTokens !== l.onTokens || b.reductionPct !== l.reductionPct) {
|
|
290
|
+
anyDrift = true;
|
|
291
|
+
lines.push(
|
|
292
|
+
`DRIFT: split "${name}": off ${b.offTokens} -> ${l.offTokens} (${l.offTokens - b.offTokens >= 0 ? '+' : ''}${l.offTokens - b.offTokens}), ` +
|
|
293
|
+
`on ${b.onTokens} -> ${l.onTokens} (${l.onTokens - b.onTokens >= 0 ? '+' : ''}${l.onTokens - b.onTokens}), ` +
|
|
294
|
+
`reduction ${b.reductionPct}% -> ${l.reductionPct}% (${round2(l.reductionPct - b.reductionPct) >= 0 ? '+' : ''}${round2(l.reductionPct - b.reductionPct)}pp)`,
|
|
295
|
+
);
|
|
296
|
+
}
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
const ba = (baseline && baseline.aggregate) || {};
|
|
300
|
+
const la = live.aggregate;
|
|
301
|
+
if (ba.offTokens !== la.offTokens || ba.onTokens !== la.onTokens || ba.reductionPct !== la.reductionPct) {
|
|
302
|
+
anyDrift = true;
|
|
303
|
+
lines.push(
|
|
304
|
+
`DRIFT: aggregate: off ${ba.offTokens} -> ${la.offTokens}, on ${ba.onTokens} -> ${la.onTokens}, ` +
|
|
305
|
+
`reduction ${ba.reductionPct}% -> ${la.reductionPct}%`,
|
|
306
|
+
);
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
if (!anyDrift) {
|
|
310
|
+
lines.push(`Baseline at ${baselinePath} is up to date with the live recompute.`);
|
|
311
|
+
} else {
|
|
312
|
+
lines.push('');
|
|
313
|
+
lines.push('Run `node scripts/benchmark-compact-content.cjs --write` to refresh the committed baseline.');
|
|
314
|
+
lines.push('(This is a REPORT, not a gate — exiting 0 regardless of drift, per this script\'s own contract.)');
|
|
315
|
+
}
|
|
316
|
+
return lines.join('\n');
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
function parseArgs(argv) {
|
|
320
|
+
const opts = { write: false, check: false, baselinePath: BASELINE_PATH };
|
|
321
|
+
for (const arg of argv) {
|
|
322
|
+
if (arg === '--write') opts.write = true;
|
|
323
|
+
else if (arg === '--check') opts.check = true;
|
|
324
|
+
else if (arg.startsWith('--baseline-path=')) opts.baselinePath = arg.slice('--baseline-path='.length);
|
|
325
|
+
}
|
|
326
|
+
return opts;
|
|
327
|
+
}
|
|
328
|
+
|
|
329
|
+
function main() {
|
|
330
|
+
const opts = parseArgs(process.argv.slice(2));
|
|
331
|
+
|
|
332
|
+
if (opts.write) {
|
|
333
|
+
const report = buildReport();
|
|
334
|
+
fs.mkdirSync(path.dirname(BASELINE_PATH), { recursive: true });
|
|
335
|
+
fs.writeFileSync(BASELINE_PATH, JSON.stringify(report, null, 2) + '\n');
|
|
336
|
+
process.stdout.write(`Wrote ${BASELINE_PATH}\n`);
|
|
337
|
+
return;
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
if (opts.check) {
|
|
341
|
+
const live = buildReport();
|
|
342
|
+
// See the module-header CRITICAL note: this branch NEVER throws or sets a
|
|
343
|
+
// non-zero exit code on a drifted/missing/invalid baseline — only
|
|
344
|
+
// `buildReport()` above (a genuine source-file read failure) can throw.
|
|
345
|
+
process.stdout.write(formatDriftReport(opts.baselinePath, live) + '\n');
|
|
346
|
+
return;
|
|
347
|
+
}
|
|
348
|
+
|
|
349
|
+
process.stdout.write(JSON.stringify(buildReport(), null, 2) + '\n');
|
|
350
|
+
}
|
|
351
|
+
|
|
352
|
+
/* c8 ignore next 3 -- CLI entry guard; this repo measures coverage with c8, which does not honor istanbul pragmas */
|
|
353
|
+
if (require.main === module) {
|
|
354
|
+
runMain(main);
|
|
355
|
+
}
|
|
356
|
+
|
|
357
|
+
module.exports = {
|
|
358
|
+
discoverRegisteredSplits,
|
|
359
|
+
computeSplitTokens,
|
|
360
|
+
computeAggregate,
|
|
361
|
+
buildReport,
|
|
362
|
+
formatDriftReport,
|
|
363
|
+
getTokenizerVersion,
|
|
364
|
+
parseArgs,
|
|
365
|
+
LABEL,
|
|
366
|
+
BASELINE_PATH,
|
|
367
|
+
WORKFLOWS_DIR,
|
|
368
|
+
};
|