forge-workflow 0.1.0-beta.3 → 0.1.0-beta.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (196) hide show
  1. package/AGENTS.md +14 -7
  2. package/CHANGELOG.md +43 -1
  3. package/README.md +6 -2
  4. package/bin/forge-cmd.js +21 -1
  5. package/bin/forge.js +16 -369
  6. package/docs/INDEX.md +1 -1
  7. package/docs/guides/BEADS_GITHUB_SYNC.md +2 -31
  8. package/docs/guides/MIGRATION.md +4 -4
  9. package/docs/guides/SETUP.md +16 -16
  10. package/docs/reference/COMMANDS.md +9 -4
  11. package/docs/reference/INSIGHTS_RECAP.md +9 -20
  12. package/docs/reference/RELEASE.md +5 -3
  13. package/docs/reference/TOOLCHAIN.md +8 -0
  14. package/docs/reference/protected-state-surfaces.md +4 -4
  15. package/docs/reference/shepherd.md +117 -17
  16. package/lefthook.yml +12 -0
  17. package/lib/activation/ensure-forge-home.js +33 -15
  18. package/lib/adapters/greptile-review-adapter.js +1 -1
  19. package/lib/adapters/pr-state-adapter.js +397 -100
  20. package/lib/agents-config.js +5 -0
  21. package/lib/audit-evidence.js +71 -110
  22. package/lib/capped-jsonl-log.js +236 -0
  23. package/lib/commands/_issue.js +31 -46
  24. package/lib/commands/_manifest.js +1 -1
  25. package/lib/commands/_registry.js +2 -2
  26. package/lib/commands/_resolve-command-opts.js +36 -29
  27. package/lib/commands/claim.js +2 -4
  28. package/lib/commands/clean.js +196 -32
  29. package/lib/commands/dev.js +4 -33
  30. package/lib/commands/hooks.js +358 -13
  31. package/lib/commands/insights.js +8 -3
  32. package/lib/commands/merge.js +600 -40
  33. package/lib/commands/plan.js +23 -115
  34. package/lib/commands/pr.js +1 -1
  35. package/lib/commands/preflight.js +11 -2
  36. package/lib/commands/prime.js +23 -3
  37. package/lib/commands/push.js +41 -51
  38. package/lib/commands/recall.js +60 -16
  39. package/lib/commands/recap.js +6 -1
  40. package/lib/commands/release.js +18 -4
  41. package/lib/commands/serve.js +5 -2
  42. package/lib/commands/setup.js +191 -95
  43. package/lib/commands/shepherd.js +49 -4
  44. package/lib/commands/ship.js +22 -23
  45. package/lib/commands/skill.js +383 -0
  46. package/lib/commands/status.js +54 -33
  47. package/lib/commands/test.js +56 -34
  48. package/lib/commands/worktree.js +247 -43
  49. package/lib/core/runtime-graph.js +89 -15
  50. package/lib/doc-assertions.js +297 -0
  51. package/lib/existing-tdd-gate.js +253 -0
  52. package/lib/forge-context.js +1 -4
  53. package/lib/forge-issues.js +64 -491
  54. package/lib/git-defaults.js +56 -0
  55. package/lib/harness-capability-matrix.js +5 -5
  56. package/lib/hook-renderer.js +147 -16
  57. package/lib/insights.js +96 -80
  58. package/lib/issue-backend.js +42 -3
  59. package/lib/kernel/backing-issue.js +14 -2
  60. package/lib/kernel/broker.js +44 -0
  61. package/lib/kernel/cli-broker-factory.js +12 -1
  62. package/lib/kernel/close-on-merge.js +154 -0
  63. package/lib/kernel/fs-class.js +42 -25
  64. package/lib/kernel/migrations.js +30 -2
  65. package/lib/kernel/schema.js +35 -0
  66. package/lib/kernel/sqlite-driver.js +292 -18
  67. package/lib/lefthook-wiring.js +21 -1
  68. package/lib/memory/router.js +16 -1
  69. package/lib/memory-digest.js +47 -15
  70. package/lib/memory-recall-events.js +145 -0
  71. package/lib/memory-recall.js +212 -0
  72. package/lib/merge-rules.js +8 -4
  73. package/lib/npm-publish-workflow.js +272 -0
  74. package/lib/orientation.js +371 -49
  75. package/lib/plugin-catalog.js +14 -4
  76. package/lib/pr-bundle.js +9 -6
  77. package/lib/pr-monitor/journal.js +18 -2
  78. package/lib/pr-monitor/reconcile-executor.js +842 -0
  79. package/lib/pr-monitor/reconcile-tick.js +138 -0
  80. package/lib/pr-monitor/reconcile.js +0 -0
  81. package/lib/pr-monitor/render-summary.js +196 -0
  82. package/lib/pr-monitor/shepherd-lease.js +252 -0
  83. package/lib/pr-monitor/watch-lifecycle.js +14 -2
  84. package/lib/pr-pull.js +98 -24
  85. package/lib/pr-shepherd.js +34 -8
  86. package/lib/preflight/gates.js +65 -18
  87. package/lib/preflight/runner.js +5 -0
  88. package/lib/project-memory.js +40 -0
  89. package/lib/protected-state-authority.js +305 -0
  90. package/lib/protected-state-surfaces.js +64 -44
  91. package/lib/release-readiness.js +51 -4
  92. package/lib/rules-sync.js +4 -0
  93. package/lib/runtime-health.js +15 -46
  94. package/lib/shell-utils.js +1 -1
  95. package/lib/skill-eval.js +750 -0
  96. package/lib/skills-sync.js +6 -3
  97. package/lib/smart-merge.js +28 -4
  98. package/lib/status/identity.js +46 -0
  99. package/lib/status/presenter.js +0 -35
  100. package/lib/status/snapshot.js +11 -16
  101. package/lib/symlink-utils.js +74 -26
  102. package/lib/upgrade-safety.js +47 -9
  103. package/lib/using-forge.js +328 -0
  104. package/lib/workflow/enforce-stage.js +5 -5
  105. package/lib/workflow/state-manager.js +23 -23
  106. package/package.json +6 -7
  107. package/rules/using-forge.md +24 -0
  108. package/scripts/doc-asserting-tests.js +158 -0
  109. package/scripts/forge-team/index.sh +0 -5
  110. package/scripts/forge-team/tests/dispatcher.test.sh +1 -1
  111. package/scripts/forge-team/tests/workflow-integration.test.sh +0 -1
  112. package/scripts/lib/behavioral-eval-runner.js +310 -0
  113. package/scripts/lib/behavioral-eval-runtime.js +456 -0
  114. package/scripts/lib/eval-evidence.js +328 -0
  115. package/scripts/lib/eval-runner.js +81 -41
  116. package/scripts/lib/immutable-eval-corpus.js +309 -0
  117. package/scripts/lib/promotion-evidence-loader.js +94 -0
  118. package/scripts/lib/promotion-scorecard.js +314 -0
  119. package/scripts/npm-release-receipt.js +134 -0
  120. package/scripts/process-tree.js +761 -0
  121. package/scripts/protected-state-check.js +47 -22
  122. package/scripts/run-command-eval.js +29 -1
  123. package/scripts/sync-d20-audit.js +172 -0
  124. package/scripts/test-full-suite.js +249 -37
  125. package/scripts/test.js +184 -44
  126. package/skills/claim-safety/SKILL.md +4 -0
  127. package/skills/claim-safety/evals/scorecard.json +41 -0
  128. package/skills/coverage.json +83 -0
  129. package/skills/dev/SKILL.md +4 -0
  130. package/skills/dev/evals/scorecard.json +41 -0
  131. package/skills/gates/SKILL.md +80 -0
  132. package/skills/gates/evals/evals.json +38 -0
  133. package/skills/gates/evals/scorecard.json +41 -0
  134. package/skills/hermes-forge/SKILL.md +1 -0
  135. package/skills/hermes-forge/evals/scorecard.json +41 -0
  136. package/skills/issue-basics/SKILL.md +1 -0
  137. package/skills/issue-basics/evals/scorecard.json +41 -0
  138. package/skills/kernel/SKILL.md +38 -0
  139. package/skills/kernel/evals/scorecard.json +41 -0
  140. package/skills/memory/SKILL.md +16 -1
  141. package/skills/memory/evals/scorecard.json +41 -0
  142. package/skills/parallel-deep-research/SKILL.md +1 -0
  143. package/skills/parallel-deep-research/evals/scorecard.json +41 -0
  144. package/skills/plan/SKILL.md +6 -0
  145. package/skills/plan/evals/scorecard.json +41 -0
  146. package/skills/portability/SKILL.md +47 -0
  147. package/skills/portability/evals/evals.json +34 -0
  148. package/skills/portability/evals/scorecard.json +41 -0
  149. package/skills/research/SKILL.md +1 -0
  150. package/skills/research/evals/scorecard.json +41 -0
  151. package/skills/review/SKILL.md +10 -11
  152. package/skills/review/evals/scorecard.json +41 -0
  153. package/skills/rollback/SKILL.md +5 -11
  154. package/skills/rollback/evals/scorecard.json +41 -0
  155. package/skills/setup/SKILL.md +91 -0
  156. package/skills/setup/evals/evals.json +42 -0
  157. package/skills/setup/evals/scorecard.json +41 -0
  158. package/skills/shepherd/SKILL.md +84 -38
  159. package/skills/shepherd/evals/evals.json +21 -9
  160. package/skills/shepherd/evals/scorecard.json +41 -0
  161. package/skills/ship/SKILL.md +10 -12
  162. package/skills/ship/evals/scorecard.json +41 -0
  163. package/skills/smith/SKILL.md +8 -0
  164. package/skills/smith/evals/scorecard.json +41 -0
  165. package/skills/sonarcloud/SKILL.md +1 -0
  166. package/skills/sonarcloud/evals/scorecard.json +41 -0
  167. package/skills/sonarcloud-analysis/SKILL.md +1 -0
  168. package/skills/sonarcloud-analysis/evals/scorecard.json +41 -0
  169. package/skills/status/SKILL.md +3 -0
  170. package/skills/status/evals/scorecard.json +41 -0
  171. package/skills/triage-ready/SKILL.md +2 -0
  172. package/skills/triage-ready/evals/scorecard.json +41 -0
  173. package/skills/using-forge/SKILL.md +104 -0
  174. package/skills/using-forge/evals/scorecard.json +41 -0
  175. package/skills/validate/SKILL.md +4 -0
  176. package/skills/validate/evals/scorecard.json +41 -0
  177. package/skills/verify/SKILL.md +4 -0
  178. package/skills/verify/evals/scorecard.json +41 -0
  179. package/skills/worktree/SKILL.md +92 -0
  180. package/skills/worktree/evals/evals.json +38 -0
  181. package/skills/worktree/evals/scorecard.json +41 -0
  182. package/lib/adapters/beads-issue-adapter.js +0 -127
  183. package/lib/beads-nudge.js +0 -91
  184. package/lib/beads-setup.js +0 -538
  185. package/lib/beads-sync-scaffold.js +0 -189
  186. package/lib/commands/board.js +0 -64
  187. package/lib/pat-setup.js +0 -207
  188. package/lib/pr-monitor/render-sticky.js +0 -192
  189. package/lib/pr-monitor/upsert-sticky.js +0 -169
  190. package/lib/status/beads-snapshot.js +0 -145
  191. package/scripts/beads-context.sh +0 -577
  192. package/scripts/beads-migrate-to-dolt.sh +0 -7
  193. package/scripts/beads-upgrade-smoke.sh +0 -284
  194. package/scripts/forge-team/lib/dashboard.sh +0 -316
  195. package/scripts/forge-team/tests/dashboard.test.sh +0 -155
  196. package/scripts/lib/beads-migrate-to-dolt.mjs +0 -503
@@ -37,13 +37,22 @@ const { gatherMonitorSnapshot } = require('../pr-monitor/gather');
37
37
  const { pollEvents } = require('../pr-monitor/monitor');
38
38
  const { watchLoop } = require('../pr-monitor/watch');
39
39
  const { startPrWatcherDetached } = require('../pr-monitor/watch-lifecycle');
40
+ const reconcileExecutor = require('../pr-monitor/reconcile-executor');
40
41
  const monitorJournal = require('../pr-monitor/journal');
42
+ const brokerMod = require('../kernel/broker');
41
43
  const { EVENT_TYPES: T } = require('../pr-monitor/events');
42
44
  const { autoShepherdRailEnabled } = require('./ship');
43
45
 
44
46
  const DEFAULT_RERUN_BUDGET = 3;
45
47
 
46
- const defaultGhRunner = (cmd, a) => execFileSync(cmd, a, { encoding: 'utf8', timeout: 30000 });
48
+ // windowsHide: true on EVERY spawn here is load-bearing, not cosmetic. The
49
+ // shepherd watcher runs detached in the background and re-polls every ~60s; on
50
+ // Windows a child process spawned WITHOUT windowsHide flashes a visible console
51
+ // window each time (Node's default is windowsHide:false). Background work's
52
+ // preferred home is the harness's managed shell (hidden + reaped); a
53
+ // Forge-spawned detached watcher is the no-session fallback and must be
54
+ // COMPLETELY silent — no console window, ever. See kernel issue 931e7924.
55
+ const defaultGhRunner = (cmd, a) => execFileSync(cmd, a, { encoding: 'utf8', timeout: 30000, windowsHide: true });
47
56
 
48
57
  /**
49
58
  * Resolve owner/repo and base branch for the shepherd pass.
@@ -189,7 +198,18 @@ async function buildMonitorContext(pr, projectRoot, deps) {
189
198
  if (!validation.valid) {
190
199
  return { error: `Invalid pr-state adapter: ${validation.errors.join('; ')}` };
191
200
  }
192
- dir = dir || monitorJournal.journalDir({ root: projectRoot || process.cwd(), repo: ctx.repo, pr: ctx.pr });
201
+ let gitCommonDir = deps.gitCommonDir;
202
+ if (!gitCommonDir) {
203
+ try {
204
+ const resolveGitCommonDir = deps.resolveGitCommonDir || brokerMod.resolveGitCommonDir;
205
+ gitCommonDir = resolveGitCommonDir(projectRoot || process.cwd(), { warn: () => {} });
206
+ } catch {
207
+ /* unavailable common-dir keeps the legacy per-root journal fallback */
208
+ }
209
+ }
210
+ dir = dir || monitorJournal.journalDir({
211
+ root: projectRoot || process.cwd(), gitCommonDir, repo: ctx.repo, pr: ctx.pr,
212
+ });
193
213
  gather = gather || (() => gatherMonitorSnapshot({ ...ctx, adapter, self: deps.self }));
194
214
  enrich = enrich || makeCheckFailureEnricher({
195
215
  ...ctx,
@@ -282,7 +302,7 @@ function wireSignals() {
282
302
  function defaultListOpenPrs(exec = execFileSync) {
283
303
  try {
284
304
  const out = exec('gh', ['pr', 'list', '--state', 'open', '--json', 'number', '-q', '.[].number'], {
285
- encoding: 'utf8', timeout: 20000, stdio: ['pipe', 'pipe', 'pipe'],
305
+ encoding: 'utf8', timeout: 20000, stdio: ['pipe', 'pipe', 'pipe'], windowsHide: true,
286
306
  });
287
307
  return String(out)
288
308
  .split(/\r?\n/)
@@ -380,6 +400,28 @@ async function handleWatch(args, projectRoot, deps = {}) {
380
400
  };
381
401
  }
382
402
 
403
+ /**
404
+ * `forge shepherd daemon` — the SINGLETON reconcile daemon (W-S4b). It acquires
405
+ * the machine-wide shepherd lease for this repo (exiting immediately if a live
406
+ * daemon already owns it), heartbeats, and converges the PR world every ~60s:
407
+ * self-registering hand-opened PRs, restarting killed watchers, reaping verified
408
+ * orphans, retiring merged/closed PRs. It self-retires (releases the lease, kills
409
+ * its verified children, exits) once no PRs remain open. Launched detached by the
410
+ * per-command `fireAndForget` trigger — not meant to be run by hand.
411
+ *
412
+ * @param {string} projectRoot
413
+ * @param {object} [deps] - injected for tests (acquire/heartbeat/gather/etc.).
414
+ * @returns {Promise<object>} result envelope.
415
+ */
416
+ async function handleDaemon(projectRoot, deps = {}) {
417
+ const res = await reconcileExecutor.runDaemon(projectRoot, { ...deps });
418
+ if (!res.ok) {
419
+ // A live foreign daemon owns the lease — this invocation is a clean no-op.
420
+ return { success: true, started: false, reason: res.reason || 'foreign-lease' };
421
+ }
422
+ return { success: true, started: true };
423
+ }
424
+
383
425
  /**
384
426
  * Command handler.
385
427
  *
@@ -401,6 +443,9 @@ async function handler(args, _flags, projectRoot, deps = {}) {
401
443
  if (positional[0] === 'watch') {
402
444
  return handleWatch(args, projectRoot, deps);
403
445
  }
446
+ if (positional[0] === 'daemon') {
447
+ return handleDaemon(projectRoot, deps);
448
+ }
404
449
 
405
450
  const pr = positional[0];
406
451
 
@@ -408,7 +453,7 @@ async function handler(args, _flags, projectRoot, deps = {}) {
408
453
  return { success: false, error: 'Usage: forge shepherd <pr> [--auto-rebase] [--bundle --json] [--pull --json]' };
409
454
  }
410
455
 
411
- const gh = deps.gh || ((cmd, a) => execFileSync(cmd, a, { encoding: 'utf8', timeout: 30000 }));
456
+ const gh = deps.gh || ((cmd, a) => execFileSync(cmd, a, { encoding: 'utf8', timeout: 30000, windowsHide: true }));
412
457
  const git = deps.git || gh;
413
458
  const buildContext = deps.buildContext || defaultBuildContext;
414
459
  const runPass = deps.runPass || runShepherdPass;
@@ -12,7 +12,7 @@ const { execFileSync } = require('node:child_process');
12
12
  const fs = require('node:fs');
13
13
  const path = require('node:path');
14
14
 
15
- const { startPrWatcherDetached } = require('../pr-monitor/watch-lifecycle');
15
+ const { fireAndForget } = require('../pr-monitor/reconcile-executor');
16
16
  const { getResolvedRuntimeGraph } = require('../core/runtime-graph');
17
17
 
18
18
  const AUTO_SHEPHERD_RAIL = 'rail.auto_shepherd';
@@ -511,34 +511,31 @@ async function createPR(options) { // NOSONAR S3776
511
511
  }
512
512
 
513
513
  /**
514
- * Best-effort, non-blocking auto-start of the constant PR monitor once a real PR
515
- * exists. Skipped on a dry run or when no PR number is known. MUST NEVER fail
516
- * ship: `startWatcher` (startPrWatcherDetached) already never throws, and this
517
- * guard keeps even a surprise error from surfacing to the ship caller.
514
+ * Best-effort wake of the repository-wide singleton after a successful real ship.
515
+ * The shared trigger owns containment and gate checks; dry-run skips before it.
518
516
  *
519
- * Gated by the default-ON `rail.auto_shepherd` rail: when a maintainer has
520
- * disabled it (`forge gate disable rail.auto_shepherd`), the watcher is skipped
521
- * so the auto-start is honestly toggleable. The rail check is fail-open and
522
- * wrapped in the same try/catch, so neither a disabled rail nor a config-read
523
- * error ever fails ship.
524
- *
525
- * @param {{ dryRun: boolean, prNumber?: string|number, startWatcher: Function, railEnabled?: Function }} params
517
+ * @param {{ dryRun: boolean, projectRoot?: string, fireAndForget?: Function }} params
526
518
  * @returns {{ started: boolean, reason?: string }}
527
519
  */
528
- function maybeStartPrWatcher({ dryRun, prNumber, startWatcher, railEnabled = autoShepherdRailEnabled }) {
529
- if (dryRun || !prNumber) return { started: false, reason: 'skipped' };
520
+ function maybeTriggerShepherdAfterShip({ dryRun, projectRoot = process.cwd(), fireAndForget: trigger = fireAndForget }) {
521
+ if (dryRun) return { started: false, reason: 'skipped' };
530
522
  try {
531
- if (!railEnabled(process.cwd())) {
532
- return { started: false, reason: 'rail.auto_shepherd disabled' };
533
- }
534
- return startWatcher({ prNumber, cwd: process.cwd() });
523
+ trigger({ projectRoot, dryRun });
524
+ return { started: true };
535
525
  } catch (err) {
536
526
  return { started: false, reason: err.message };
537
527
  }
538
528
  }
539
529
 
540
530
  async function executeShip(options) {
541
- const { featureSlug, title, dryRun = false, startWatcher = startPrWatcherDetached, railEnabled = autoShepherdRailEnabled } = options || {};
531
+ const {
532
+ featureSlug,
533
+ title,
534
+ dryRun = false,
535
+ projectRoot = process.cwd(),
536
+ fireAndForget: trigger = fireAndForget,
537
+ createPr = createPR,
538
+ } = options || {};
542
539
 
543
540
  // Validate feature slug
544
541
  if (!featureSlug || typeof featureSlug !== 'string' || featureSlug.trim() === '') {
@@ -588,9 +585,9 @@ async function executeShip(options) {
588
585
  testScenarios,
589
586
  coverage,
590
587
  });
591
- const result = await createPR({ title, body: prBody, dryRun });
588
+ const result = await createPr({ title, body: prBody, dryRun });
592
589
  if (!result.success) return result;
593
- maybeStartPrWatcher({ dryRun, prNumber: result.prNumber, startWatcher, railEnabled });
590
+ maybeTriggerShepherdAfterShip({ dryRun, projectRoot, fireAndForget: trigger });
594
591
  return {
595
592
  success: true,
596
593
  prUrl: result.prUrl,
@@ -605,11 +602,13 @@ async function executeShip(options) {
605
602
  module.exports = {
606
603
  name: 'ship',
607
604
  description: 'Create a pull request from validated feature work',
608
- handler: async (args, flags = {}) => {
605
+ handler: async (args, flags = {}, projectRoot = process.cwd(), deps = {}) => {
609
606
  const result = await executeShip({
610
607
  featureSlug: args[0],
611
608
  title: args[1],
612
609
  dryRun: Boolean(flags.dryRun || flags['--dry-run']),
610
+ projectRoot,
611
+ fireAndForget: deps.fireAndForget,
613
612
  });
614
613
  if (!result.success) {
615
614
  return result;
@@ -623,7 +622,7 @@ module.exports = {
623
622
  output: lines.join('\n'),
624
623
  };
625
624
  },
626
- maybeStartPrWatcher,
625
+ maybeTriggerShepherdAfterShip,
627
626
  autoShepherdRailEnabled,
628
627
  extractKeyDecisions,
629
628
  extractTestScenarios,
@@ -0,0 +1,383 @@
1
+ 'use strict';
2
+
3
+ /**
4
+ * Forge Skill Command -- the unified "forge skill <verb>" noun.
5
+ *
6
+ * "forge skill for <situation>" is a DETERMINISTIC intent-to-skill router: it reads the canonical
7
+ * skill catalog (skills/*\/SKILL.md frontmatter) and prints the best-fit Forge skill(s) plus WHY,
8
+ * as the reasoning fallback for harnesses without a SessionStart hook that can auto-inject the
9
+ * using-forge dispatch skill. It NEVER opens the kernel and NEVER throws.
10
+ *
11
+ * The noun is structured so later waves can add sibling verbs (forge skill eval, forge skill
12
+ * scores) without a new top-level command -- the owner's command-surface rule: unify related
13
+ * commands under one self-explanatory noun instead of scattering verbs.
14
+ *
15
+ * @module commands/skill
16
+ */
17
+
18
+ const fs = require('node:fs');
19
+ const path = require('node:path');
20
+ const { routeSkill, loadSkillCatalog } = require('../using-forge');
21
+ const skillEval = require('../skill-eval');
22
+ const { runBehavioralEvaluation } = require('../../scripts/lib/behavioral-eval-runner');
23
+ const { resolveBehavioralEvaluation } = require('../../scripts/lib/behavioral-eval-runtime');
24
+ const { loadPromotionEvidence } = require('../../scripts/lib/promotion-evidence-loader');
25
+ const { scorePromotion } = require('../../scripts/lib/promotion-scorecard');
26
+
27
+ const BEHAVIORAL_TIERS = Object.freeze([30, 100, 300]);
28
+ const BEHAVIORAL_TIER_VALUES = new Set(BEHAVIORAL_TIERS.map(String));
29
+ const BEHAVIORAL_TIER_USAGE = BEHAVIORAL_TIERS.join('|');
30
+
31
+ const USAGE = 'Usage: forge skill for "<situation>" [--json]\n' +
32
+ ' forge skill eval [name] --static [--json]\n' +
33
+ ` forge skill eval <name> --full --tier ${BEHAVIORAL_TIER_USAGE} [--json]\n` +
34
+ ' forge skill scores [--json]\n' +
35
+ ' forge skill coverage [--json]';
36
+
37
+ /** Extract the situation text (all non-flag args after the verb) and json flag. */
38
+ function parseForArgs(rest, flags) {
39
+ const positional = rest.filter(a => typeof a === 'string' && !a.startsWith('--'));
40
+ const json = flags.json === true || flags['--json'] === true || rest.includes('--json');
41
+ return { situation: positional.join(' ').trim(), json };
42
+ }
43
+
44
+ /** Render the human-readable routing answer. */
45
+ function formatRouting(result) {
46
+ const lines = ['Best skill for: "' + result.situation + '"', ''];
47
+ if (result.unknown) {
48
+ lines.push(
49
+ 'No confident match. This may not need a Forge skill -- or describe it more concretely.',
50
+ 'Fallbacks: `forge ready` for what to work on, or the kernel skill to see the whole surface.',
51
+ );
52
+ return lines.join('\n');
53
+ }
54
+ const [top, ...rest] = result.matches;
55
+ lines.push(
56
+ '-> ' + top.name + ' (' + top.why + ')',
57
+ ' Announce: "Using ' + top.name + ' to ..." then follow the skill.',
58
+ );
59
+ if (rest.length > 0) {
60
+ lines.push('', 'Also consider:');
61
+ for (const m of rest) lines.push(' - ' + m.name + ' (' + m.why + ')');
62
+ }
63
+ return lines.join('\n');
64
+ }
65
+
66
+ /** "forge skill for <situation>" -- deterministic router. */
67
+ function handleFor(rest, flags) {
68
+ const { situation, json } = parseForArgs(rest, flags);
69
+ if (!situation) {
70
+ return { success: false, error: 'Missing situation.\n' + USAGE };
71
+ }
72
+ // Read the canonical catalog from the Forge PACKAGE (not projectRoot): a set-up consumer
73
+ // project has no root skills/, so the routable skills live in the package assets.
74
+ const catalog = loadSkillCatalog();
75
+ const result = routeSkill(situation, { catalog });
76
+ if (json) {
77
+ return { success: true, result, output: JSON.stringify(result, null, 2) + '\n' };
78
+ }
79
+ return { success: true, result, output: formatRouting(result) };
80
+ }
81
+
82
+ /** Write a scorecard to skills/<name>/evals/scorecard.json (stable 2-space JSON + trailing NL). */
83
+ function writeScorecard(skillsDir, name, card) {
84
+ const dir = path.join(skillsDir, name, 'evals');
85
+ fs.mkdirSync(dir, { recursive: true });
86
+ fs.writeFileSync(path.join(dir, 'scorecard.json'), JSON.stringify(card, null, 2) + '\n');
87
+ }
88
+
89
+ function readFlagValue(rest, flags, name) {
90
+ const direct = flags[name] ?? flags[`--${name}`];
91
+ if (direct !== undefined && direct !== true) return direct;
92
+ const exact = rest.indexOf(`--${name}`);
93
+ if (exact >= 0) return rest[exact + 1];
94
+ const prefix = `--${name}=`;
95
+ const joined = rest.find(arg => typeof arg === 'string' && arg.startsWith(prefix));
96
+ return joined ? joined.slice(prefix.length) : undefined;
97
+ }
98
+
99
+ function hasFlag(rest, flags, name) {
100
+ return flags[name] === true || flags[`--${name}`] === true ||
101
+ rest.includes(`--${name}`) || rest.some(arg => typeof arg === 'string' && arg.startsWith(`--${name}=`));
102
+ }
103
+
104
+ function positionalArgs(rest, valueFlags = []) {
105
+ const positional = [];
106
+ for (let index = 0; index < rest.length; index += 1) {
107
+ const arg = rest[index];
108
+ if (typeof arg !== 'string') continue;
109
+ if (!arg.startsWith('--')) {
110
+ positional.push(arg);
111
+ continue;
112
+ }
113
+ const flagName = arg.slice(2).split('=', 1)[0];
114
+ if (valueFlags.includes(flagName) && !arg.includes('=')) index += 1;
115
+ }
116
+ return positional;
117
+ }
118
+
119
+ async function handleFullEval(rest, flags, projectRoot, opts) {
120
+ const json = hasFlag(rest, flags, 'json');
121
+ const tierValue = readFlagValue(rest, flags, 'tier');
122
+ const name = positionalArgs(rest, ['tier'])[0];
123
+ if (!name) return { success: false, error: 'Behavioral evaluation requires a skill name.\n' + USAGE };
124
+ const tier = Number(tierValue);
125
+ if (!BEHAVIORAL_TIER_VALUES.has(String(tierValue))) {
126
+ return { success: false, error: `Behavioral evaluation requires --tier ${BEHAVIORAL_TIER_USAGE}.\n` + USAGE };
127
+ }
128
+
129
+ const ctx = skillEval.resolveSkillsContext(projectRoot);
130
+ if (!ctx || !fs.existsSync(path.join(ctx.skillsDir, name, 'SKILL.md'))) {
131
+ return { success: false, error: `Skill '${name}' not found.` };
132
+ }
133
+
134
+ const scoreResult = (result) => {
135
+ const loaded = loadPromotionEvidence({ tier, findings: result.findings });
136
+ const scorecard = scorePromotion({ tier, pairs: loaded.ok ? loaded.pairs : [] });
137
+ if (!loaded.ok) scorecard.reasons = [loaded.reason];
138
+ return { ...result, scorecard };
139
+ };
140
+ const formatResult = (result) => {
141
+ const successfulScore = tier === BEHAVIORAL_TIERS[0] || result.scorecard.status === 'PASS';
142
+ const success = result.status === 'PASS' && successfulScore;
143
+ let error;
144
+ if (!success) {
145
+ error = result.status !== 'PASS'
146
+ ? `Behavioral evaluation ${result.status}.`
147
+ : `Promotion ${result.scorecard.phase || 'evaluation'} ${result.scorecard.status}.`;
148
+ }
149
+ const output = json
150
+ ? JSON.stringify(result, null, 2) + '\n'
151
+ : `Behavioral evaluation: ${result.status}\nTier: ${tier}\n` +
152
+ `Completed: ${result.completedRuns}/${result.expectedRuns}\n` +
153
+ `Promotion: ${result.scorecard.phase || 'unavailable'} / ${result.scorecard.status}`;
154
+ return { success, error, behavioral: result, output };
155
+ };
156
+
157
+ const runner = opts.runBehavioralEvaluation || runBehavioralEvaluation;
158
+ let behavioralOptions = opts.behavioralOptions || {};
159
+ if (!opts.runBehavioralEvaluation || opts.resolveBehavioralEvaluation) {
160
+ const resolver = opts.resolveBehavioralEvaluation || resolveBehavioralEvaluation;
161
+ const resolved = await resolver({
162
+ projectRoot,
163
+ skillName: name,
164
+ skillPath: path.join(ctx.skillsDir, name, 'SKILL.md'),
165
+ tier,
166
+ env: opts.env || process.env,
167
+ });
168
+ if (!resolved.ok) return formatResult(scoreResult(resolved.result));
169
+ behavioralOptions = { ...resolved.options, ...behavioralOptions };
170
+ }
171
+ const result = scoreResult(await runner({
172
+ ...behavioralOptions,
173
+ projectRoot,
174
+ skillName: name,
175
+ tier,
176
+ }));
177
+ return formatResult(result);
178
+ }
179
+
180
+ /**
181
+ * "forge skill eval [name] --static [--json]" -- compute + persist the DETERMINISTIC scorecard(s).
182
+ * All skills when no name. --static is accepted and implied; behavioral execution is isolated
183
+ * behind the explicit --full branch above so existing static scoring remains unchanged.
184
+ */
185
+ function handleEval(rest, flags, projectRoot, opts = {}) {
186
+ const full = hasFlag(rest, flags, 'full');
187
+ if (full) return handleFullEval(rest, flags, projectRoot, opts);
188
+ if (readFlagValue(rest, flags, 'tier') !== undefined) {
189
+ return { success: false, error: '--tier is valid only with --full.\n' + USAGE };
190
+ }
191
+ const json = flags.json === true || flags['--json'] === true || rest.includes('--json');
192
+ const positional = rest.filter(a => typeof a === 'string' && !a.startsWith('--'));
193
+ const name = positional[0];
194
+ const ctx = skillEval.resolveSkillsContext(projectRoot);
195
+ if (!ctx) {
196
+ return { success: false, error: 'No canonical skills/ directory found (looked in the project and the packaged Forge root).' };
197
+ }
198
+ const { skillsDir, catalog } = ctx;
199
+ const targets = name
200
+ ? [name]
201
+ : fs.readdirSync(skillsDir, { withFileTypes: true })
202
+ .filter(e => e.isDirectory() && fs.existsSync(path.join(skillsDir, e.name, 'SKILL.md')))
203
+ .map(e => e.name)
204
+ .sort();
205
+
206
+ const written = [];
207
+ const cards = {};
208
+ for (const target of targets) {
209
+ const card = skillEval.buildScorecard({ skillsDir, name: target, catalog });
210
+ if (!card) {
211
+ if (name) return { success: false, error: `Skill '${name}' not found under ${skillsDir}.` };
212
+ continue;
213
+ }
214
+ writeScorecard(skillsDir, target, card);
215
+ written.push(target);
216
+ cards[target] = card;
217
+ }
218
+
219
+ if (json) {
220
+ return { success: true, cards, output: JSON.stringify(name ? cards[name] : cards, null, 2) + '\n' };
221
+ }
222
+ const lines = ['Static scorecards written (deterministic tier):'];
223
+ for (const t of written) lines.push(' ' + t + ' composite=' + cards[t].composite);
224
+ lines.push('', 'Use --full --tier 30|100|300 for controlled behavioral evidence.');
225
+ return { success: true, cards, output: lines.join('\n') };
226
+ }
227
+
228
+ /** Render the worst-first league table from a scorecards map. */
229
+ function formatScores(scorecards, gate) {
230
+ const rows = Object.values(scorecards)
231
+ .map(c => ({
232
+ skill: c.skill,
233
+ composite: c.composite,
234
+ dq: c.static.description_quality.score,
235
+ tok: c.static.token_cost.score,
236
+ caps: c.static.caps.score,
237
+ fixtures: c.fixtures,
238
+ }))
239
+ .sort((a, b) => a.composite - b.composite || a.skill.localeCompare(b.skill));
240
+
241
+ const lines = ['Skill scores (static tier — worst first). Composite = 0.5*desc-quality + 0.3*token-cost + 0.2*caps.', ''];
242
+ lines.push(' COMPOSITE DESC-Q TOKEN CAPS FIXTURES SKILL');
243
+ for (const r of rows) {
244
+ lines.push(
245
+ ' ' + String(r.composite).padStart(9) +
246
+ ' ' + String(r.dq).padStart(6) +
247
+ ' ' + String(r.tok).padStart(5) +
248
+ ' ' + String(r.caps).padStart(4) +
249
+ ' ' + r.fixtures.padEnd(11) +
250
+ ' ' + r.skill,
251
+ );
252
+ }
253
+ if (gate.warnings.length > 0) {
254
+ lines.push('', 'Router-reachability warnings (paraphrase gap — W5 judge is the fix, not blocking):');
255
+ for (const w of gate.warnings) lines.push(' - ' + w.skill + ': ' + w.detail);
256
+ }
257
+ lines.push('', gate.passed ? 'CI gate: PASS' : 'CI gate: FAIL (' + gate.failures.length + ')');
258
+ for (const f of gate.failures) lines.push(' x ' + f.skill + ': ' + f.kind + ' — ' + f.detail);
259
+ return lines.join('\n');
260
+ }
261
+
262
+ /** Render the command-coverage section (a summary line + any failures/warnings). */
263
+ function formatCoverage(coverage) {
264
+ const lines = [
265
+ 'Skill coverage (every registered command must own a skill or be exempt):',
266
+ ' commands=' + coverage.total + ' mapped=' + coverage.mapped + ' exempt=' + coverage.exempt +
267
+ ' gaps=' + coverage.failures.length,
268
+ ];
269
+ if (coverage.warnings.length > 0) {
270
+ lines.push('', 'Stale coverage.json entries (non-blocking — remove them):');
271
+ for (const w of coverage.warnings) lines.push(' - ' + w.command + ': ' + w.kind);
272
+ }
273
+ lines.push('', coverage.passed ? 'Coverage gate: PASS' : 'Coverage gate: FAIL (' + coverage.failures.length + ')');
274
+ for (const f of coverage.failures) lines.push(' x ' + f.command + ': ' + f.kind + ' — ' + f.detail);
275
+ return lines.join('\n');
276
+ }
277
+
278
+ /** Build the combined gate error string (static scorecard gate + command-coverage gate). */
279
+ function buildScoresError(gate, coverage) {
280
+ const parts = [];
281
+ if (!gate.passed) {
282
+ parts.push('static gate (' + gate.failures.length + '): ' + gate.failures.map(f => f.skill + ' — ' + f.kind).join('; '));
283
+ }
284
+ if (coverage && !coverage.passed) {
285
+ parts.push('coverage gate (' + coverage.failures.length + '): ' + coverage.failures.map(f => f.command + ' — ' + f.kind).join('; '));
286
+ }
287
+ return 'Skill CI gate FAILED — ' + parts.join(' | ');
288
+ }
289
+
290
+ /** "forge skill scores [--json]" -- the league table. Gate state is drift-aware so it agrees with CI. */
291
+ function handleScores(rest, flags, projectRoot) {
292
+ const json = flags.json === true || flags['--json'] === true || rest.includes('--json');
293
+ const ctx = skillEval.resolveSkillsContext(projectRoot);
294
+ if (!ctx) {
295
+ return { success: false, error: 'No canonical skills/ directory found (looked in the project and the packaged Forge root).' };
296
+ }
297
+ const { skillsDir, catalog, source } = ctx;
298
+ const scorecards = skillEval.buildAllScorecards(skillsDir, catalog);
299
+ // Compare the recomputed cards against the COMMITTED artifacts (canonical skills/ AND the
300
+ // .agents/skills mirror). A stale/missing committed scorecard is drift, so the gate reported here
301
+ // is FAIL — matching the CI drift test — instead of a hollow PASS over freshly-rebuilt cards.
302
+ // Gate the mirror by CONTEXT, not existence: a SOURCE checkout (source==='project') is EXPECTED
303
+ // to ship the committed .agents/skills mirror, so pass mirrorDir UNCONDITIONALLY — a deleted or
304
+ // never-checked-out mirror then REPORTS drift instead of silently passing. From the PACKAGED root
305
+ // (source==='package', a consumer install) no mirror ships, so omit the check to avoid false drift.
306
+ const mirrorDir = source === 'project' ? path.join(path.dirname(skillsDir), '.agents', 'skills') : null;
307
+ const drift = skillEval.detectScorecardDrift({ skillsDir, freshCards: scorecards, mirrorDir });
308
+ const gate = skillEval.evaluateGate(scorecards, { drift });
309
+ // Command-coverage gate (§3.3): a registered command with no owning skill (and not exempt) must
310
+ // FAIL scores too, so CI running `forge skill scores` catches a new unrouted command — not only
311
+ // the dedicated `forge skill coverage`.
312
+ const coverage = skillEval.buildCoverageReport(projectRoot);
313
+ // The gate verdict MUST drive the command's exit status: a failing gate (scorecard drift, caps
314
+ // violation, invalid fixtures, OR a coverage gap) returns success:false so the registry runner
315
+ // exits non-zero and a CI job running `forge skill scores` actually FAILS — instead of exiting 0
316
+ // while the output says "gate: FAIL". The full league table + gate detail still ride along.
317
+ const coveragePassed = !coverage || coverage.passed === true;
318
+ const passed = gate.passed === true && coveragePassed;
319
+ const gateError = passed ? undefined : buildScoresError(gate, coverage);
320
+ if (json) {
321
+ return { success: passed, error: gateError, scorecards, gate, coverage, drift, output: JSON.stringify({ scorecards, gate, coverage, drift }, null, 2) + '\n' };
322
+ }
323
+ const text = coverage ? formatScores(scorecards, gate) + '\n\n' + formatCoverage(coverage) : formatScores(scorecards, gate);
324
+ return { success: passed, error: gateError, scorecards, gate, coverage, drift, output: text };
325
+ }
326
+
327
+ /** "forge skill coverage [--json]" -- the dedicated command→skill coverage gate. */
328
+ function handleCoverage(rest, flags, projectRoot) {
329
+ const json = flags.json === true || flags['--json'] === true || rest.includes('--json');
330
+ const report = skillEval.buildCoverageReport(projectRoot);
331
+ if (!report) {
332
+ return { success: false, error: 'No canonical skills/ directory found (looked in the project and the packaged Forge root).' };
333
+ }
334
+ const passed = report.passed === true;
335
+ const gateError = passed
336
+ ? undefined
337
+ : 'Skill coverage gate FAILED (' + report.failures.length + '): ' +
338
+ report.failures.map(f => f.command + ' — ' + f.kind).join('; ');
339
+ if (json) {
340
+ return { success: passed, error: gateError, report, output: JSON.stringify(report, null, 2) + '\n' };
341
+ }
342
+ return { success: passed, error: gateError, report, output: formatCoverage(report) };
343
+ }
344
+
345
+ module.exports = {
346
+ name: 'skill',
347
+ description: 'Route, evaluate, score, and coverage-check Forge skills (forge skill for | eval | scores | coverage)',
348
+ usage: USAGE,
349
+ flags: {
350
+ '--json': 'Emit the machine-readable result',
351
+ '--static': 'Score only the deterministic static tier',
352
+ '--full': 'Run the controlled behavioral evaluation tier',
353
+ '--tier': `Behavioral corpus tier: ${BEHAVIORAL_TIER_USAGE}`,
354
+ },
355
+ // The registry passes (args, flags, projectRoot, opts). Only the trailing `opts` carries a
356
+ // default so SonarCloud S1788 stays satisfied; `flags` is normalized in the body. The router
357
+ // reads the canonical catalog from the package root; eval/scores read the canonical skills dir.
358
+ handler: (args, rawFlags, projectRoot, opts = {}) => {
359
+ const flags = rawFlags || {};
360
+ const verb = args[0];
361
+ if (verb === 'for') {
362
+ return handleFor(args.slice(1), flags);
363
+ }
364
+ if (verb === 'eval') {
365
+ return handleEval(args.slice(1), flags, projectRoot, opts);
366
+ }
367
+ if (verb === 'scores') {
368
+ return handleScores(args.slice(1), flags, projectRoot);
369
+ }
370
+ if (verb === 'coverage') {
371
+ return handleCoverage(args.slice(1), flags, projectRoot);
372
+ }
373
+ if (!verb) {
374
+ return { success: false, error: 'Missing verb.\n' + USAGE };
375
+ }
376
+ return {
377
+ success: false,
378
+ error: "Unknown verb '" + verb + "'. Supported: for, eval, scores, coverage.\n" + USAGE,
379
+ };
380
+ },
381
+ // Exposed for unit tests; not part of the CLI surface.
382
+ _internal: { parseForArgs, formatRouting, handleFor, handleEval, handleFullEval, handleScores, handleCoverage, formatScores, formatCoverage },
383
+ };