forge-workflow 0.1.0-beta.4 → 0.1.0-beta.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/AGENTS.md +14 -7
  2. package/CHANGELOG.md +43 -1
  3. package/README.md +6 -2
  4. package/bin/forge-cmd.js +20 -0
  5. package/bin/forge.js +16 -374
  6. package/docs/INDEX.md +1 -1
  7. package/docs/guides/BEADS_GITHUB_SYNC.md +2 -31
  8. package/docs/guides/MIGRATION.md +4 -4
  9. package/docs/guides/SETUP.md +16 -16
  10. package/docs/reference/COMMANDS.md +8 -5
  11. package/docs/reference/INSIGHTS_RECAP.md +9 -20
  12. package/docs/reference/RELEASE.md +5 -3
  13. package/docs/reference/TOOLCHAIN.md +8 -0
  14. package/docs/reference/protected-state-surfaces.md +4 -4
  15. package/docs/reference/shepherd.md +54 -25
  16. package/lefthook.yml +12 -0
  17. package/lib/activation/ensure-forge-home.js +33 -15
  18. package/lib/adapters/pr-state-adapter.js +344 -142
  19. package/lib/audit-evidence.js +71 -110
  20. package/lib/capped-jsonl-log.js +236 -0
  21. package/lib/commands/_registry.js +2 -2
  22. package/lib/commands/clean.js +196 -32
  23. package/lib/commands/dev.js +4 -33
  24. package/lib/commands/hooks.js +223 -25
  25. package/lib/commands/insights.js +8 -3
  26. package/lib/commands/merge.js +600 -40
  27. package/lib/commands/pr.js +1 -1
  28. package/lib/commands/preflight.js +11 -2
  29. package/lib/commands/prime.js +21 -8
  30. package/lib/commands/push.js +41 -51
  31. package/lib/commands/recall.js +60 -16
  32. package/lib/commands/recap.js +6 -1
  33. package/lib/commands/release.js +17 -2
  34. package/lib/commands/setup.js +191 -94
  35. package/lib/commands/shepherd.js +13 -1
  36. package/lib/commands/ship.js +22 -23
  37. package/lib/commands/skill.js +119 -11
  38. package/lib/commands/status.js +17 -1
  39. package/lib/commands/test.js +24 -34
  40. package/lib/commands/worktree.js +220 -42
  41. package/lib/core/runtime-graph.js +1 -1
  42. package/lib/doc-assertions.js +297 -0
  43. package/lib/existing-tdd-gate.js +253 -0
  44. package/lib/forge-context.js +1 -4
  45. package/lib/forge-issues.js +56 -32
  46. package/lib/git-defaults.js +56 -0
  47. package/lib/harness-capability-matrix.js +3 -3
  48. package/lib/hook-renderer.js +93 -4
  49. package/lib/insights.js +96 -80
  50. package/lib/kernel/backing-issue.js +14 -2
  51. package/lib/kernel/broker.js +16 -0
  52. package/lib/kernel/cli-broker-factory.js +12 -1
  53. package/lib/kernel/close-on-merge.js +154 -0
  54. package/lib/kernel/fs-class.js +42 -25
  55. package/lib/kernel/sqlite-driver.js +153 -29
  56. package/lib/lefthook-wiring.js +21 -1
  57. package/lib/memory/router.js +16 -1
  58. package/lib/memory-digest.js +47 -15
  59. package/lib/memory-recall-events.js +145 -0
  60. package/lib/memory-recall.js +71 -10
  61. package/lib/merge-rules.js +8 -4
  62. package/lib/npm-publish-workflow.js +272 -0
  63. package/lib/orientation.js +68 -43
  64. package/lib/plugin-catalog.js +14 -4
  65. package/lib/pr-bundle.js +5 -6
  66. package/lib/pr-monitor/journal.js +18 -2
  67. package/lib/pr-monitor/reconcile-executor.js +224 -41
  68. package/lib/pr-monitor/render-summary.js +196 -0
  69. package/lib/pr-monitor/shepherd-lease.js +10 -1
  70. package/lib/pr-monitor/watch-lifecycle.js +13 -1
  71. package/lib/pr-pull.js +33 -14
  72. package/lib/pr-shepherd.js +34 -8
  73. package/lib/preflight/gates.js +65 -18
  74. package/lib/preflight/runner.js +5 -0
  75. package/lib/project-memory.js +33 -1
  76. package/lib/protected-state-authority.js +305 -0
  77. package/lib/protected-state-surfaces.js +64 -44
  78. package/lib/release-readiness.js +51 -4
  79. package/lib/shell-utils.js +1 -1
  80. package/lib/skills-sync.js +6 -3
  81. package/lib/smart-merge.js +28 -4
  82. package/lib/symlink-utils.js +74 -26
  83. package/lib/upgrade-safety.js +39 -0
  84. package/lib/using-forge.js +19 -6
  85. package/package.json +6 -7
  86. package/scripts/doc-asserting-tests.js +158 -0
  87. package/scripts/lib/behavioral-eval-runner.js +310 -0
  88. package/scripts/lib/behavioral-eval-runtime.js +456 -0
  89. package/scripts/lib/eval-evidence.js +328 -0
  90. package/scripts/lib/eval-runner.js +81 -41
  91. package/scripts/lib/immutable-eval-corpus.js +309 -0
  92. package/scripts/lib/promotion-evidence-loader.js +94 -0
  93. package/scripts/lib/promotion-scorecard.js +314 -0
  94. package/scripts/npm-release-receipt.js +134 -0
  95. package/scripts/process-tree.js +761 -0
  96. package/scripts/protected-state-check.js +47 -22
  97. package/scripts/run-command-eval.js +29 -1
  98. package/scripts/sync-d20-audit.js +172 -0
  99. package/scripts/test-full-suite.js +249 -37
  100. package/scripts/test.js +176 -43
  101. package/skills/review/SKILL.md +4 -11
  102. package/skills/review/evals/scorecard.json +3 -3
  103. package/skills/rollback/SKILL.md +4 -11
  104. package/skills/rollback/evals/scorecard.json +3 -3
  105. package/skills/shepherd/SKILL.md +20 -14
  106. package/skills/shepherd/evals/scorecard.json +2 -2
  107. package/skills/ship/SKILL.md +4 -12
  108. package/skills/ship/evals/scorecard.json +3 -3
  109. package/skills/worktree/SKILL.md +6 -1
  110. package/skills/worktree/evals/scorecard.json +2 -2
  111. package/lib/beads-setup.js +0 -538
  112. package/lib/beads-sync-scaffold.js +0 -189
  113. package/lib/pat-setup.js +0 -207
  114. package/lib/pr-monitor/render-sticky.js +0 -206
  115. package/lib/pr-monitor/upsert-sticky.js +0 -169
  116. package/scripts/beads-context.sh +0 -577
  117. package/scripts/beads-migrate-to-dolt.sh +0 -7
  118. package/scripts/beads-upgrade-smoke.sh +0 -284
  119. package/scripts/lib/beads-migrate-to-dolt.mjs +0 -503
@@ -0,0 +1,310 @@
1
+ 'use strict';
2
+
3
+ const { loadTier, evaluateCase } = require('./immutable-eval-corpus');
4
+ const { createEvalEvidence, appendEvalEvidence } = require('./eval-evidence');
5
+
6
+ const RESULT_FIELDS = Object.freeze(['evidence', 'attribution']);
7
+ const ATTRIBUTION_FIELDS = Object.freeze([
8
+ 'model', 'effort', 'role', 'hashes', 'startedAt', 'endedAt', 'activeMs',
9
+ 'passiveMs', 'tokens', 'retries', 'compactions',
10
+ ]);
11
+ const ATTRIBUTION_HASH_FIELDS = Object.freeze(['prompt', 'skill', 'tool']);
12
+ const BINDING_FIELDS = Object.freeze(['repoSha', 'configHash', 'budgetHash']);
13
+ const ARM_FIELDS = Object.freeze(['id', 'model', 'config', 'budget']);
14
+ const STRUCTURAL_FAILURE_PREFIXES = Object.freeze([
15
+ 'evidence.', 'binding.', 'case_id.', 'packet.', 'split.', 'trial.', 'metrics.',
16
+ 'manifest.', 'observation.',
17
+ ]);
18
+ const SAFE_RUNTIME_FAILURES = new Set([
19
+ 'runtime.execution_failed', 'runtime.token_budget_exceeded', 'runtime.usage_unparseable',
20
+ ]);
21
+
22
+ function hasExactFields(value, fields) {
23
+ if (!value || typeof value !== 'object' || Array.isArray(value)) return false;
24
+ const keys = Object.keys(value);
25
+ return keys.length === fields.length && fields.every((field) => Object.hasOwn(value, field));
26
+ }
27
+
28
+ function validExecutorResult(result) {
29
+ return hasExactFields(result, RESULT_FIELDS) &&
30
+ hasExactFields(result.attribution, ATTRIBUTION_FIELDS) &&
31
+ hasExactFields(result.attribution.hashes, ATTRIBUTION_HASH_FIELDS);
32
+ }
33
+
34
+ function isStructuralFailure(failure) {
35
+ return STRUCTURAL_FAILURE_PREFIXES.some((prefix) => failure.startsWith(prefix));
36
+ }
37
+
38
+ function snapshotBinding(binding) {
39
+ if (!hasExactFields(binding, BINDING_FIELDS)) return null;
40
+ if (!/^[0-9a-f]{40}$/.test(binding.repoSha)) return null;
41
+ if (!/^[0-9a-f]{64}$/.test(binding.configHash)) return null;
42
+ if (!/^[0-9a-f]{64}$/.test(binding.budgetHash)) return null;
43
+ return Object.freeze({
44
+ repoSha: binding.repoSha,
45
+ configHash: binding.configHash,
46
+ budgetHash: binding.budgetHash,
47
+ });
48
+ }
49
+
50
+ function snapshotArms(arms) {
51
+ if (!Array.isArray(arms) || arms.length !== 4) return null;
52
+ const snapshots = [];
53
+ for (const arm of arms) {
54
+ if (!hasExactFields(arm, ARM_FIELDS)) return null;
55
+ if ([arm.id, arm.model, arm.config, arm.budget]
56
+ .some((value) => typeof value !== 'string' || value.length === 0)) return null;
57
+ if (!['current', 'bounded'].includes(arm.config)) return null;
58
+ snapshots.push(Object.freeze({ ...arm }));
59
+ }
60
+ if (new Set(snapshots.map((arm) => arm.id)).size !== 4) return null;
61
+ if (new Set(snapshots.map((arm) => arm.model)).size !== 2) return null;
62
+ const matrix = new Set(snapshots.map((arm) => `${arm.model}\0${arm.config}`));
63
+ if (matrix.size !== 4) return null;
64
+ return Object.freeze(snapshots);
65
+ }
66
+
67
+ function explicitHardFailure(evaluation, result) {
68
+ return result.evidence?.observation?.hardFailure === true
69
+ || evaluation.hardFailure === true;
70
+ }
71
+
72
+ function buildEnvelope(input, identity, binding, evaluation, result, hardFailure) {
73
+ const attribution = result.attribution;
74
+ return createEvalEvidence({
75
+ issue_id: input.issueId,
76
+ pr: input.pr,
77
+ head_sha: binding.repoSha,
78
+ model: attribution.model,
79
+ effort: attribution.effort,
80
+ role: attribution.role,
81
+ hashes: {
82
+ eval_set: result.evidence.packetHash,
83
+ prompt: attribution.hashes.prompt,
84
+ skill: attribution.hashes.skill,
85
+ tool: attribution.hashes.tool,
86
+ },
87
+ started_at: attribution.startedAt,
88
+ ended_at: attribution.endedAt,
89
+ active_ms: attribution.activeMs,
90
+ passive_ms: attribution.passiveMs,
91
+ tokens: {
92
+ input: attribution.tokens.input,
93
+ output: attribution.tokens.output,
94
+ cached: attribution.tokens.cached,
95
+ },
96
+ retries: attribution.retries,
97
+ compactions: attribution.compactions,
98
+ gates: [{ name: 'behavioral-case', passed: evaluation.passed }],
99
+ run_identity: {
100
+ arm_id: identity.armId,
101
+ case_id: identity.caseId,
102
+ risk: identity.risk,
103
+ split: identity.split,
104
+ model: identity.model,
105
+ config: identity.config,
106
+ budget: identity.budget,
107
+ tier: input.tier,
108
+ trial_index: identity.trialIndex,
109
+ config_hash: binding.configHash,
110
+ budget_hash: binding.budgetHash,
111
+ },
112
+ case_result: {
113
+ status: evaluation.passed ? 'PASS' : 'FAIL',
114
+ hard_failure: hardFailure,
115
+ latency_ms: result.evidence.metrics.durationMs,
116
+ tokens: result.evidence.metrics.tokensUsed,
117
+ },
118
+ });
119
+ }
120
+
121
+ function finding(identity, status, failures, evidence, hardFailure = false) {
122
+ return {
123
+ caseId: identity.caseId,
124
+ risk: identity.risk,
125
+ split: identity.split,
126
+ model: identity.model,
127
+ config: identity.config,
128
+ budget: identity.budget,
129
+ trialIndex: identity.trialIndex,
130
+ status,
131
+ hardFailure,
132
+ latencyMs: Number.isFinite(evidence?.metrics?.durationMs) ? evidence.metrics.durationMs : 0,
133
+ tokens: Number.isFinite(evidence?.metrics?.tokensUsed) ? evidence.metrics.tokensUsed : 0,
134
+ failures,
135
+ };
136
+ }
137
+
138
+ function incompleteResult(tier, arms, expectedRuns, reason) {
139
+ return {
140
+ status: 'INCOMPLETE',
141
+ tier,
142
+ arms,
143
+ expectedRuns,
144
+ completedRuns: 0,
145
+ passedRuns: 0,
146
+ failedRuns: 0,
147
+ incompleteRuns: expectedRuns,
148
+ findings: [{ status: 'INCOMPLETE', failures: [reason] }],
149
+ };
150
+ }
151
+
152
+ function identityFor(packet, trialIndex, arm) {
153
+ return {
154
+ armId: arm.id,
155
+ caseId: packet.caseId,
156
+ risk: packet.risk,
157
+ split: packet.split,
158
+ model: arm.model,
159
+ config: arm.config,
160
+ budget: arm.budget,
161
+ trialIndex,
162
+ };
163
+ }
164
+
165
+ async function invokeExecutor(input, packet, trialIndex, arm, binding) {
166
+ try {
167
+ return await input.executor(Object.freeze({
168
+ armId: arm.id,
169
+ model: arm.model,
170
+ config: arm.config,
171
+ budget: arm.budget,
172
+ packet,
173
+ trialIndex,
174
+ binding,
175
+ skillName: input.skillName,
176
+ }));
177
+ } catch (error) {
178
+ const failure = error?.message;
179
+ return SAFE_RUNTIME_FAILURES.has(failure) ? { runtimeFailure: failure } : null;
180
+ }
181
+ }
182
+
183
+ async function persistEvidence(input, append, identity, binding, evaluation, result, hardFailure) {
184
+ try {
185
+ const envelope = buildEnvelope(input, identity, binding, evaluation, result, hardFailure);
186
+ const appended = await append(input.projectRoot, envelope, input.appendOptions || {});
187
+ if (appended?.conflict) return 'evidence.conflict';
188
+ if (!appended || appended.ok !== true) return 'evidence.append_failed';
189
+ return null;
190
+ } catch {
191
+ return 'evidence.append_failed';
192
+ }
193
+ }
194
+
195
+ async function executeArm({ input, corpus, packet, trialIndex, arm, binding, append }) {
196
+ const identity = identityFor(packet, trialIndex, arm);
197
+ const result = await invokeExecutor(input, packet, trialIndex, arm, binding);
198
+ if (result?.runtimeFailure) {
199
+ return {
200
+ status: 'INCOMPLETE',
201
+ finding: finding(identity, 'INCOMPLETE', [result.runtimeFailure]),
202
+ };
203
+ }
204
+ if (!validExecutorResult(result)) {
205
+ return { status: 'INCOMPLETE', finding: finding(identity, 'INCOMPLETE', ['evidence.malformed']) };
206
+ }
207
+ if (result.attribution.model !== arm.model) {
208
+ return {
209
+ status: 'INCOMPLETE',
210
+ finding: finding(identity, 'INCOMPLETE', ['attribution.model_mismatch'], result.evidence),
211
+ };
212
+ }
213
+
214
+ const evaluation = evaluateCase({
215
+ packet,
216
+ allPackets: corpus.allPackets,
217
+ manifest: corpus.manifest,
218
+ evidence: result.evidence,
219
+ expectedBinding: binding,
220
+ });
221
+ const hardFailure = explicitHardFailure(evaluation, result);
222
+ if (evaluation.failures.some(isStructuralFailure)) {
223
+ return {
224
+ status: 'INCOMPLETE',
225
+ finding: finding(identity, 'INCOMPLETE', evaluation.failures, result.evidence, hardFailure),
226
+ };
227
+ }
228
+
229
+ const persistenceFailure = await persistEvidence(
230
+ input, append, identity, binding, evaluation, result, hardFailure,
231
+ );
232
+ if (persistenceFailure) {
233
+ return {
234
+ status: 'INCOMPLETE',
235
+ finding: finding(identity, 'INCOMPLETE', [persistenceFailure], result.evidence, hardFailure),
236
+ };
237
+ }
238
+ const status = evaluation.passed ? 'PASS' : 'FAIL';
239
+ return {
240
+ status,
241
+ finding: finding(
242
+ identity, status, evaluation.passed ? [] : evaluation.failures, result.evidence, hardFailure,
243
+ ),
244
+ };
245
+ }
246
+
247
+ function recordOutcome(counts, outcome, findings) {
248
+ findings.push(outcome.finding);
249
+ if (outcome.status === 'INCOMPLETE') {
250
+ counts.incompleteRuns += 1;
251
+ return;
252
+ }
253
+ counts.completedRuns += 1;
254
+ if (outcome.status === 'PASS') counts.passedRuns += 1;
255
+ else counts.failedRuns += 1;
256
+ }
257
+
258
+ function finalStatus(counts, expectedRuns) {
259
+ if (counts.incompleteRuns > 0 || counts.completedRuns !== expectedRuns) return 'INCOMPLETE';
260
+ return counts.failedRuns > 0 ? 'FAIL' : 'PASS';
261
+ }
262
+
263
+ /**
264
+ * Execute the frozen corpus through four opaque, matched executor arms.
265
+ * The executor may observe arm ids and immutable packets, but it receives no
266
+ * merge capability. Only a strict, privacy-safe attribution envelope persists.
267
+ */
268
+ async function runBehavioralEvaluation(input) {
269
+ let corpus;
270
+ try {
271
+ corpus = loadTier(input.tier);
272
+ } catch (error) {
273
+ return incompleteResult(input.tier, input.arms || [], 0, error.message);
274
+ }
275
+
276
+ const suppliedArms = Array.isArray(input.arms) ? input.arms : [];
277
+ const trialIndices = corpus.manifest.trialIndices;
278
+ const expectedRuns = corpus.cases.length * trialIndices.length * 4;
279
+ const arms = snapshotArms(suppliedArms);
280
+ if (!arms) return incompleteResult(input.tier, suppliedArms, expectedRuns, 'arms.invalid');
281
+ if (typeof input.executor !== 'function') {
282
+ return incompleteResult(input.tier, arms, expectedRuns, 'executor.missing');
283
+ }
284
+ const binding = snapshotBinding(input.binding);
285
+ if (!binding) return incompleteResult(input.tier, arms, expectedRuns, 'binding.invalid');
286
+
287
+ const append = input.appendEvidence || appendEvalEvidence;
288
+ const findings = [];
289
+ const counts = { completedRuns: 0, passedRuns: 0, failedRuns: 0, incompleteRuns: 0 };
290
+
291
+ for (const packet of corpus.cases) {
292
+ for (const trialIndex of trialIndices) {
293
+ for (const arm of arms) {
294
+ const outcome = await executeArm({ input, corpus, packet, trialIndex, arm, binding, append });
295
+ recordOutcome(counts, outcome, findings);
296
+ }
297
+ }
298
+ }
299
+
300
+ return {
301
+ status: finalStatus(counts, expectedRuns),
302
+ tier: input.tier,
303
+ arms,
304
+ expectedRuns,
305
+ ...counts,
306
+ findings,
307
+ };
308
+ }
309
+
310
+ module.exports = { runBehavioralEvaluation };